basic-benchmark/compare-report.py

367 lines
13 KiB
Python
Raw Permalink Normal View History

#!/usr/bin/env python3
"""basic-benchmark: compare benchmark results across devices and before/after.
python3 compare-report.py # every tag found in results/
python3 compare-report.py --tags t14-1 server-1-headless
python3 compare-report.py --label t14-1="ThinkPad T14" --label t14-1=...
python3 compare-report.py --no-graphs # table only
Repeats can be collapsed to a per-device median by grouping tags:
python3 compare-report.py --group desktop=before1,before2,before3 \
--exclude t14-1 --group t14=t14-2,t14-3
With --group, each group becomes one column holding the median of its members'
values; --exclude drops tags (e.g. an outlier run). Without --group every tag
is its own column.
Primary source is results/scores-<tag>.csv. For metrics that predate the score
file (older runs), it falls back to the raw results:
fio-<job>-<tag>.json openssl-aes/sha-<tag>.txt 7z-<tag>.txt
sysbench-mem-<tag>.txt glmark2-<tag>.txt vkmark-<tag>.txt
Writes results/compare-scores.csv and one bar chart per benchmark to
results/graphs/compare-<metric>.png, then prints the table.
"""
import argparse
import csv
import glob
import json
import os
import re
import statistics
HERE = os.path.dirname(os.path.abspath(__file__))
RES = os.path.join(HERE, "results")
GRAPHS = os.path.join(RES, "graphs")
def fnum(v):
try:
return float(v)
except (TypeError, ValueError):
return None
def read_text(path):
try:
with open(path, encoding="utf-8", errors="replace") as fh:
return fh.read()
except OSError:
return None
def fmt(v):
if v is None:
return "-"
if abs(v) >= 10000:
return f"{v:,.0f}"
if abs(v) >= 100:
return f"{v:,.1f}"
if abs(v) >= 1:
return f"{v:.2f}"
return f"{v:.3f}"
# --- raw-file fallbacks -------------------------------------------------------
def raw_7zip(tag):
txt = read_text(os.path.join(RES, f"7z-{tag}.txt"))
if not txt:
return None
for line in txt.splitlines():
if line.startswith("Avr:"):
parts = line.split()
if len(parts) > 4:
return fnum(parts[4])
return None
def _openssl(tag, kind):
txt = read_text(os.path.join(RES, f"openssl-{kind}-{tag}.txt"))
if not txt:
return None
val = None
for line in txt.splitlines():
parts = line.split()
if parts and parts[-1].endswith("k"):
v = fnum(parts[-1][:-1])
if v is not None:
val = v / 1e6 # kbytes/s -> GB/s
return val
def raw_openssl_aes(tag):
return _openssl(tag, "aes")
def raw_openssl_sha(tag):
return _openssl(tag, "sha")
def raw_sysbench_mem(tag):
txt = read_text(os.path.join(RES, f"sysbench-mem-{tag}.txt"))
if not txt:
return None
m = re.search(r"\(([\d.]+) MiB/sec\)", txt)
return fnum(m.group(1)) if m else None
def _fio(tag, job, rw, field):
try:
with open(os.path.join(RES, f"fio-{job}-{tag}.json")) as fh:
data = json.load(fh)
return fnum(data["jobs"][0].get(rw, {}).get(field))
except (OSError, ValueError, KeyError, IndexError):
return None
def raw_fio_seqread(tag):
v = _fio(tag, "seqread", "read", "bw_bytes")
return v / 1e6 if v is not None else None
def raw_fio_seqwrite(tag):
v = _fio(tag, "seqwrite", "write", "bw_bytes")
return v / 1e6 if v is not None else None
def raw_fio_randread(tag):
return _fio(tag, "randread", "read", "iops")
def raw_fio_randwrite(tag):
return _fio(tag, "randwrite", "write", "iops")
def _gpu_score(tag, tool):
txt = read_text(os.path.join(RES, f"{tool}-{tag}.txt"))
if not txt:
return None
m = re.search(rf"{tool} Score:\s*([\d.]+)", txt)
return fnum(m.group(1)) if m else None
def raw_glmark2(tag):
return _gpu_score(tag, "glmark2")
def raw_vkmark(tag):
return _gpu_score(tag, "vkmark")
# key, title, unit, higher-is-better, raw fallback
METRICS = [
("PassMark-cpu", "PassMark CPU Mark", "mark", True, None),
("PassMark-cpu-single", "PassMark CPU Single", "mark", True, None),
("PassMark-mem", "PassMark Memory Mark", "mark", True, None),
("7zip", "7-Zip compression (all)", "MIPS", True, raw_7zip),
("7zip-1t", "7-Zip compression (1 thread)", "MIPS", True, None),
("openssl-aes", "OpenSSL AES-256-GCM", "GB/s", True, raw_openssl_aes),
("openssl-sha", "OpenSSL SHA-256", "GB/s", True, raw_openssl_sha),
("sysbench-1t", "sysbench CPU (1 thread)", "events/s", True, None),
("sysbench-nt", "sysbench CPU (all threads)", "events/s", True, None),
("sysbench-mem", "sysbench memory", "MiB/s", True, raw_sysbench_mem),
("llama-tg", "llama.cpp generation", "tok/s", True, None),
("libreoffice", "LibreOffice documents->PDF", "docs/s", True, None),
("gegl", "GEGL image operations", "ops/s", True, None),
("inkscape", "Inkscape SVG->PNG", "img/s", True, None),
("fio-seqread", "fio sequential read", "MB/s", True, raw_fio_seqread),
("fio-seqwrite", "fio sequential write", "MB/s", True, raw_fio_seqwrite),
("fio-randread", "fio random read 4k", "IOPS", True, raw_fio_randread),
("fio-randwrite", "fio random write 4k", "IOPS", True, raw_fio_randwrite),
("glmark2", "glmark2", "score", True, raw_glmark2),
("vkmark", "vkmark", "score", True, raw_vkmark),
]
RAW_TAG_GLOBS = [
("7z-*.txt", "7z-", ".txt"),
("openssl-aes-*.txt", "openssl-aes-", ".txt"),
("sysbench-mem-*.txt", "sysbench-mem-", ".txt"),
("fio-seqread-*.json", "fio-seqread-", ".json"),
("glmark2-*.txt", "glmark2-", ".txt"),
("vkmark-*.txt", "vkmark-", ".txt"),
]
def discover_tags():
tags = set()
for path in glob.glob(os.path.join(RES, "scores-*.csv")):
tags.add(os.path.basename(path)[len("scores-"):-len(".csv")])
for path in glob.glob(os.path.join(RES, "env-*.txt")):
tags.add(os.path.basename(path)[len("env-"):-len(".txt")])
for pattern, prefix, suffix in RAW_TAG_GLOBS:
for path in glob.glob(os.path.join(RES, pattern)):
base = os.path.basename(path)
if prefix == "7z-" and "-1t-" in base:
continue # 7z-1t-<tag>.txt is a different metric
tags.add(base[len(prefix):-len(suffix)])
return {t for t in tags if t and "merged" not in t}
def load_scores(tag):
out = {}
path = os.path.join(RES, f"scores-{tag}.csv")
if os.path.exists(path):
with open(path) as fh:
for line in fh:
parts = [p.strip() for p in line.strip().split(",")]
if len(parts) >= 2 and fnum(parts[1]) is not None:
out[parts[0]] = fnum(parts[1])
return out
def bar_chart(key, title, unit, tags, values, color):
try:
import matplotlib
matplotlib.use("Agg")
import matplotlib.pyplot as plt
except ImportError:
return None
present = [(t, v) for t, v in zip(tags, values) if v is not None]
names = [p[0] for p in present]
vals = [p[1] for p in present]
colors = [color[p[0]] for p in present]
fig, ax = plt.subplots(figsize=(max(6.0, 1.1 * len(names) + 2.0), 4.5))
bars = ax.bar(names, vals, color=colors)
for b, v in zip(bars, vals):
ax.annotate(fmt(v), (b.get_x() + b.get_width() / 2, b.get_height()),
ha="center", va="bottom", fontsize=8)
ax.set_title(f"{title} ({unit})")
ax.set_ylabel(unit)
ax.tick_params(axis="x", labelrotation=30)
ax.grid(axis="y", alpha=0.3)
for lbl in ax.get_xticklabels():
lbl.set_horizontalalignment("right")
fig.tight_layout()
os.makedirs(GRAPHS, exist_ok=True)
out = os.path.join(GRAPHS, f"compare-{key}.png")
fig.savefig(out, dpi=110)
plt.close(fig)
return out
def median(vals):
vals = [v for v in vals if v is not None]
if not vals:
return None
return statistics.median(vals)
def main():
global RES, GRAPHS
ap = argparse.ArgumentParser()
ap.add_argument("--results", default=RES, help="results directory (default ./results)")
ap.add_argument("--tags", nargs="*", help="tags in display order (default: all found)")
ap.add_argument("--label", action="append", default=[], metavar="TAG=NAME",
help="friendly name for a tag (repeatable)")
ap.add_argument("--group", action="append", default=[], metavar="NAME=t1,t2,...",
help="collapse tags into one median column (repeatable); "
"grouped columns are shown instead of the raw tags")
ap.add_argument("--exclude", nargs="*", default=[], metavar="TAG",
help="tags to drop entirely (e.g. an outlier run)")
ap.add_argument("--no-graphs", action="store_true")
args = ap.parse_args()
RES = os.path.abspath(args.results)
GRAPHS = os.path.join(RES, "graphs")
labels = {}
for spec in args.label:
if "=" in spec:
k, v = spec.split("=", 1)
labels[k] = v
def label(t):
return labels.get(t, t)
groups = [] # (name, [tags])
for spec in args.group:
if "=" in spec:
name, members = spec.split("=", 1)
groups.append((name, [m for m in members.split(",") if m]))
excluded = set(args.exclude)
if args.tags:
all_tags = [t for t in args.tags if t not in excluded]
else:
all_tags = [t for t in sorted(discover_tags()) if t not in excluded]
if not all_tags:
raise SystemExit(f"no results found in {RES}")
scoremaps = {t: load_scores(t) for t in all_tags}
def value(tag, key, raw):
v = scoremaps.get(tag, {}).get(key)
if v is None and raw is not None:
v = raw(tag)
return v
# Build column list: either grouped medians or the raw tags.
if groups:
need = set(all_tags)
for _, members in groups:
need.update(members)
for m in need:
if m not in scoremaps:
scoremaps[m] = load_scores(m)
columns = [(name, members) for name, members in groups]
col_names = [name for name, _ in columns]
else:
columns = [(label(t), [t]) for t in all_tags]
col_names = [label(t) for t in all_tags]
palette = ("#c0392b", "#2980b9", "#27ae60", "#8e44ad", "#d35400",
"#16a085", "#2c3e50", "#c2185b", "#7f8c8d", "#f39c12")
color = {col_names[i]: palette[i % len(palette)] for i in range(len(col_names))}
rows = []
for key, title, unit, higher, raw in METRICS:
vals = []
for _, members in columns:
vals.append(median([value(m, key, raw) for m in members]))
if any(v is not None for v in vals):
rows.append((key, title, unit, higher, vals))
if not rows:
raise SystemExit("no benchmark values found")
name_w = max([len("benchmark")] + [len(r[1]) for r in rows])
col_w = [max(len(n), 9) for n in col_names]
unit_w = max([len("unit")] + [len(r[2]) for r in rows])
header = f"{'benchmark':<{name_w}} {'unit':<{unit_w}} " + \
" ".join(f"{col_names[i]:>{col_w[i]}}" for i in range(len(col_names)))
print(header)
print("-" * len(header))
for key, title, unit, higher, vals in rows:
cells = " ".join(f"{fmt(v):>{col_w[i]}}" for i, v in enumerate(vals))
print(f"{title:<{name_w}} {unit:<{unit_w}} {cells}")
if groups:
print("\ngroups (median of runs):")
for name, members in columns:
print(f" {name:<12} <- {', '.join(members)}")
if excluded:
print(f"excluded: {', '.join(sorted(excluded))}")
csv_path = os.path.join(RES, "compare-scores.csv")
with open(csv_path, "w", newline="") as fh:
w = csv.writer(fh)
w.writerow(["benchmark", "unit"] + col_names)
for key, title, unit, higher, vals in rows:
w.writerow([title, unit] + ["" if v is None else f"{v:g}" for v in vals])
print(f"\ntable: {csv_path}")
if args.no_graphs:
return
graphs = []
for key, title, unit, higher, vals in rows:
out = bar_chart(key, title, unit, col_names, vals, color)
if out:
graphs.append(out)
if graphs:
print(f"graphs: {len(graphs)} -> {GRAPHS}/compare-*.png")
else:
print("graphs: skipped (matplotlib not installed)")
if __name__ == "__main__":
main()