2026-09-29 21:06:24 -04:00
|
|
|
#!/usr/bin/env python3
|
|
|
|
|
"""basic-benchmark: compare benchmark results across devices and before/after.
|
|
|
|
|
|
|
|
|
|
python3 compare-report.py # every tag found in results/
|
|
|
|
|
python3 compare-report.py --tags t14-1 server-1-headless
|
|
|
|
|
python3 compare-report.py --label t14-1="ThinkPad T14" --label t14-1=...
|
|
|
|
|
python3 compare-report.py --no-graphs # table only
|
|
|
|
|
|
2026-09-29 23:56:54 -04:00
|
|
|
Repeats can be collapsed to a per-device median by grouping tags:
|
|
|
|
|
python3 compare-report.py --group desktop=before1,before2,before3 \
|
|
|
|
|
--exclude t14-1 --group t14=t14-2,t14-3
|
|
|
|
|
With --group, each group becomes one column holding the median of its members'
|
|
|
|
|
values; --exclude drops tags (e.g. an outlier run). Without --group every tag
|
|
|
|
|
is its own column.
|
|
|
|
|
|
2026-09-29 21:06:24 -04:00
|
|
|
Primary source is results/scores-<tag>.csv. For metrics that predate the score
|
|
|
|
|
file (older runs), it falls back to the raw results:
|
|
|
|
|
fio-<job>-<tag>.json openssl-aes/sha-<tag>.txt 7z-<tag>.txt
|
|
|
|
|
sysbench-mem-<tag>.txt glmark2-<tag>.txt vkmark-<tag>.txt
|
|
|
|
|
|
|
|
|
|
Writes results/compare-scores.csv and one bar chart per benchmark to
|
|
|
|
|
results/graphs/compare-<metric>.png, then prints the table.
|
|
|
|
|
"""
|
|
|
|
|
import argparse
|
|
|
|
|
import csv
|
|
|
|
|
import glob
|
|
|
|
|
import json
|
|
|
|
|
import os
|
|
|
|
|
import re
|
2026-09-29 23:56:54 -04:00
|
|
|
import statistics
|
2026-09-29 21:06:24 -04:00
|
|
|
|
|
|
|
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
|
|
|
RES = os.path.join(HERE, "results")
|
|
|
|
|
GRAPHS = os.path.join(RES, "graphs")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def fnum(v):
|
|
|
|
|
try:
|
|
|
|
|
return float(v)
|
|
|
|
|
except (TypeError, ValueError):
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def read_text(path):
|
|
|
|
|
try:
|
|
|
|
|
with open(path, encoding="utf-8", errors="replace") as fh:
|
|
|
|
|
return fh.read()
|
|
|
|
|
except OSError:
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def fmt(v):
|
|
|
|
|
if v is None:
|
|
|
|
|
return "-"
|
|
|
|
|
if abs(v) >= 10000:
|
|
|
|
|
return f"{v:,.0f}"
|
|
|
|
|
if abs(v) >= 100:
|
|
|
|
|
return f"{v:,.1f}"
|
|
|
|
|
if abs(v) >= 1:
|
|
|
|
|
return f"{v:.2f}"
|
|
|
|
|
return f"{v:.3f}"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --- raw-file fallbacks -------------------------------------------------------
|
|
|
|
|
def raw_7zip(tag):
|
|
|
|
|
txt = read_text(os.path.join(RES, f"7z-{tag}.txt"))
|
|
|
|
|
if not txt:
|
|
|
|
|
return None
|
|
|
|
|
for line in txt.splitlines():
|
|
|
|
|
if line.startswith("Avr:"):
|
|
|
|
|
parts = line.split()
|
|
|
|
|
if len(parts) > 4:
|
|
|
|
|
return fnum(parts[4])
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _openssl(tag, kind):
|
|
|
|
|
txt = read_text(os.path.join(RES, f"openssl-{kind}-{tag}.txt"))
|
|
|
|
|
if not txt:
|
|
|
|
|
return None
|
|
|
|
|
val = None
|
|
|
|
|
for line in txt.splitlines():
|
|
|
|
|
parts = line.split()
|
|
|
|
|
if parts and parts[-1].endswith("k"):
|
|
|
|
|
v = fnum(parts[-1][:-1])
|
|
|
|
|
if v is not None:
|
|
|
|
|
val = v / 1e6 # kbytes/s -> GB/s
|
|
|
|
|
return val
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def raw_openssl_aes(tag):
|
|
|
|
|
return _openssl(tag, "aes")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def raw_openssl_sha(tag):
|
|
|
|
|
return _openssl(tag, "sha")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def raw_sysbench_mem(tag):
|
|
|
|
|
txt = read_text(os.path.join(RES, f"sysbench-mem-{tag}.txt"))
|
|
|
|
|
if not txt:
|
|
|
|
|
return None
|
|
|
|
|
m = re.search(r"\(([\d.]+) MiB/sec\)", txt)
|
|
|
|
|
return fnum(m.group(1)) if m else None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _fio(tag, job, rw, field):
|
|
|
|
|
try:
|
|
|
|
|
with open(os.path.join(RES, f"fio-{job}-{tag}.json")) as fh:
|
|
|
|
|
data = json.load(fh)
|
|
|
|
|
return fnum(data["jobs"][0].get(rw, {}).get(field))
|
|
|
|
|
except (OSError, ValueError, KeyError, IndexError):
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def raw_fio_seqread(tag):
|
|
|
|
|
v = _fio(tag, "seqread", "read", "bw_bytes")
|
|
|
|
|
return v / 1e6 if v is not None else None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def raw_fio_seqwrite(tag):
|
|
|
|
|
v = _fio(tag, "seqwrite", "write", "bw_bytes")
|
|
|
|
|
return v / 1e6 if v is not None else None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def raw_fio_randread(tag):
|
|
|
|
|
return _fio(tag, "randread", "read", "iops")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def raw_fio_randwrite(tag):
|
|
|
|
|
return _fio(tag, "randwrite", "write", "iops")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _gpu_score(tag, tool):
|
|
|
|
|
txt = read_text(os.path.join(RES, f"{tool}-{tag}.txt"))
|
|
|
|
|
if not txt:
|
|
|
|
|
return None
|
|
|
|
|
m = re.search(rf"{tool} Score:\s*([\d.]+)", txt)
|
|
|
|
|
return fnum(m.group(1)) if m else None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def raw_glmark2(tag):
|
|
|
|
|
return _gpu_score(tag, "glmark2")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def raw_vkmark(tag):
|
|
|
|
|
return _gpu_score(tag, "vkmark")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# key, title, unit, higher-is-better, raw fallback
|
|
|
|
|
METRICS = [
|
|
|
|
|
("PassMark-cpu", "PassMark CPU Mark", "mark", True, None),
|
|
|
|
|
("PassMark-cpu-single", "PassMark CPU Single", "mark", True, None),
|
|
|
|
|
("PassMark-mem", "PassMark Memory Mark", "mark", True, None),
|
|
|
|
|
("7zip", "7-Zip compression (all)", "MIPS", True, raw_7zip),
|
|
|
|
|
("7zip-1t", "7-Zip compression (1 thread)", "MIPS", True, None),
|
|
|
|
|
("openssl-aes", "OpenSSL AES-256-GCM", "GB/s", True, raw_openssl_aes),
|
|
|
|
|
("openssl-sha", "OpenSSL SHA-256", "GB/s", True, raw_openssl_sha),
|
|
|
|
|
("sysbench-1t", "sysbench CPU (1 thread)", "events/s", True, None),
|
|
|
|
|
("sysbench-nt", "sysbench CPU (all threads)", "events/s", True, None),
|
|
|
|
|
("sysbench-mem", "sysbench memory", "MiB/s", True, raw_sysbench_mem),
|
|
|
|
|
("llama-tg", "llama.cpp generation", "tok/s", True, None),
|
|
|
|
|
("libreoffice", "LibreOffice documents->PDF", "docs/s", True, None),
|
|
|
|
|
("gegl", "GEGL image operations", "ops/s", True, None),
|
|
|
|
|
("inkscape", "Inkscape SVG->PNG", "img/s", True, None),
|
|
|
|
|
("fio-seqread", "fio sequential read", "MB/s", True, raw_fio_seqread),
|
|
|
|
|
("fio-seqwrite", "fio sequential write", "MB/s", True, raw_fio_seqwrite),
|
|
|
|
|
("fio-randread", "fio random read 4k", "IOPS", True, raw_fio_randread),
|
|
|
|
|
("fio-randwrite", "fio random write 4k", "IOPS", True, raw_fio_randwrite),
|
|
|
|
|
("glmark2", "glmark2", "score", True, raw_glmark2),
|
|
|
|
|
("vkmark", "vkmark", "score", True, raw_vkmark),
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
RAW_TAG_GLOBS = [
|
|
|
|
|
("7z-*.txt", "7z-", ".txt"),
|
|
|
|
|
("openssl-aes-*.txt", "openssl-aes-", ".txt"),
|
|
|
|
|
("sysbench-mem-*.txt", "sysbench-mem-", ".txt"),
|
|
|
|
|
("fio-seqread-*.json", "fio-seqread-", ".json"),
|
|
|
|
|
("glmark2-*.txt", "glmark2-", ".txt"),
|
|
|
|
|
("vkmark-*.txt", "vkmark-", ".txt"),
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def discover_tags():
|
|
|
|
|
tags = set()
|
|
|
|
|
for path in glob.glob(os.path.join(RES, "scores-*.csv")):
|
|
|
|
|
tags.add(os.path.basename(path)[len("scores-"):-len(".csv")])
|
|
|
|
|
for path in glob.glob(os.path.join(RES, "env-*.txt")):
|
|
|
|
|
tags.add(os.path.basename(path)[len("env-"):-len(".txt")])
|
|
|
|
|
for pattern, prefix, suffix in RAW_TAG_GLOBS:
|
|
|
|
|
for path in glob.glob(os.path.join(RES, pattern)):
|
|
|
|
|
base = os.path.basename(path)
|
|
|
|
|
if prefix == "7z-" and "-1t-" in base:
|
|
|
|
|
continue # 7z-1t-<tag>.txt is a different metric
|
|
|
|
|
tags.add(base[len(prefix):-len(suffix)])
|
|
|
|
|
return {t for t in tags if t and "merged" not in t}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def load_scores(tag):
|
|
|
|
|
out = {}
|
|
|
|
|
path = os.path.join(RES, f"scores-{tag}.csv")
|
|
|
|
|
if os.path.exists(path):
|
|
|
|
|
with open(path) as fh:
|
|
|
|
|
for line in fh:
|
|
|
|
|
parts = [p.strip() for p in line.strip().split(",")]
|
|
|
|
|
if len(parts) >= 2 and fnum(parts[1]) is not None:
|
|
|
|
|
out[parts[0]] = fnum(parts[1])
|
|
|
|
|
return out
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def bar_chart(key, title, unit, tags, values, color):
|
|
|
|
|
try:
|
|
|
|
|
import matplotlib
|
|
|
|
|
matplotlib.use("Agg")
|
|
|
|
|
import matplotlib.pyplot as plt
|
|
|
|
|
except ImportError:
|
|
|
|
|
return None
|
|
|
|
|
present = [(t, v) for t, v in zip(tags, values) if v is not None]
|
|
|
|
|
names = [p[0] for p in present]
|
|
|
|
|
vals = [p[1] for p in present]
|
|
|
|
|
colors = [color[p[0]] for p in present]
|
|
|
|
|
fig, ax = plt.subplots(figsize=(max(6.0, 1.1 * len(names) + 2.0), 4.5))
|
|
|
|
|
bars = ax.bar(names, vals, color=colors)
|
|
|
|
|
for b, v in zip(bars, vals):
|
|
|
|
|
ax.annotate(fmt(v), (b.get_x() + b.get_width() / 2, b.get_height()),
|
|
|
|
|
ha="center", va="bottom", fontsize=8)
|
|
|
|
|
ax.set_title(f"{title} ({unit})")
|
|
|
|
|
ax.set_ylabel(unit)
|
|
|
|
|
ax.tick_params(axis="x", labelrotation=30)
|
|
|
|
|
ax.grid(axis="y", alpha=0.3)
|
|
|
|
|
for lbl in ax.get_xticklabels():
|
|
|
|
|
lbl.set_horizontalalignment("right")
|
|
|
|
|
fig.tight_layout()
|
|
|
|
|
os.makedirs(GRAPHS, exist_ok=True)
|
|
|
|
|
out = os.path.join(GRAPHS, f"compare-{key}.png")
|
|
|
|
|
fig.savefig(out, dpi=110)
|
|
|
|
|
plt.close(fig)
|
|
|
|
|
return out
|
|
|
|
|
|
|
|
|
|
|
2026-09-29 23:56:54 -04:00
|
|
|
def median(vals):
|
|
|
|
|
vals = [v for v in vals if v is not None]
|
|
|
|
|
if not vals:
|
|
|
|
|
return None
|
|
|
|
|
return statistics.median(vals)
|
|
|
|
|
|
|
|
|
|
|
2026-09-29 21:06:24 -04:00
|
|
|
def main():
|
2026-09-29 23:56:54 -04:00
|
|
|
global RES, GRAPHS
|
2026-09-29 21:06:24 -04:00
|
|
|
ap = argparse.ArgumentParser()
|
2026-09-29 23:56:54 -04:00
|
|
|
ap.add_argument("--results", default=RES, help="results directory (default ./results)")
|
2026-09-29 21:06:24 -04:00
|
|
|
ap.add_argument("--tags", nargs="*", help="tags in display order (default: all found)")
|
|
|
|
|
ap.add_argument("--label", action="append", default=[], metavar="TAG=NAME",
|
|
|
|
|
help="friendly name for a tag (repeatable)")
|
2026-09-29 23:56:54 -04:00
|
|
|
ap.add_argument("--group", action="append", default=[], metavar="NAME=t1,t2,...",
|
|
|
|
|
help="collapse tags into one median column (repeatable); "
|
|
|
|
|
"grouped columns are shown instead of the raw tags")
|
|
|
|
|
ap.add_argument("--exclude", nargs="*", default=[], metavar="TAG",
|
|
|
|
|
help="tags to drop entirely (e.g. an outlier run)")
|
2026-09-29 21:06:24 -04:00
|
|
|
ap.add_argument("--no-graphs", action="store_true")
|
|
|
|
|
args = ap.parse_args()
|
|
|
|
|
|
2026-09-29 23:56:54 -04:00
|
|
|
RES = os.path.abspath(args.results)
|
|
|
|
|
GRAPHS = os.path.join(RES, "graphs")
|
|
|
|
|
|
2026-09-29 21:06:24 -04:00
|
|
|
labels = {}
|
|
|
|
|
for spec in args.label:
|
|
|
|
|
if "=" in spec:
|
|
|
|
|
k, v = spec.split("=", 1)
|
|
|
|
|
labels[k] = v
|
|
|
|
|
|
|
|
|
|
def label(t):
|
|
|
|
|
return labels.get(t, t)
|
|
|
|
|
|
2026-09-29 23:56:54 -04:00
|
|
|
groups = [] # (name, [tags])
|
|
|
|
|
for spec in args.group:
|
|
|
|
|
if "=" in spec:
|
|
|
|
|
name, members = spec.split("=", 1)
|
|
|
|
|
groups.append((name, [m for m in members.split(",") if m]))
|
|
|
|
|
|
|
|
|
|
excluded = set(args.exclude)
|
|
|
|
|
if args.tags:
|
|
|
|
|
all_tags = [t for t in args.tags if t not in excluded]
|
|
|
|
|
else:
|
|
|
|
|
all_tags = [t for t in sorted(discover_tags()) if t not in excluded]
|
|
|
|
|
if not all_tags:
|
2026-09-29 21:06:24 -04:00
|
|
|
raise SystemExit(f"no results found in {RES}")
|
|
|
|
|
|
2026-09-29 23:56:54 -04:00
|
|
|
scoremaps = {t: load_scores(t) for t in all_tags}
|
|
|
|
|
|
|
|
|
|
def value(tag, key, raw):
|
|
|
|
|
v = scoremaps.get(tag, {}).get(key)
|
|
|
|
|
if v is None and raw is not None:
|
|
|
|
|
v = raw(tag)
|
|
|
|
|
return v
|
|
|
|
|
|
|
|
|
|
# Build column list: either grouped medians or the raw tags.
|
|
|
|
|
if groups:
|
|
|
|
|
need = set(all_tags)
|
|
|
|
|
for _, members in groups:
|
|
|
|
|
need.update(members)
|
|
|
|
|
for m in need:
|
|
|
|
|
if m not in scoremaps:
|
|
|
|
|
scoremaps[m] = load_scores(m)
|
|
|
|
|
columns = [(name, members) for name, members in groups]
|
|
|
|
|
col_names = [name for name, _ in columns]
|
|
|
|
|
else:
|
|
|
|
|
columns = [(label(t), [t]) for t in all_tags]
|
|
|
|
|
col_names = [label(t) for t in all_tags]
|
2026-09-29 21:06:24 -04:00
|
|
|
|
|
|
|
|
palette = ("#c0392b", "#2980b9", "#27ae60", "#8e44ad", "#d35400",
|
|
|
|
|
"#16a085", "#2c3e50", "#c2185b", "#7f8c8d", "#f39c12")
|
2026-09-29 23:56:54 -04:00
|
|
|
color = {col_names[i]: palette[i % len(palette)] for i in range(len(col_names))}
|
2026-09-29 21:06:24 -04:00
|
|
|
|
|
|
|
|
rows = []
|
|
|
|
|
for key, title, unit, higher, raw in METRICS:
|
|
|
|
|
vals = []
|
2026-09-29 23:56:54 -04:00
|
|
|
for _, members in columns:
|
|
|
|
|
vals.append(median([value(m, key, raw) for m in members]))
|
2026-09-29 21:06:24 -04:00
|
|
|
if any(v is not None for v in vals):
|
|
|
|
|
rows.append((key, title, unit, higher, vals))
|
|
|
|
|
|
|
|
|
|
if not rows:
|
|
|
|
|
raise SystemExit("no benchmark values found")
|
|
|
|
|
|
|
|
|
|
name_w = max([len("benchmark")] + [len(r[1]) for r in rows])
|
2026-09-29 23:56:54 -04:00
|
|
|
col_w = [max(len(n), 9) for n in col_names]
|
2026-09-29 21:06:24 -04:00
|
|
|
unit_w = max([len("unit")] + [len(r[2]) for r in rows])
|
|
|
|
|
|
|
|
|
|
header = f"{'benchmark':<{name_w}} {'unit':<{unit_w}} " + \
|
2026-09-29 23:56:54 -04:00
|
|
|
" ".join(f"{col_names[i]:>{col_w[i]}}" for i in range(len(col_names)))
|
2026-09-29 21:06:24 -04:00
|
|
|
print(header)
|
|
|
|
|
print("-" * len(header))
|
|
|
|
|
for key, title, unit, higher, vals in rows:
|
2026-09-29 23:56:54 -04:00
|
|
|
cells = " ".join(f"{fmt(v):>{col_w[i]}}" for i, v in enumerate(vals))
|
2026-09-29 21:06:24 -04:00
|
|
|
print(f"{title:<{name_w}} {unit:<{unit_w}} {cells}")
|
|
|
|
|
|
2026-09-29 23:56:54 -04:00
|
|
|
if groups:
|
|
|
|
|
print("\ngroups (median of runs):")
|
|
|
|
|
for name, members in columns:
|
|
|
|
|
print(f" {name:<12} <- {', '.join(members)}")
|
|
|
|
|
if excluded:
|
|
|
|
|
print(f"excluded: {', '.join(sorted(excluded))}")
|
|
|
|
|
|
2026-09-29 21:06:24 -04:00
|
|
|
csv_path = os.path.join(RES, "compare-scores.csv")
|
|
|
|
|
with open(csv_path, "w", newline="") as fh:
|
|
|
|
|
w = csv.writer(fh)
|
2026-09-29 23:56:54 -04:00
|
|
|
w.writerow(["benchmark", "unit"] + col_names)
|
2026-09-29 21:06:24 -04:00
|
|
|
for key, title, unit, higher, vals in rows:
|
|
|
|
|
w.writerow([title, unit] + ["" if v is None else f"{v:g}" for v in vals])
|
|
|
|
|
print(f"\ntable: {csv_path}")
|
|
|
|
|
|
|
|
|
|
if args.no_graphs:
|
|
|
|
|
return
|
|
|
|
|
graphs = []
|
|
|
|
|
for key, title, unit, higher, vals in rows:
|
2026-09-29 23:56:54 -04:00
|
|
|
out = bar_chart(key, title, unit, col_names, vals, color)
|
2026-09-29 21:06:24 -04:00
|
|
|
if out:
|
|
|
|
|
graphs.append(out)
|
|
|
|
|
if graphs:
|
|
|
|
|
print(f"graphs: {len(graphs)} -> {GRAPHS}/compare-*.png")
|
|
|
|
|
else:
|
|
|
|
|
print("graphs: skipped (matplotlib not installed)")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if __name__ == "__main__":
|
|
|
|
|
main()
|