compare-report: median-of-reps grouping, --exclude, --results

--group NAME=t1,t2 collapses repetitions into one median column per device,
--exclude drops bad runs, and --results reads a collected results directory.
This commit is contained in:
smill 2026-09-29 23:56:54 -04:00
commit 531ed18082
2 changed files with 84 additions and 14 deletions

View file

@ -383,6 +383,17 @@ are back-filled from the raw results (`fio` JSON, `openssl`, `7z`,
`sysbench-mem`, `glmark2`/`vkmark`), so existing runs compare without re-running.
Needs `matplotlib`; use `--no-graphs` for the table alone.
With **median-of-3**, collapse repetitions into one column per device and drop
any bad run:
```bash
python3 compare-report.py --results ~/Documents/results \
--group "desktop=before1-headless,before2-headless,before3-headless" \
--group "server=server1-headless,server2-headless,server3-headless" \
--group "t14=t14-2,t14-3" # t14-1 dropped (low-battery run)
```
`--group NAME=t1,t2,...` makes one median column; `--exclude t1 t2` drops tags.
`--results DIR` reads another machine's collected `results/`.
Fill the results table in `benchmarks.txt` with the medians after each side.
---

View file

@ -6,6 +6,13 @@
python3 compare-report.py --label t14-1="ThinkPad T14" --label t14-1=...
python3 compare-report.py --no-graphs # table only
Repeats can be collapsed to a per-device median by grouping tags:
python3 compare-report.py --group desktop=before1,before2,before3 \
--exclude t14-1 --group t14=t14-2,t14-3
With --group, each group becomes one column holding the median of its members'
values; --exclude drops tags (e.g. an outlier run). Without --group every tag
is its own column.
Primary source is results/scores-<tag>.csv. For metrics that predate the score
file (older runs), it falls back to the raw results:
fio-<job>-<tag>.json openssl-aes/sha-<tag>.txt 7z-<tag>.txt
@ -20,6 +27,7 @@ import glob
import json
import os
import re
import statistics
HERE = os.path.dirname(os.path.abspath(__file__))
RES = os.path.join(HERE, "results")
@ -230,14 +238,31 @@ def bar_chart(key, title, unit, tags, values, color):
return out
def median(vals):
vals = [v for v in vals if v is not None]
if not vals:
return None
return statistics.median(vals)
def main():
global RES, GRAPHS
ap = argparse.ArgumentParser()
ap.add_argument("--results", default=RES, help="results directory (default ./results)")
ap.add_argument("--tags", nargs="*", help="tags in display order (default: all found)")
ap.add_argument("--label", action="append", default=[], metavar="TAG=NAME",
help="friendly name for a tag (repeatable)")
ap.add_argument("--group", action="append", default=[], metavar="NAME=t1,t2,...",
help="collapse tags into one median column (repeatable); "
"grouped columns are shown instead of the raw tags")
ap.add_argument("--exclude", nargs="*", default=[], metavar="TAG",
help="tags to drop entirely (e.g. an outlier run)")
ap.add_argument("--no-graphs", action="store_true")
args = ap.parse_args()
RES = os.path.abspath(args.results)
GRAPHS = os.path.join(RES, "graphs")
labels = {}
for spec in args.label:
if "=" in spec:
@ -247,24 +272,51 @@ def main():
def label(t):
return labels.get(t, t)
tags = args.tags or sorted(discover_tags())
if not tags:
groups = [] # (name, [tags])
for spec in args.group:
if "=" in spec:
name, members = spec.split("=", 1)
groups.append((name, [m for m in members.split(",") if m]))
excluded = set(args.exclude)
if args.tags:
all_tags = [t for t in args.tags if t not in excluded]
else:
all_tags = [t for t in sorted(discover_tags()) if t not in excluded]
if not all_tags:
raise SystemExit(f"no results found in {RES}")
scoremaps = {t: load_scores(t) for t in tags}
scoremaps = {t: load_scores(t) for t in all_tags}
def value(tag, key, raw):
v = scoremaps.get(tag, {}).get(key)
if v is None and raw is not None:
v = raw(tag)
return v
# Build column list: either grouped medians or the raw tags.
if groups:
need = set(all_tags)
for _, members in groups:
need.update(members)
for m in need:
if m not in scoremaps:
scoremaps[m] = load_scores(m)
columns = [(name, members) for name, members in groups]
col_names = [name for name, _ in columns]
else:
columns = [(label(t), [t]) for t in all_tags]
col_names = [label(t) for t in all_tags]
palette = ("#c0392b", "#2980b9", "#27ae60", "#8e44ad", "#d35400",
"#16a085", "#2c3e50", "#c2185b", "#7f8c8d", "#f39c12")
color = {t: palette[i % len(palette)] for i, t in enumerate(tags)}
color = {col_names[i]: palette[i % len(palette)] for i in range(len(col_names))}
rows = []
for key, title, unit, higher, raw in METRICS:
vals = []
for t in tags:
v = scoremaps[t].get(key)
if v is None and raw is not None:
v = raw(t)
vals.append(v)
for _, members in columns:
vals.append(median([value(m, key, raw) for m in members]))
if any(v is not None for v in vals):
rows.append((key, title, unit, higher, vals))
@ -272,21 +324,28 @@ def main():
raise SystemExit("no benchmark values found")
name_w = max([len("benchmark")] + [len(r[1]) for r in rows])
tag_w = [max(len(label(t)), 9) for t in tags]
col_w = [max(len(n), 9) for n in col_names]
unit_w = max([len("unit")] + [len(r[2]) for r in rows])
header = f"{'benchmark':<{name_w}} {'unit':<{unit_w}} " + \
" ".join(f"{label(t):>{tag_w[i]}}" for i, t in enumerate(tags))
" ".join(f"{col_names[i]:>{col_w[i]}}" for i in range(len(col_names)))
print(header)
print("-" * len(header))
for key, title, unit, higher, vals in rows:
cells = " ".join(f"{fmt(v):>{tag_w[i]}}" for i, v in enumerate(vals))
cells = " ".join(f"{fmt(v):>{col_w[i]}}" for i, v in enumerate(vals))
print(f"{title:<{name_w}} {unit:<{unit_w}} {cells}")
if groups:
print("\ngroups (median of runs):")
for name, members in columns:
print(f" {name:<12} <- {', '.join(members)}")
if excluded:
print(f"excluded: {', '.join(sorted(excluded))}")
csv_path = os.path.join(RES, "compare-scores.csv")
with open(csv_path, "w", newline="") as fh:
w = csv.writer(fh)
w.writerow(["benchmark", "unit"] + [label(t) for t in tags])
w.writerow(["benchmark", "unit"] + col_names)
for key, title, unit, higher, vals in rows:
w.writerow([title, unit] + ["" if v is None else f"{v:g}" for v in vals])
print(f"\ntable: {csv_path}")
@ -295,7 +354,7 @@ def main():
return
graphs = []
for key, title, unit, higher, vals in rows:
out = bar_chart(key, title, unit, tags, vals, color)
out = bar_chart(key, title, unit, col_names, vals, color)
if out:
graphs.append(out)
if graphs: