From 531ed18082be6aafe9325574849f68e6350d82a0 Mon Sep 17 00:00:00 2001 From: smill Date: Tue, 29 Sep 2026 23:56:54 -0400 Subject: [PATCH] compare-report: median-of-reps grouping, --exclude, --results --group NAME=t1,t2 collapses repetitions into one median column per device, --exclude drops bad runs, and --results reads a collected results directory. --- README.md | 11 ++++++ compare-report.py | 87 +++++++++++++++++++++++++++++++++++++++-------- 2 files changed, 84 insertions(+), 14 deletions(-) diff --git a/README.md b/README.md index 98b9cd7..dbf8043 100644 --- a/README.md +++ b/README.md @@ -383,6 +383,17 @@ are back-filled from the raw results (`fio` JSON, `openssl`, `7z`, `sysbench-mem`, `glmark2`/`vkmark`), so existing runs compare without re-running. Needs `matplotlib`; use `--no-graphs` for the table alone. +With **median-of-3**, collapse repetitions into one column per device and drop +any bad run: +```bash +python3 compare-report.py --results ~/Documents/results \ + --group "desktop=before1-headless,before2-headless,before3-headless" \ + --group "server=server1-headless,server2-headless,server3-headless" \ + --group "t14=t14-2,t14-3" # t14-1 dropped (low-battery run) +``` +`--group NAME=t1,t2,...` makes one median column; `--exclude t1 t2` drops tags. +`--results DIR` reads another machine's collected `results/`. + Fill the results table in `benchmarks.txt` with the medians after each side. --- diff --git a/compare-report.py b/compare-report.py index 679a59f..de9c26f 100644 --- a/compare-report.py +++ b/compare-report.py @@ -6,6 +6,13 @@ python3 compare-report.py --label t14-1="ThinkPad T14" --label t14-1=... python3 compare-report.py --no-graphs # table only +Repeats can be collapsed to a per-device median by grouping tags: + python3 compare-report.py --group desktop=before1,before2,before3 \ + --exclude t14-1 --group t14=t14-2,t14-3 +With --group, each group becomes one column holding the median of its members' +values; --exclude drops tags (e.g. an outlier run). Without --group every tag +is its own column. + Primary source is results/scores-.csv. For metrics that predate the score file (older runs), it falls back to the raw results: fio--.json openssl-aes/sha-.txt 7z-.txt @@ -20,6 +27,7 @@ import glob import json import os import re +import statistics HERE = os.path.dirname(os.path.abspath(__file__)) RES = os.path.join(HERE, "results") @@ -230,14 +238,31 @@ def bar_chart(key, title, unit, tags, values, color): return out +def median(vals): + vals = [v for v in vals if v is not None] + if not vals: + return None + return statistics.median(vals) + + def main(): + global RES, GRAPHS ap = argparse.ArgumentParser() + ap.add_argument("--results", default=RES, help="results directory (default ./results)") ap.add_argument("--tags", nargs="*", help="tags in display order (default: all found)") ap.add_argument("--label", action="append", default=[], metavar="TAG=NAME", help="friendly name for a tag (repeatable)") + ap.add_argument("--group", action="append", default=[], metavar="NAME=t1,t2,...", + help="collapse tags into one median column (repeatable); " + "grouped columns are shown instead of the raw tags") + ap.add_argument("--exclude", nargs="*", default=[], metavar="TAG", + help="tags to drop entirely (e.g. an outlier run)") ap.add_argument("--no-graphs", action="store_true") args = ap.parse_args() + RES = os.path.abspath(args.results) + GRAPHS = os.path.join(RES, "graphs") + labels = {} for spec in args.label: if "=" in spec: @@ -247,24 +272,51 @@ def main(): def label(t): return labels.get(t, t) - tags = args.tags or sorted(discover_tags()) - if not tags: + groups = [] # (name, [tags]) + for spec in args.group: + if "=" in spec: + name, members = spec.split("=", 1) + groups.append((name, [m for m in members.split(",") if m])) + + excluded = set(args.exclude) + if args.tags: + all_tags = [t for t in args.tags if t not in excluded] + else: + all_tags = [t for t in sorted(discover_tags()) if t not in excluded] + if not all_tags: raise SystemExit(f"no results found in {RES}") - scoremaps = {t: load_scores(t) for t in tags} + scoremaps = {t: load_scores(t) for t in all_tags} + + def value(tag, key, raw): + v = scoremaps.get(tag, {}).get(key) + if v is None and raw is not None: + v = raw(tag) + return v + + # Build column list: either grouped medians or the raw tags. + if groups: + need = set(all_tags) + for _, members in groups: + need.update(members) + for m in need: + if m not in scoremaps: + scoremaps[m] = load_scores(m) + columns = [(name, members) for name, members in groups] + col_names = [name for name, _ in columns] + else: + columns = [(label(t), [t]) for t in all_tags] + col_names = [label(t) for t in all_tags] palette = ("#c0392b", "#2980b9", "#27ae60", "#8e44ad", "#d35400", "#16a085", "#2c3e50", "#c2185b", "#7f8c8d", "#f39c12") - color = {t: palette[i % len(palette)] for i, t in enumerate(tags)} + color = {col_names[i]: palette[i % len(palette)] for i in range(len(col_names))} rows = [] for key, title, unit, higher, raw in METRICS: vals = [] - for t in tags: - v = scoremaps[t].get(key) - if v is None and raw is not None: - v = raw(t) - vals.append(v) + for _, members in columns: + vals.append(median([value(m, key, raw) for m in members])) if any(v is not None for v in vals): rows.append((key, title, unit, higher, vals)) @@ -272,21 +324,28 @@ def main(): raise SystemExit("no benchmark values found") name_w = max([len("benchmark")] + [len(r[1]) for r in rows]) - tag_w = [max(len(label(t)), 9) for t in tags] + col_w = [max(len(n), 9) for n in col_names] unit_w = max([len("unit")] + [len(r[2]) for r in rows]) header = f"{'benchmark':<{name_w}} {'unit':<{unit_w}} " + \ - " ".join(f"{label(t):>{tag_w[i]}}" for i, t in enumerate(tags)) + " ".join(f"{col_names[i]:>{col_w[i]}}" for i in range(len(col_names))) print(header) print("-" * len(header)) for key, title, unit, higher, vals in rows: - cells = " ".join(f"{fmt(v):>{tag_w[i]}}" for i, v in enumerate(vals)) + cells = " ".join(f"{fmt(v):>{col_w[i]}}" for i, v in enumerate(vals)) print(f"{title:<{name_w}} {unit:<{unit_w}} {cells}") + if groups: + print("\ngroups (median of runs):") + for name, members in columns: + print(f" {name:<12} <- {', '.join(members)}") + if excluded: + print(f"excluded: {', '.join(sorted(excluded))}") + csv_path = os.path.join(RES, "compare-scores.csv") with open(csv_path, "w", newline="") as fh: w = csv.writer(fh) - w.writerow(["benchmark", "unit"] + [label(t) for t in tags]) + w.writerow(["benchmark", "unit"] + col_names) for key, title, unit, higher, vals in rows: w.writerow([title, unit] + ["" if v is None else f"{v:g}" for v in vals]) print(f"\ntable: {csv_path}") @@ -295,7 +354,7 @@ def main(): return graphs = [] for key, title, unit, higher, vals in rows: - out = bar_chart(key, title, unit, tags, vals, color) + out = bar_chart(key, title, unit, col_names, vals, color) if out: graphs.append(out) if graphs: