compare-report: median-of-reps grouping, --exclude, --results
--group NAME=t1,t2 collapses repetitions into one median column per device, --exclude drops bad runs, and --results reads a collected results directory.
This commit is contained in:
parent
f9c1e2d40e
commit
531ed18082
2 changed files with 84 additions and 14 deletions
11
README.md
11
README.md
|
|
@ -383,6 +383,17 @@ are back-filled from the raw results (`fio` JSON, `openssl`, `7z`,
|
||||||
`sysbench-mem`, `glmark2`/`vkmark`), so existing runs compare without re-running.
|
`sysbench-mem`, `glmark2`/`vkmark`), so existing runs compare without re-running.
|
||||||
Needs `matplotlib`; use `--no-graphs` for the table alone.
|
Needs `matplotlib`; use `--no-graphs` for the table alone.
|
||||||
|
|
||||||
|
With **median-of-3**, collapse repetitions into one column per device and drop
|
||||||
|
any bad run:
|
||||||
|
```bash
|
||||||
|
python3 compare-report.py --results ~/Documents/results \
|
||||||
|
--group "desktop=before1-headless,before2-headless,before3-headless" \
|
||||||
|
--group "server=server1-headless,server2-headless,server3-headless" \
|
||||||
|
--group "t14=t14-2,t14-3" # t14-1 dropped (low-battery run)
|
||||||
|
```
|
||||||
|
`--group NAME=t1,t2,...` makes one median column; `--exclude t1 t2` drops tags.
|
||||||
|
`--results DIR` reads another machine's collected `results/`.
|
||||||
|
|
||||||
Fill the results table in `benchmarks.txt` with the medians after each side.
|
Fill the results table in `benchmarks.txt` with the medians after each side.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
|
||||||
|
|
@ -6,6 +6,13 @@
|
||||||
python3 compare-report.py --label t14-1="ThinkPad T14" --label t14-1=...
|
python3 compare-report.py --label t14-1="ThinkPad T14" --label t14-1=...
|
||||||
python3 compare-report.py --no-graphs # table only
|
python3 compare-report.py --no-graphs # table only
|
||||||
|
|
||||||
|
Repeats can be collapsed to a per-device median by grouping tags:
|
||||||
|
python3 compare-report.py --group desktop=before1,before2,before3 \
|
||||||
|
--exclude t14-1 --group t14=t14-2,t14-3
|
||||||
|
With --group, each group becomes one column holding the median of its members'
|
||||||
|
values; --exclude drops tags (e.g. an outlier run). Without --group every tag
|
||||||
|
is its own column.
|
||||||
|
|
||||||
Primary source is results/scores-<tag>.csv. For metrics that predate the score
|
Primary source is results/scores-<tag>.csv. For metrics that predate the score
|
||||||
file (older runs), it falls back to the raw results:
|
file (older runs), it falls back to the raw results:
|
||||||
fio-<job>-<tag>.json openssl-aes/sha-<tag>.txt 7z-<tag>.txt
|
fio-<job>-<tag>.json openssl-aes/sha-<tag>.txt 7z-<tag>.txt
|
||||||
|
|
@ -20,6 +27,7 @@ import glob
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
|
import statistics
|
||||||
|
|
||||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||||
RES = os.path.join(HERE, "results")
|
RES = os.path.join(HERE, "results")
|
||||||
|
|
@ -230,14 +238,31 @@ def bar_chart(key, title, unit, tags, values, color):
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def median(vals):
|
||||||
|
vals = [v for v in vals if v is not None]
|
||||||
|
if not vals:
|
||||||
|
return None
|
||||||
|
return statistics.median(vals)
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
|
global RES, GRAPHS
|
||||||
ap = argparse.ArgumentParser()
|
ap = argparse.ArgumentParser()
|
||||||
|
ap.add_argument("--results", default=RES, help="results directory (default ./results)")
|
||||||
ap.add_argument("--tags", nargs="*", help="tags in display order (default: all found)")
|
ap.add_argument("--tags", nargs="*", help="tags in display order (default: all found)")
|
||||||
ap.add_argument("--label", action="append", default=[], metavar="TAG=NAME",
|
ap.add_argument("--label", action="append", default=[], metavar="TAG=NAME",
|
||||||
help="friendly name for a tag (repeatable)")
|
help="friendly name for a tag (repeatable)")
|
||||||
|
ap.add_argument("--group", action="append", default=[], metavar="NAME=t1,t2,...",
|
||||||
|
help="collapse tags into one median column (repeatable); "
|
||||||
|
"grouped columns are shown instead of the raw tags")
|
||||||
|
ap.add_argument("--exclude", nargs="*", default=[], metavar="TAG",
|
||||||
|
help="tags to drop entirely (e.g. an outlier run)")
|
||||||
ap.add_argument("--no-graphs", action="store_true")
|
ap.add_argument("--no-graphs", action="store_true")
|
||||||
args = ap.parse_args()
|
args = ap.parse_args()
|
||||||
|
|
||||||
|
RES = os.path.abspath(args.results)
|
||||||
|
GRAPHS = os.path.join(RES, "graphs")
|
||||||
|
|
||||||
labels = {}
|
labels = {}
|
||||||
for spec in args.label:
|
for spec in args.label:
|
||||||
if "=" in spec:
|
if "=" in spec:
|
||||||
|
|
@ -247,24 +272,51 @@ def main():
|
||||||
def label(t):
|
def label(t):
|
||||||
return labels.get(t, t)
|
return labels.get(t, t)
|
||||||
|
|
||||||
tags = args.tags or sorted(discover_tags())
|
groups = [] # (name, [tags])
|
||||||
if not tags:
|
for spec in args.group:
|
||||||
|
if "=" in spec:
|
||||||
|
name, members = spec.split("=", 1)
|
||||||
|
groups.append((name, [m for m in members.split(",") if m]))
|
||||||
|
|
||||||
|
excluded = set(args.exclude)
|
||||||
|
if args.tags:
|
||||||
|
all_tags = [t for t in args.tags if t not in excluded]
|
||||||
|
else:
|
||||||
|
all_tags = [t for t in sorted(discover_tags()) if t not in excluded]
|
||||||
|
if not all_tags:
|
||||||
raise SystemExit(f"no results found in {RES}")
|
raise SystemExit(f"no results found in {RES}")
|
||||||
|
|
||||||
scoremaps = {t: load_scores(t) for t in tags}
|
scoremaps = {t: load_scores(t) for t in all_tags}
|
||||||
|
|
||||||
|
def value(tag, key, raw):
|
||||||
|
v = scoremaps.get(tag, {}).get(key)
|
||||||
|
if v is None and raw is not None:
|
||||||
|
v = raw(tag)
|
||||||
|
return v
|
||||||
|
|
||||||
|
# Build column list: either grouped medians or the raw tags.
|
||||||
|
if groups:
|
||||||
|
need = set(all_tags)
|
||||||
|
for _, members in groups:
|
||||||
|
need.update(members)
|
||||||
|
for m in need:
|
||||||
|
if m not in scoremaps:
|
||||||
|
scoremaps[m] = load_scores(m)
|
||||||
|
columns = [(name, members) for name, members in groups]
|
||||||
|
col_names = [name for name, _ in columns]
|
||||||
|
else:
|
||||||
|
columns = [(label(t), [t]) for t in all_tags]
|
||||||
|
col_names = [label(t) for t in all_tags]
|
||||||
|
|
||||||
palette = ("#c0392b", "#2980b9", "#27ae60", "#8e44ad", "#d35400",
|
palette = ("#c0392b", "#2980b9", "#27ae60", "#8e44ad", "#d35400",
|
||||||
"#16a085", "#2c3e50", "#c2185b", "#7f8c8d", "#f39c12")
|
"#16a085", "#2c3e50", "#c2185b", "#7f8c8d", "#f39c12")
|
||||||
color = {t: palette[i % len(palette)] for i, t in enumerate(tags)}
|
color = {col_names[i]: palette[i % len(palette)] for i in range(len(col_names))}
|
||||||
|
|
||||||
rows = []
|
rows = []
|
||||||
for key, title, unit, higher, raw in METRICS:
|
for key, title, unit, higher, raw in METRICS:
|
||||||
vals = []
|
vals = []
|
||||||
for t in tags:
|
for _, members in columns:
|
||||||
v = scoremaps[t].get(key)
|
vals.append(median([value(m, key, raw) for m in members]))
|
||||||
if v is None and raw is not None:
|
|
||||||
v = raw(t)
|
|
||||||
vals.append(v)
|
|
||||||
if any(v is not None for v in vals):
|
if any(v is not None for v in vals):
|
||||||
rows.append((key, title, unit, higher, vals))
|
rows.append((key, title, unit, higher, vals))
|
||||||
|
|
||||||
|
|
@ -272,21 +324,28 @@ def main():
|
||||||
raise SystemExit("no benchmark values found")
|
raise SystemExit("no benchmark values found")
|
||||||
|
|
||||||
name_w = max([len("benchmark")] + [len(r[1]) for r in rows])
|
name_w = max([len("benchmark")] + [len(r[1]) for r in rows])
|
||||||
tag_w = [max(len(label(t)), 9) for t in tags]
|
col_w = [max(len(n), 9) for n in col_names]
|
||||||
unit_w = max([len("unit")] + [len(r[2]) for r in rows])
|
unit_w = max([len("unit")] + [len(r[2]) for r in rows])
|
||||||
|
|
||||||
header = f"{'benchmark':<{name_w}} {'unit':<{unit_w}} " + \
|
header = f"{'benchmark':<{name_w}} {'unit':<{unit_w}} " + \
|
||||||
" ".join(f"{label(t):>{tag_w[i]}}" for i, t in enumerate(tags))
|
" ".join(f"{col_names[i]:>{col_w[i]}}" for i in range(len(col_names)))
|
||||||
print(header)
|
print(header)
|
||||||
print("-" * len(header))
|
print("-" * len(header))
|
||||||
for key, title, unit, higher, vals in rows:
|
for key, title, unit, higher, vals in rows:
|
||||||
cells = " ".join(f"{fmt(v):>{tag_w[i]}}" for i, v in enumerate(vals))
|
cells = " ".join(f"{fmt(v):>{col_w[i]}}" for i, v in enumerate(vals))
|
||||||
print(f"{title:<{name_w}} {unit:<{unit_w}} {cells}")
|
print(f"{title:<{name_w}} {unit:<{unit_w}} {cells}")
|
||||||
|
|
||||||
|
if groups:
|
||||||
|
print("\ngroups (median of runs):")
|
||||||
|
for name, members in columns:
|
||||||
|
print(f" {name:<12} <- {', '.join(members)}")
|
||||||
|
if excluded:
|
||||||
|
print(f"excluded: {', '.join(sorted(excluded))}")
|
||||||
|
|
||||||
csv_path = os.path.join(RES, "compare-scores.csv")
|
csv_path = os.path.join(RES, "compare-scores.csv")
|
||||||
with open(csv_path, "w", newline="") as fh:
|
with open(csv_path, "w", newline="") as fh:
|
||||||
w = csv.writer(fh)
|
w = csv.writer(fh)
|
||||||
w.writerow(["benchmark", "unit"] + [label(t) for t in tags])
|
w.writerow(["benchmark", "unit"] + col_names)
|
||||||
for key, title, unit, higher, vals in rows:
|
for key, title, unit, higher, vals in rows:
|
||||||
w.writerow([title, unit] + ["" if v is None else f"{v:g}" for v in vals])
|
w.writerow([title, unit] + ["" if v is None else f"{v:g}" for v in vals])
|
||||||
print(f"\ntable: {csv_path}")
|
print(f"\ntable: {csv_path}")
|
||||||
|
|
@ -295,7 +354,7 @@ def main():
|
||||||
return
|
return
|
||||||
graphs = []
|
graphs = []
|
||||||
for key, title, unit, higher, vals in rows:
|
for key, title, unit, higher, vals in rows:
|
||||||
out = bar_chart(key, title, unit, tags, vals, color)
|
out = bar_chart(key, title, unit, col_names, vals, color)
|
||||||
if out:
|
if out:
|
||||||
graphs.append(out)
|
graphs.append(out)
|
||||||
if graphs:
|
if graphs:
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue