Skip to content

Commit 8220a25

Browse files
committed
ci: run performance benchmarks on PRs with a per-platform verdict
Add a benchmarks workflow that builds the PR head and its merge-base, runs the full bench suite in interleaved base/head passes on dedicated self-hosted runners (Linux epoll+uring, Windows iocp, macOS kqueue), and posts one advisory PR comment, updated in place on each push. A row is flagged only when its mean paired delta exceeds the larger of a fixed minimum effect size and three times the spread observed across the base-side runs of the same session, so the noise floor is measured live on the same machine rather than maintained as stored calibration. Iterations alternate starting side so linear drift cancels. The comparison and report generation live in a stdlib-only script, .github/bench/compare.py. It matches benchmark suites dynamically: benchmarks present on only one side are reported as added or removed rather than compared, unrecognized metrics degrade to an unsupported list, and malformed input files are warned about and skipped — the report never fails a job over suite shape. The workflow currently carries a temporary smoke profile and branch push trigger for bring-up on the dedicated machines; the production profile and gated pull_request_target trigger replace them once the runners are validated.
1 parent 54e28fd commit 8220a25

2 files changed

Lines changed: 546 additions & 0 deletions

File tree

‎.github/bench/compare.py‎

Lines changed: 293 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,293 @@
1+
#!/usr/bin/env python3
2+
"""Compare interleaved corosio benchmark runs.
3+
4+
summarize: reduce raw per-iteration JSON (base/head × backend) to one
5+
per-platform summary with flagging.
6+
report: merge per-platform summaries into the PR comment markdown.
7+
8+
Input files: <side>-<backend>-<iter>.json as written by
9+
corosio_bench --output. Never exits nonzero because of suite-shape
10+
differences between base and head; only real I/O or usage errors fail.
11+
"""
12+
import argparse
13+
import json
14+
import re
15+
import statistics
16+
import sys
17+
from pathlib import Path
18+
19+
MIN_EFFECT_PCT = 2.0
20+
NOISE_FACTOR = 3.0
21+
HIGHER_BETTER = ("bytes_per_sec", "items_per_sec", "ops_per_sec")
22+
FNAME = re.compile(r"^(base|head)-([A-Za-z0-9_]+)-(\d+)\.json$")
23+
24+
25+
def primary_metric(category, metrics):
26+
"""Pick the compared metric and its direction for one benchmark."""
27+
if "latency" in category and "latency_mean_ns" in metrics:
28+
return "latency_mean_ns", "lower"
29+
for m in HIGHER_BETTER:
30+
if m in metrics:
31+
return m, "higher"
32+
if "latency_mean_ns" in metrics:
33+
return "latency_mean_ns", "lower"
34+
return None, None
35+
36+
37+
def human(value, metric):
38+
"""Format a metric value with readable units."""
39+
if metric.endswith("_ns"):
40+
for factor, unit in ((1e9, "s"), (1e6, "ms"), (1e3, "µs")):
41+
if abs(value) >= factor:
42+
return f"{value / factor:.2f} {unit}"
43+
return f"{value:.0f} ns"
44+
suffix = "B/s" if metric == "bytes_per_sec" else "/s"
45+
for factor, prefix in ((1e9, "G"), (1e6, "M"), (1e3, "K")):
46+
if abs(value) >= factor:
47+
n = value / factor
48+
return f"{n:.2f} {prefix}{suffix}" if n < 100 else f"{n:.1f} {prefix}{suffix}"
49+
return f"{value:.1f} {suffix}"
50+
51+
52+
MARKER = "<!-- corosio-bench-report -->"
53+
54+
55+
def _row_line(platform, r):
56+
arrow = "🔻" if r["delta_pct"] < 0 else "🔺"
57+
noise = "—" if r["noise_pct"] is None else f"{r['noise_pct']:.2f}%"
58+
return (f"| {platform} | {r['backend']} | {r['category']} | {r['name']} "
59+
f"| {r['metric']} | {human(r['base_mean'], r['metric'])} "
60+
f"| {human(r['head_mean'], r['metric'])} "
61+
f"| {arrow} {r['delta_pct']:+.2f}% | {noise} |")
62+
63+
64+
def report(summaries, base_sha, head_sha, run_url):
65+
lines = [MARKER, "## Benchmark report", ""]
66+
mode = next((s["mode"] for s in summaries.values() if s), "ab")
67+
if mode == "aa":
68+
lines += ["**A/A validation run** — base compared against itself; "
69+
"every flag below is a false positive.", ""]
70+
lines += [f"`{base_sha[:12]}` (base) vs `{head_sha[:12]}` (head)", ""]
71+
72+
for platform, s in summaries.items():
73+
if s is None:
74+
lines.append(f"- ❌ **{platform}** — no results "
75+
"(job failed or runner offline)")
76+
elif s["flagged_count"]:
77+
lines.append(f"- ⚠️ **{platform}** — {s['flagged_count']} flagged "
78+
f"({', '.join(s['backends'])})")
79+
else:
80+
lines.append(f"- ✅ **{platform}** — clean "
81+
f"({', '.join(s['backends'])})")
82+
lines.append("")
83+
84+
header = ("| Platform | Backend | Category | Benchmark | Metric "
85+
"| Base | Head | Δ | Noise |")
86+
rule = "|---|---|---|---|---|---|---|---|---|"
87+
88+
flagged = [(p, r) for p, s in summaries.items() if s
89+
for r in s["rows"] if r["flagged"]]
90+
if flagged:
91+
lines += ["### ⚠️ Flagged", "", header, rule]
92+
lines += [_row_line(p, r) for p, r in flagged]
93+
lines.append("")
94+
95+
for platform, s in summaries.items():
96+
if s is None:
97+
continue
98+
lines += [f"<details><summary>{platform} — full results "
99+
f"({len(s['rows'])} benchmarks, {s['iterations']} iterations, "
100+
f"{s['duration_s']}s each)</summary>", "", header, rule]
101+
lines += [_row_line(platform, r) for r in s["rows"]]
102+
lines.append("")
103+
if s["new"]:
104+
lines += ["**New benchmarks (no baseline):**", ""]
105+
lines += [f"- `{n['category']}/{n['name']}` [{n['backend']}] "
106+
f"{human(n['head_mean'], n['metric'])}" for n in s["new"]]
107+
lines.append("")
108+
if s["removed"]:
109+
lines += ["**Removed benchmarks:** " +
110+
", ".join(f"`{r['category']}/{r['name']}`"
111+
for r in s["removed"]), ""]
112+
if s["unsupported"]:
113+
lines += ["**Unsupported (no recognized metric):** " +
114+
", ".join(f"`{u['category']}/{u['name']}`"
115+
for u in s["unsupported"]), ""]
116+
lines += ["</details>", ""]
117+
118+
lines += [f"[Run & raw JSON artifacts]({run_url}) · "
119+
"flag rule: |Δ| > max(2%, 3×CV of base runs) · advisory only"]
120+
return "\n".join(lines) + "\n"
121+
122+
123+
def load_runs(input_dir):
124+
"""Return {(side, backend, iter): {(category, name): {metric: value}}}."""
125+
runs = {}
126+
for p in sorted(Path(input_dir).iterdir()):
127+
m = FNAME.match(p.name)
128+
if not m:
129+
continue
130+
side, backend, it = m.group(1), m.group(2), int(m.group(3))
131+
try:
132+
payload = json.loads(p.read_text())
133+
except (OSError, json.JSONDecodeError) as e:
134+
print(f"warning: skipping unreadable {p.name}: {e}", file=sys.stderr)
135+
continue
136+
if not isinstance(payload, dict):
137+
print(f"warning: skipping non-object JSON {p.name}", file=sys.stderr)
138+
continue
139+
benchmarks = payload.get("benchmarks", [])
140+
if not isinstance(benchmarks, list):
141+
print(f"warning: skipping {p.name}: benchmarks field is not a list", file=sys.stderr)
142+
continue
143+
table = {}
144+
for b in benchmarks:
145+
if not isinstance(b, dict):
146+
print(f"warning: skipping non-object benchmark entry in {p.name}", file=sys.stderr)
147+
continue
148+
key = (b.get("category", ""), b.get("name", ""))
149+
table[key] = {
150+
k: v for k, v in b.items()
151+
if isinstance(v, (int, float)) and not isinstance(v, bool)
152+
}
153+
runs[(side, backend, it)] = table
154+
return runs
155+
156+
157+
def _values(runs, side, backend, key, metric):
158+
"""Metric samples for one benchmark on one side, ordered by iteration."""
159+
out = []
160+
for (s, b, it), table in sorted(runs.items(), key=lambda kv: kv[0][2]):
161+
if s == side and b == backend and key in table and metric in table[key]:
162+
out.append((it, table[key][metric]))
163+
return out
164+
165+
166+
def summarize(input_dir, platform, mode="ab"):
167+
runs = load_runs(input_dir)
168+
backends = sorted({b for (_, b, _) in runs})
169+
iterations = max((it for (_, _, it) in runs), default=0)
170+
duration = 0.0
171+
rows, new, removed, unsupported = [], [], [], []
172+
173+
for backend in backends:
174+
base_keys, head_keys = set(), set()
175+
sample = {}
176+
for (s, b, it), table in runs.items():
177+
if b != backend:
178+
continue
179+
(base_keys if s == "base" else head_keys).update(table)
180+
for key, metrics in table.items():
181+
sample.setdefault(key, metrics)
182+
183+
for key in sorted(base_keys | head_keys):
184+
category, name = key
185+
metric, direction = primary_metric(category, sample.get(key, {}))
186+
if metric is None:
187+
unsupported.append(
188+
{"backend": backend, "category": category, "name": name})
189+
continue
190+
if key not in base_keys:
191+
head = _values(runs, "head", backend, key, metric)
192+
mean = statistics.fmean(v for _, v in head) if head else 0.0
193+
new.append({"backend": backend, "category": category,
194+
"name": name, "metric": metric, "head_mean": mean})
195+
continue
196+
if key not in head_keys:
197+
removed.append(
198+
{"backend": backend, "category": category, "name": name})
199+
continue
200+
201+
base = dict(_values(runs, "base", backend, key, metric))
202+
head = dict(_values(runs, "head", backend, key, metric))
203+
common = sorted(set(base) & set(head))
204+
deltas = []
205+
for it in common:
206+
b_v, h_v = base[it], head[it]
207+
if b_v == 0:
208+
continue
209+
d = (h_v - b_v) / b_v * 100.0
210+
if direction == "lower":
211+
d = -d
212+
deltas.append(d)
213+
if not deltas:
214+
unsupported.append(
215+
{"backend": backend, "category": category, "name": name})
216+
continue
217+
218+
base_vals = [base[it] for it in common]
219+
base_mean = statistics.fmean(base_vals)
220+
head_mean = statistics.fmean(head[it] for it in common)
221+
noise_pct = None
222+
if len(base_vals) >= 2 and base_mean != 0:
223+
noise_pct = statistics.stdev(base_vals) / abs(base_mean) * 100.0
224+
delta_pct = statistics.fmean(deltas)
225+
flagged = (
226+
noise_pct is not None
227+
and abs(delta_pct) > max(MIN_EFFECT_PCT, NOISE_FACTOR * noise_pct)
228+
)
229+
rows.append({
230+
"backend": backend, "category": category, "name": name,
231+
"metric": metric, "direction": direction,
232+
"base_mean": base_mean, "head_mean": head_mean,
233+
"delta_pct": delta_pct, "noise_pct": noise_pct,
234+
"flagged": flagged,
235+
})
236+
237+
for p in Path(input_dir).iterdir():
238+
if FNAME.match(p.name):
239+
try:
240+
duration = json.loads(p.read_text())["metadata"]["duration_s"]
241+
break
242+
except Exception:
243+
pass
244+
245+
return {
246+
"platform": platform, "backends": backends,
247+
"iterations": iterations, "duration_s": duration, "mode": mode,
248+
"rows": rows, "new": new, "removed": removed,
249+
"unsupported": unsupported,
250+
"flagged_count": sum(1 for r in rows if r["flagged"]),
251+
}
252+
253+
254+
def main(argv=None):
255+
ap = argparse.ArgumentParser(prog="compare.py")
256+
sub = ap.add_subparsers(dest="cmd", required=True)
257+
s = sub.add_parser("summarize")
258+
s.add_argument("--platform", required=True)
259+
s.add_argument("--input-dir", required=True)
260+
s.add_argument("--output", required=True)
261+
s.add_argument("--mode", default="ab", choices=("ab", "aa"))
262+
r = sub.add_parser("report")
263+
r.add_argument("--summaries", required=True,
264+
help="dir containing bench-<platform>/summary.json")
265+
r.add_argument("--expect", required=True,
266+
help="comma-separated platform list")
267+
r.add_argument("--base-sha", required=True)
268+
r.add_argument("--head-sha", required=True)
269+
r.add_argument("--run-url", required=True)
270+
r.add_argument("--output", required=True)
271+
args = ap.parse_args(argv)
272+
273+
if args.cmd == "summarize":
274+
summary = summarize(args.input_dir, args.platform, args.mode)
275+
Path(args.output).write_text(json.dumps(summary, indent=2))
276+
print(f"{args.platform}: {len(summary['rows'])} rows, "
277+
f"{summary['flagged_count']} flagged")
278+
elif args.cmd == "report":
279+
summaries = {}
280+
for platform in args.expect.split(","):
281+
p = Path(args.summaries) / f"bench-{platform}" / "summary.json"
282+
try:
283+
summaries[platform] = json.loads(p.read_text())
284+
except (OSError, json.JSONDecodeError):
285+
summaries[platform] = None
286+
md = report(summaries, args.base_sha, args.head_sha, args.run_url)
287+
Path(args.output).write_text(md)
288+
print(f"report written: {args.output}")
289+
return 0
290+
291+
292+
if __name__ == "__main__":
293+
sys.exit(main())

0 commit comments

Comments
 (0)