Skip to content

Commit d928b08

Browse files
fix(ci): skip C4 deepeval tests when dep missing, drop pr_drafts links from docs nav, run pre-commit auto-fixes
1 parent 24c023a commit d928b08

24 files changed

Lines changed: 819 additions & 761 deletions

‎.pre-commit-config.yaml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -22,7 +22,7 @@ repos:
2222
hooks:
2323
- id: codespell
2424
args:
25-
- "--ignore-words-list=iff,nd,te,hist,ot,ist,ment,worl,bu,som,covert,strat,intoto,unparseable,re-use,socio-economic,rouge"
25+
- "--ignore-words-list=iff,nd,te,hist,ot,ist,ment,worl,bu,som,covert,strat,intoto,unparseable,re-use,socio-economic,rouge,unx,ord,cov"
2626
exclude: |
2727
(?x)^(
2828
CHANGELOG\.md|

‎benchmarks/paper/experiments/01_controlled_noise.py‎

Lines changed: 14 additions & 44 deletions
Original file line numberDiff line numberDiff line change
@@ -192,18 +192,11 @@ def run_experiment(
192192
user_instruction=task.user_instruction,
193193
tools=task.tools,
194194
)
195-
_, subs, passed = _score_trajectory(
196-
predicted, task.reference_actions
197-
)
195+
_, subs, passed = _score_trajectory(predicted, task.reference_actions)
198196
traj_path = (
199-
output_root
200-
/ domain
201-
/ noise.name
202-
/ f"seed{seed}_{task.task_id}.jsonl"
203-
)
204-
_write_trajectory_jsonl(
205-
traj_path, predicted, task.reference_actions
197+
output_root / domain / noise.name / f"seed{seed}_{task.task_id}.jsonl"
206198
)
199+
_write_trajectory_jsonl(traj_path, predicted, task.reference_actions)
207200
timestamp = datetime.now(timezone.utc).isoformat()
208201

209202
manifest_row: dict[str, Any] = {
@@ -219,9 +212,7 @@ def run_experiment(
219212
"passed": bool(passed),
220213
"reference_action_count": len(task.reference_actions),
221214
"predicted_action_count": len(predicted),
222-
"trajectory_path": str(
223-
traj_path.relative_to(output_root)
224-
),
215+
"trajectory_path": str(traj_path.relative_to(output_root)),
225216
"timestamp_utc": timestamp,
226217
"model_version_sha": None,
227218
"benchmark_sha": None,
@@ -254,25 +245,19 @@ def run_experiment(
254245
fh.write(json.dumps(row) + "\n")
255246

256247
summary = _build_summary(jsonl_rows)
257-
summary_path.write_text(
258-
json.dumps(summary, indent=2, sort_keys=True), encoding="utf-8"
259-
)
248+
summary_path.write_text(json.dumps(summary, indent=2, sort_keys=True), encoding="utf-8")
260249

261250
manifest = {
262251
"schema": "https://checkllm.dev/schemas/paper_manifest/v1",
263252
"experiment_id": EXPERIMENT_ID,
264253
"generated_at_utc": datetime.now(timezone.utc).isoformat(),
265-
"noise_levels": [
266-
{"name": n.name, **n.as_params()} for n in NOISE_LEVELS
267-
],
254+
"noise_levels": [{"name": n.name, **n.as_params()} for n in NOISE_LEVELS],
268255
"seeds": list(SEEDS),
269256
"domains": list(DOMAINS),
270257
"limit_tasks": limit_tasks,
271258
"rows": rows,
272259
}
273-
manifest_path.write_text(
274-
json.dumps(manifest, indent=2), encoding="utf-8"
275-
)
260+
manifest_path.write_text(json.dumps(manifest, indent=2), encoding="utf-8")
276261

277262
return {
278263
"total_rows": len(rows),
@@ -302,35 +287,23 @@ def _build_summary(jsonl_rows: list[dict[str, Any]]) -> dict[str, Any]:
302287
domain_noise_means[domain] = {}
303288
for noise in NOISE_LEVELS:
304289
bucket = [
305-
r
306-
for r in jsonl_rows
307-
if r["domain"] == domain and r["noise_level"] == noise.name
290+
r for r in jsonl_rows if r["domain"] == domain and r["noise_level"] == noise.name
308291
]
309292
stats: dict[str, dict[str, float | int]] = {}
310293
for key in sub_keys:
311294
values = [float(r[key]) for r in bucket]
312295
stats[key] = {
313296
"mean": float(statistics.fmean(values)) if values else 0.0,
314-
"stdev": (
315-
float(statistics.pstdev(values))
316-
if len(values) > 1
317-
else 0.0
318-
),
297+
"stdev": (float(statistics.pstdev(values)) if len(values) > 1 else 0.0),
319298
"n": len(values),
320299
}
321300
domain_noise_stats[domain][noise.name] = stats
322-
domain_noise_means[domain][noise.name] = float(
323-
stats["overall"]["mean"]
324-
)
301+
domain_noise_means[domain][noise.name] = float(stats["overall"]["mean"])
325302

326303
monotonicity_check: dict[str, bool] = {}
327304
for domain in DOMAINS:
328-
ordered_means = [
329-
domain_noise_means[domain][n.name] for n in NOISE_LEVELS
330-
]
331-
monotonicity_check[domain] = _is_monotonic_non_increasing(
332-
ordered_means
333-
)
305+
ordered_means = [domain_noise_means[domain][n.name] for n in NOISE_LEVELS]
306+
monotonicity_check[domain] = _is_monotonic_non_increasing(ordered_means)
334307

335308
return {
336309
"experiment_id": EXPERIMENT_ID,
@@ -347,8 +320,7 @@ def _parse_args(argv: list[str] | None = None) -> argparse.Namespace:
347320
"""Build the argparse namespace for the CLI entry point."""
348321
parser = argparse.ArgumentParser(
349322
description=(
350-
"Controlled-noise validation experiment for TrajectoryMetric "
351-
"(Task C1-zero)."
323+
"Controlled-noise validation experiment for TrajectoryMetric " "(Task C1-zero)."
352324
)
353325
)
354326
parser.add_argument(
@@ -376,9 +348,7 @@ def main(argv: list[str] | None = None) -> int:
376348
Exit code; ``0`` on success.
377349
"""
378350
args = _parse_args(argv)
379-
summary = run_experiment(
380-
output_dir=args.output_dir, limit_tasks=args.limit_tasks
381-
)
351+
summary = run_experiment(output_dir=args.output_dir, limit_tasks=args.limit_tasks)
382352
print(json.dumps(summary, indent=2))
383353
return 0
384354

‎benchmarks/paper/experiments/02_metric_vs_truth.py‎

Lines changed: 12 additions & 37 deletions
Original file line numberDiff line numberDiff line change
@@ -140,9 +140,7 @@ def _score_trajectory(
140140
return metric.compute_subscores(predicted_names).as_dict()
141141

142142

143-
def _auroc_mann_whitney(
144-
scores: np.ndarray, labels: np.ndarray
145-
) -> float:
143+
def _auroc_mann_whitney(scores: np.ndarray, labels: np.ndarray) -> float:
146144
"""Compute AUROC via the Mann-Whitney U statistic on average ranks.
147145
148146
Equivalent to :func:`sklearn.metrics.roc_auc_score` for binary
@@ -232,9 +230,7 @@ def _bootstrap_ci(
232230
resample produced a defined statistic.
233231
"""
234232

235-
def _vectorized_statistic(
236-
sample_scores: np.ndarray, sample_labels: np.ndarray
237-
) -> float:
233+
def _vectorized_statistic(sample_scores: np.ndarray, sample_labels: np.ndarray) -> float:
238234
# scipy.stats.bootstrap calls the statistic on each resample;
239235
# vectorize=False (the default for non-vectorised callables) is
240236
# set in the bootstrap call so each resample arrives as 1-D.
@@ -297,9 +293,7 @@ def _statistics_block(
297293
("auroc", _auroc_mann_whitney),
298294
):
299295
value = fn(scores, labels)
300-
ci_lower, ci_upper = _bootstrap_ci(
301-
scores, labels, fn, n_bootstrap=n_bootstrap, rng=rng
302-
)
296+
ci_lower, ci_upper = _bootstrap_ci(scores, labels, fn, n_bootstrap=n_bootstrap, rng=rng)
303297
block[name] = {
304298
"value": float(value),
305299
"ci_lower": float(ci_lower),
@@ -410,26 +404,20 @@ def run_experiment(
410404
fh.write(json.dumps(row) + "\n")
411405

412406
summary = _build_summary(score_rows, n_bootstrap=n_bootstrap)
413-
summary_path.write_text(
414-
json.dumps(summary, indent=2, sort_keys=True), encoding="utf-8"
415-
)
407+
summary_path.write_text(json.dumps(summary, indent=2, sort_keys=True), encoding="utf-8")
416408

417409
manifest = {
418410
"schema": "https://checkllm.dev/schemas/paper_manifest/v1",
419411
"experiment_id": EXPERIMENT_ID,
420412
"generated_at_utc": datetime.now(timezone.utc).isoformat(),
421-
"noise_levels": [
422-
{"name": n.name, **n.as_params()} for n in NOISE_LEVELS
423-
],
413+
"noise_levels": [{"name": n.name, **n.as_params()} for n in NOISE_LEVELS],
424414
"seeds": list(SEEDS),
425415
"domains": list(DOMAINS),
426416
"limit_tasks": limit_tasks,
427417
"n_bootstrap": n_bootstrap,
428418
"rows": manifest_rows,
429419
}
430-
manifest_path.write_text(
431-
json.dumps(manifest, indent=2), encoding="utf-8"
432-
)
420+
manifest_path.write_text(json.dumps(manifest, indent=2), encoding="utf-8")
433421

434422
return {
435423
"n_trajectories": len(manifest_rows),
@@ -439,9 +427,7 @@ def run_experiment(
439427
}
440428

441429

442-
def _build_summary(
443-
score_rows: list[dict[str, Any]], n_bootstrap: int
444-
) -> dict[str, Any]:
430+
def _build_summary(score_rows: list[dict[str, Any]], n_bootstrap: int) -> dict[str, Any]:
445431
"""Aggregate per-trajectory rows into the paper-summary statistics.
446432
447433
Args:
@@ -458,26 +444,20 @@ def _build_summary(
458444
labels = np.asarray([r["label"] for r in score_rows], dtype=int)
459445
overall_scores = np.asarray([r["score"] for r in score_rows], dtype=float)
460446

461-
overall_block = _statistics_block(
462-
overall_scores, labels, n_bootstrap=n_bootstrap, rng=rng
463-
)
447+
overall_block = _statistics_block(overall_scores, labels, n_bootstrap=n_bootstrap, rng=rng)
464448

465449
per_domain: dict[str, dict[str, dict[str, float]]] = {}
466450
auroc_gate_passed: dict[str, bool] = {}
467451
for domain in DOMAINS:
468452
bucket = [r for r in score_rows if r["domain"] == domain]
469453
d_scores = np.asarray([r["score"] for r in bucket], dtype=float)
470454
d_labels = np.asarray([r["label"] for r in bucket], dtype=int)
471-
block = _statistics_block(
472-
d_scores, d_labels, n_bootstrap=n_bootstrap, rng=rng
473-
)
455+
block = _statistics_block(d_scores, d_labels, n_bootstrap=n_bootstrap, rng=rng)
474456
per_domain[domain] = block
475457
ci_lower = block["auroc"]["ci_lower"]
476458
# Gate passes when the 95% CI lower bound is at or above 0.70.
477459
# NaN (undefined AUROC) fails the gate.
478-
auroc_gate_passed[domain] = bool(
479-
not np.isnan(ci_lower) and ci_lower >= AUROC_GATE_LOWER
480-
)
460+
auroc_gate_passed[domain] = bool(not np.isnan(ci_lower) and ci_lower >= AUROC_GATE_LOWER)
481461

482462
per_metric: dict[str, dict[str, dict[str, float]]] = {}
483463
for sub_metric in SUB_METRIC_KEYS:
@@ -507,9 +487,7 @@ def _build_summary(
507487
def _parse_args(argv: list[str] | None = None) -> argparse.Namespace:
508488
"""Build the argparse namespace for the CLI entry point."""
509489
parser = argparse.ArgumentParser(
510-
description=(
511-
"Metric-vs-synthetic-truth correlation study (Task C2-zero)."
512-
)
490+
description=("Metric-vs-synthetic-truth correlation study (Task C2-zero).")
513491
)
514492
parser.add_argument(
515493
"--output-dir",
@@ -527,10 +505,7 @@ def _parse_args(argv: list[str] | None = None) -> argparse.Namespace:
527505
"--n-bootstrap",
528506
type=int,
529507
default=DEFAULT_N_BOOTSTRAP,
530-
help=(
531-
"Number of percentile bootstrap resamples for 95%% CIs "
532-
"(default: 1000)."
533-
),
508+
help=("Number of percentile bootstrap resamples for 95%% CIs " "(default: 1000)."),
534509
)
535510
return parser.parse_args(argv)
536511

‎benchmarks/paper/experiments/03_trajectory_ablation.py‎

Lines changed: 8 additions & 24 deletions
Original file line numberDiff line numberDiff line change
@@ -204,12 +204,9 @@ def _build_trajectory_pool(
204204
"task_id": task.task_id,
205205
"seed": seed,
206206
"noise_level": noise.name,
207-
"predicted_names": [
208-
str(a.get("name", "")) for a in predicted
209-
],
207+
"predicted_names": [str(a.get("name", "")) for a in predicted],
210208
"reference_names": [
211-
str(a.get("name", ""))
212-
for a in task.reference_actions
209+
str(a.get("name", "")) for a in task.reference_actions
213210
],
214211
"label": 1 if noise.name == "clean" else 0,
215212
}
@@ -342,9 +339,7 @@ def _build_heatmap(rows: list[dict[str, Any]]) -> dict[str, dict[str, Any]]:
342339
vals = [
343340
r["spearman_rho"]
344341
for r in rows
345-
if r[key_a] == va
346-
and r[key_b] == vb
347-
and not np.isnan(r["spearman_rho"])
342+
if r[key_a] == va and r[key_b] == vb and not np.isnan(r["spearman_rho"])
348343
]
349344
row.append(float(np.mean(vals)) if vals else None)
350345
grid.append(row)
@@ -386,9 +381,7 @@ def _weight_correlations(rows: list[dict[str, Any]]) -> dict[str, dict[str, floa
386381
return out
387382

388383

389-
def _is_pareto_optimal(
390-
rows: list[dict[str, Any]], default_row: dict[str, Any]
391-
) -> bool:
384+
def _is_pareto_optimal(rows: list[dict[str, Any]], default_row: dict[str, Any]) -> bool:
392385
"""True iff no other cell strictly dominates the default on both metrics.
393386
394387
A cell strictly dominates iff its Spearman rho and its AUROC are
@@ -410,10 +403,7 @@ def _is_pareto_optimal(
410403
continue
411404
if np.isnan(r["spearman_rho"]) or np.isnan(r["auroc"]):
412405
continue
413-
if (
414-
r["spearman_rho"] > default_rho
415-
and r["auroc"] > default_auc
416-
):
406+
if r["spearman_rho"] > default_rho and r["auroc"] > default_auc:
417407
return False
418408
return True
419409

@@ -448,9 +438,7 @@ def run_experiment(
448438
pool = _build_trajectory_pool(seeds, noise_levels, domains, limit_tasks)
449439

450440
all_cells = list(_iter_grid_cells())
451-
n_skipped_degenerate = sum(
452-
1 for cell in all_cells if _is_degenerate(cell[:4])
453-
)
441+
n_skipped_degenerate = sum(1 for cell in all_cells if _is_degenerate(cell[:4]))
454442
valid_cells = [cell for cell in all_cells if not _is_degenerate(cell[:4])]
455443

456444
if limit_cells is not None and limit_cells < len(valid_cells):
@@ -558,14 +546,10 @@ def run_experiment(
558546
"weight_correlations": _weight_correlations(rows),
559547
"wall_seconds": float(wall_seconds),
560548
}
561-
summary_path.write_text(
562-
json.dumps(summary, indent=2, sort_keys=True), encoding="utf-8"
563-
)
549+
summary_path.write_text(json.dumps(summary, indent=2, sort_keys=True), encoding="utf-8")
564550

565551
heatmap = _build_heatmap(rows)
566-
heatmap_path.write_text(
567-
json.dumps(heatmap, indent=2, sort_keys=True), encoding="utf-8"
568-
)
552+
heatmap_path.write_text(json.dumps(heatmap, indent=2, sort_keys=True), encoding="utf-8")
569553

570554
return {
571555
"n_cells": len(rows),

0 commit comments

Comments
 (0)