synthbench-eval 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. synthbench/__init__.py +16 -0
  2. synthbench/__main__.py +5 -0
  3. synthbench/adapter.py +131 -0
  4. synthbench/anomaly.py +503 -0
  5. synthbench/baseline_floors.py +213 -0
  6. synthbench/baselines.py +902 -0
  7. synthbench/cli.py +2929 -0
  8. synthbench/config_id.py +424 -0
  9. synthbench/contamination.py +783 -0
  10. synthbench/convergence/__init__.py +67 -0
  11. synthbench/convergence/baseline.py +172 -0
  12. synthbench/convergence/bootstrap.py +60 -0
  13. synthbench/convergence/cli_report.py +556 -0
  14. synthbench/convergence/curves.py +111 -0
  15. synthbench/convergence/real_sampling.py +146 -0
  16. synthbench/convergence/thresholds.py +49 -0
  17. synthbench/datasets/__init__.py +37 -0
  18. synthbench/datasets/base.py +147 -0
  19. synthbench/datasets/eurobarometer.py +335 -0
  20. synthbench/datasets/globalopinionqa.py +252 -0
  21. synthbench/datasets/gss.py +323 -0
  22. synthbench/datasets/michigan.py +505 -0
  23. synthbench/datasets/ntia.py +357 -0
  24. synthbench/datasets/opinionsqa.py +412 -0
  25. synthbench/datasets/pewtech.py +334 -0
  26. synthbench/datasets/policy.py +142 -0
  27. synthbench/datasets/subpop.py +302 -0
  28. synthbench/datasets/wvs.py +229 -0
  29. synthbench/findings.py +988 -0
  30. synthbench/holdout.py +347 -0
  31. synthbench/human_distributions.py +243 -0
  32. synthbench/leaderboard.py +715 -0
  33. synthbench/leaderboard_pr.py +274 -0
  34. synthbench/metrics/__init__.py +38 -0
  35. synthbench/metrics/composite.py +72 -0
  36. synthbench/metrics/conditioning.py +39 -0
  37. synthbench/metrics/distributional.py +41 -0
  38. synthbench/metrics/ranking.py +37 -0
  39. synthbench/metrics/refusal.py +270 -0
  40. synthbench/metrics/subgroup.py +58 -0
  41. synthbench/private_holdout.py +240 -0
  42. synthbench/providers/__init__.py +43 -0
  43. synthbench/providers/_parsing.py +150 -0
  44. synthbench/providers/_retry.py +107 -0
  45. synthbench/providers/base.py +212 -0
  46. synthbench/providers/http.py +108 -0
  47. synthbench/providers/majority_baseline.py +23 -0
  48. synthbench/providers/ollama.py +109 -0
  49. synthbench/providers/openrouter.py +144 -0
  50. synthbench/providers/population_baseline.py +87 -0
  51. synthbench/providers/random_baseline.py +24 -0
  52. synthbench/providers/raw_anthropic.py +149 -0
  53. synthbench/providers/raw_gemini.py +145 -0
  54. synthbench/providers/raw_openai.py +139 -0
  55. synthbench/providers/synthpanel.py +978 -0
  56. synthbench/publish.py +2772 -0
  57. synthbench/r2_upload.py +178 -0
  58. synthbench/recompute.py +235 -0
  59. synthbench/report.py +493 -0
  60. synthbench/run_hash.py +111 -0
  61. synthbench/run_validity.py +232 -0
  62. synthbench/runner.py +834 -0
  63. synthbench/stats.py +1492 -0
  64. synthbench/submission.py +284 -0
  65. synthbench/submission_pr.py +436 -0
  66. synthbench/submit_adapter.py +748 -0
  67. synthbench/suite.py +424 -0
  68. synthbench/suites/__init__.py +94 -0
  69. synthbench/topics.py +256 -0
  70. synthbench/user_config.py +409 -0
  71. synthbench/validation.py +1556 -0
  72. synthbench/visualize.py +455 -0
  73. synthbench_eval-0.4.0.dist-info/METADATA +279 -0
  74. synthbench_eval-0.4.0.dist-info/RECORD +78 -0
  75. synthbench_eval-0.4.0.dist-info/WHEEL +5 -0
  76. synthbench_eval-0.4.0.dist-info/entry_points.txt +2 -0
  77. synthbench_eval-0.4.0.dist-info/licenses/LICENSE +21 -0
  78. synthbench_eval-0.4.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,213 @@
1
+ """Null-agent baseline floor discovery and drift detection (sb-lhoh).
2
+
3
+ Per Berkeley paper recommendation ("run a null agent; if it's not zero,
4
+ something is wrong"), SynthBench's equivalent null agents are the
5
+ ``random-baseline`` and ``majority-baseline`` providers. Their composite
6
+ parity ("SPS") must stay bounded; upward drift on a stable dataset
7
+ signals a scoring-function bug, not a success.
8
+
9
+ This module discovers the canonical per-dataset baseline SPS from
10
+ ``leaderboard-results/`` (mirroring leaderboard.build_baseline_scores:
11
+ max composite_parity per (provider, dataset)) and checks it against
12
+ configured ceilings.
13
+
14
+ Thresholds:
15
+ * ``MAJORITY_MAX_SPS = 0.85`` — Berkeley target. Current observed
16
+ max across datasets is ~0.71, leaving >0.13 headroom.
17
+ * ``RANDOM_MAX_SPS = 0.80`` — calibrated. Berkeley's aspirational
18
+ target is 0.70, but SynthBench's composite_parity weights
19
+ ``p_refuse`` (where uniform-random naturally matches human DK/
20
+ Refused rates), pushing observed random SPS to ~0.71-0.76.
21
+ We set the hard CI gate at 0.80 (drift detection over current
22
+ max) and surface 0.70 as the published aspirational floor.
23
+
24
+ See ``docs/benchmark-hardening-analysis.md`` §5.4 for full context.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import json
30
+ from dataclasses import dataclass
31
+ from pathlib import Path
32
+
33
+ BASELINE_PROVIDERS = ("random-baseline", "majority-baseline")
34
+
35
+ # Hard CI thresholds. A baseline run whose SPS meets-or-exceeds the
36
+ # threshold signals upward scoring-function drift and fails CI.
37
+ MAJORITY_MAX_SPS = 0.85
38
+ RANDOM_MAX_SPS = 0.80
39
+
40
+ # Aspirational floors from Wang et al. (UC Berkeley, 2026). Published on
41
+ # the methodology page as "the floor any benchmark-serious model must
42
+ # clear." The hard CI gate above is looser because SynthBench's
43
+ # composite_parity rewards random agents for matching human DK/Refused
44
+ # rates (high p_refuse) — a known property of the scoring protocol, not
45
+ # a bug. If p_refuse is rebalanced in future scoring revisions, the
46
+ # hard gate should drop toward these Berkeley targets.
47
+ ASPIRATIONAL_RANDOM_MAX_SPS = 0.70
48
+ ASPIRATIONAL_MAJORITY_MAX_SPS = 0.85
49
+
50
+
51
+ @dataclass(frozen=True)
52
+ class BaselineRun:
53
+ """One null-agent submission discovered under leaderboard-results/."""
54
+
55
+ provider: str
56
+ dataset: str
57
+ sps: float
58
+ n_evaluated: int
59
+ timestamp: str
60
+ source_file: str
61
+
62
+
63
+ @dataclass(frozen=True)
64
+ class FloorViolation:
65
+ """One canonical-baseline SPS that met-or-exceeded its ceiling."""
66
+
67
+ provider: str
68
+ dataset: str
69
+ sps: float
70
+ threshold: float
71
+ source_file: str
72
+
73
+ def format(self) -> str:
74
+ return (
75
+ f"{self.provider} on {self.dataset}: SPS={self.sps:.4f} "
76
+ f">= threshold {self.threshold:.2f} (source: {self.source_file})"
77
+ )
78
+
79
+
80
+ def _load_result(path: Path) -> dict | None:
81
+ try:
82
+ data = json.loads(path.read_text())
83
+ except (json.JSONDecodeError, OSError):
84
+ return None
85
+ if not isinstance(data, dict):
86
+ return None
87
+ cfg = data.get("config") or {}
88
+ if cfg.get("provider") not in BASELINE_PROVIDERS:
89
+ return None
90
+ agg = data.get("aggregate") or {}
91
+ scores = data.get("scores") or {}
92
+ sps = scores.get("sps")
93
+ if sps is None:
94
+ sps = agg.get("composite_parity")
95
+ if not isinstance(sps, (int, float)):
96
+ return None
97
+ return {
98
+ "provider": cfg["provider"],
99
+ "dataset": cfg.get("dataset", "unknown"),
100
+ "sps": float(sps),
101
+ "n_evaluated": int(cfg.get("n_evaluated") or agg.get("n_questions") or 0),
102
+ "timestamp": str(data.get("timestamp", "")),
103
+ }
104
+
105
+
106
+ def discover_baseline_runs(results_dir: Path | str) -> list[BaselineRun]:
107
+ """Enumerate every baseline submission under ``results_dir``.
108
+
109
+ Non-baseline submissions, malformed files, and files without an
110
+ SPS are silently skipped. Sorted by timestamp ascending to make
111
+ the output stable and usable as a drift log.
112
+ """
113
+ root = Path(results_dir)
114
+ runs: list[BaselineRun] = []
115
+ for path in sorted(root.glob("*.json")):
116
+ rec = _load_result(path)
117
+ if rec is None:
118
+ continue
119
+ runs.append(
120
+ BaselineRun(
121
+ provider=rec["provider"],
122
+ dataset=rec["dataset"],
123
+ sps=rec["sps"],
124
+ n_evaluated=rec["n_evaluated"],
125
+ timestamp=rec["timestamp"],
126
+ source_file=path.name,
127
+ )
128
+ )
129
+ runs.sort(key=lambda r: (r.provider, r.dataset, r.timestamp))
130
+ return runs
131
+
132
+
133
+ def canonical_baselines(
134
+ runs: list[BaselineRun],
135
+ ) -> dict[tuple[str, str], BaselineRun]:
136
+ """Pick the canonical (max-SPS) run per (provider, dataset).
137
+
138
+ Mirrors ``leaderboard.build_baseline_scores``: the value shown to
139
+ users as "the random baseline" on the leaderboard is the max
140
+ composite_parity across all runs for that provider/dataset. That
141
+ is the number a new submission must beat, so that is what we gate
142
+ on.
143
+ """
144
+ best: dict[tuple[str, str], BaselineRun] = {}
145
+ for run in runs:
146
+ key = (run.provider, run.dataset)
147
+ if key not in best or run.sps > best[key].sps:
148
+ best[key] = run
149
+ return best
150
+
151
+
152
+ def threshold_for(provider: str) -> float:
153
+ if provider == "random-baseline":
154
+ return RANDOM_MAX_SPS
155
+ if provider == "majority-baseline":
156
+ return MAJORITY_MAX_SPS
157
+ raise ValueError(f"unknown baseline provider: {provider!r}")
158
+
159
+
160
+ def check_floors(
161
+ canonicals: dict[tuple[str, str], BaselineRun],
162
+ ) -> list[FloorViolation]:
163
+ """Return any canonical-baseline entry whose SPS >= its threshold."""
164
+ violations: list[FloorViolation] = []
165
+ for (provider, dataset), run in sorted(canonicals.items()):
166
+ t = threshold_for(provider)
167
+ if run.sps >= t:
168
+ violations.append(
169
+ FloorViolation(
170
+ provider=provider,
171
+ dataset=dataset,
172
+ sps=run.sps,
173
+ threshold=t,
174
+ source_file=run.source_file,
175
+ )
176
+ )
177
+ return violations
178
+
179
+
180
+ def summary_report(
181
+ canonicals: dict[tuple[str, str], BaselineRun],
182
+ ) -> str:
183
+ """Render a human-readable floor summary. Used by tests and scripts."""
184
+ lines = ["Null-agent baseline floors (max SPS per provider/dataset):"]
185
+ by_provider: dict[str, list[BaselineRun]] = {}
186
+ for run in canonicals.values():
187
+ by_provider.setdefault(run.provider, []).append(run)
188
+ for provider in sorted(by_provider):
189
+ t = threshold_for(provider)
190
+ lines.append(f" {provider} (CI threshold: SPS < {t:.2f})")
191
+ for run in sorted(by_provider[provider], key=lambda r: r.dataset):
192
+ status = "OK " if run.sps < t else "FAIL"
193
+ lines.append(
194
+ f" [{status}] {run.dataset:20s} SPS={run.sps:.4f} "
195
+ f"n={run.n_evaluated:4d} ({run.source_file})"
196
+ )
197
+ return "\n".join(lines)
198
+
199
+
200
+ def history_report(runs: list[BaselineRun]) -> str:
201
+ """Render every observed baseline run in time order.
202
+
203
+ This is the persisted drift log: every baseline submission ever
204
+ checked into leaderboard-results/ is listed with its SPS. Grep
205
+ this output to watch a specific (provider, dataset) over time.
206
+ """
207
+ lines = ["Null-agent baseline SPS history (ascending by time):"]
208
+ for run in sorted(runs, key=lambda r: (r.provider, r.dataset, r.timestamp)):
209
+ lines.append(
210
+ f" {run.timestamp:30s} {run.provider:20s} "
211
+ f"{run.dataset:20s} SPS={run.sps:.4f} n={run.n_evaluated:4d}"
212
+ )
213
+ return "\n".join(lines)