synthbench-eval 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- synthbench/__init__.py +16 -0
- synthbench/__main__.py +5 -0
- synthbench/adapter.py +131 -0
- synthbench/anomaly.py +503 -0
- synthbench/baseline_floors.py +213 -0
- synthbench/baselines.py +902 -0
- synthbench/cli.py +2929 -0
- synthbench/config_id.py +424 -0
- synthbench/contamination.py +783 -0
- synthbench/convergence/__init__.py +67 -0
- synthbench/convergence/baseline.py +172 -0
- synthbench/convergence/bootstrap.py +60 -0
- synthbench/convergence/cli_report.py +556 -0
- synthbench/convergence/curves.py +111 -0
- synthbench/convergence/real_sampling.py +146 -0
- synthbench/convergence/thresholds.py +49 -0
- synthbench/datasets/__init__.py +37 -0
- synthbench/datasets/base.py +147 -0
- synthbench/datasets/eurobarometer.py +335 -0
- synthbench/datasets/globalopinionqa.py +252 -0
- synthbench/datasets/gss.py +323 -0
- synthbench/datasets/michigan.py +505 -0
- synthbench/datasets/ntia.py +357 -0
- synthbench/datasets/opinionsqa.py +412 -0
- synthbench/datasets/pewtech.py +334 -0
- synthbench/datasets/policy.py +142 -0
- synthbench/datasets/subpop.py +302 -0
- synthbench/datasets/wvs.py +229 -0
- synthbench/findings.py +988 -0
- synthbench/holdout.py +347 -0
- synthbench/human_distributions.py +243 -0
- synthbench/leaderboard.py +715 -0
- synthbench/leaderboard_pr.py +274 -0
- synthbench/metrics/__init__.py +38 -0
- synthbench/metrics/composite.py +72 -0
- synthbench/metrics/conditioning.py +39 -0
- synthbench/metrics/distributional.py +41 -0
- synthbench/metrics/ranking.py +37 -0
- synthbench/metrics/refusal.py +270 -0
- synthbench/metrics/subgroup.py +58 -0
- synthbench/private_holdout.py +240 -0
- synthbench/providers/__init__.py +43 -0
- synthbench/providers/_parsing.py +150 -0
- synthbench/providers/_retry.py +107 -0
- synthbench/providers/base.py +212 -0
- synthbench/providers/http.py +108 -0
- synthbench/providers/majority_baseline.py +23 -0
- synthbench/providers/ollama.py +109 -0
- synthbench/providers/openrouter.py +144 -0
- synthbench/providers/population_baseline.py +87 -0
- synthbench/providers/random_baseline.py +24 -0
- synthbench/providers/raw_anthropic.py +149 -0
- synthbench/providers/raw_gemini.py +145 -0
- synthbench/providers/raw_openai.py +139 -0
- synthbench/providers/synthpanel.py +978 -0
- synthbench/publish.py +2772 -0
- synthbench/r2_upload.py +178 -0
- synthbench/recompute.py +235 -0
- synthbench/report.py +493 -0
- synthbench/run_hash.py +111 -0
- synthbench/run_validity.py +232 -0
- synthbench/runner.py +834 -0
- synthbench/stats.py +1492 -0
- synthbench/submission.py +284 -0
- synthbench/submission_pr.py +436 -0
- synthbench/submit_adapter.py +748 -0
- synthbench/suite.py +424 -0
- synthbench/suites/__init__.py +94 -0
- synthbench/topics.py +256 -0
- synthbench/user_config.py +409 -0
- synthbench/validation.py +1556 -0
- synthbench/visualize.py +455 -0
- synthbench_eval-0.4.0.dist-info/METADATA +279 -0
- synthbench_eval-0.4.0.dist-info/RECORD +78 -0
- synthbench_eval-0.4.0.dist-info/WHEEL +5 -0
- synthbench_eval-0.4.0.dist-info/entry_points.txt +2 -0
- synthbench_eval-0.4.0.dist-info/licenses/LICENSE +21 -0
- synthbench_eval-0.4.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""Null-agent baseline floor discovery and drift detection (sb-lhoh).
|
|
2
|
+
|
|
3
|
+
Per Berkeley paper recommendation ("run a null agent; if it's not zero,
|
|
4
|
+
something is wrong"), SynthBench's equivalent null agents are the
|
|
5
|
+
``random-baseline`` and ``majority-baseline`` providers. Their composite
|
|
6
|
+
parity ("SPS") must stay bounded; upward drift on a stable dataset
|
|
7
|
+
signals a scoring-function bug, not a success.
|
|
8
|
+
|
|
9
|
+
This module discovers the canonical per-dataset baseline SPS from
|
|
10
|
+
``leaderboard-results/`` (mirroring leaderboard.build_baseline_scores:
|
|
11
|
+
max composite_parity per (provider, dataset)) and checks it against
|
|
12
|
+
configured ceilings.
|
|
13
|
+
|
|
14
|
+
Thresholds:
|
|
15
|
+
* ``MAJORITY_MAX_SPS = 0.85`` — Berkeley target. Current observed
|
|
16
|
+
max across datasets is ~0.71, leaving >0.13 headroom.
|
|
17
|
+
* ``RANDOM_MAX_SPS = 0.80`` — calibrated. Berkeley's aspirational
|
|
18
|
+
target is 0.70, but SynthBench's composite_parity weights
|
|
19
|
+
``p_refuse`` (where uniform-random naturally matches human DK/
|
|
20
|
+
Refused rates), pushing observed random SPS to ~0.71-0.76.
|
|
21
|
+
We set the hard CI gate at 0.80 (drift detection over current
|
|
22
|
+
max) and surface 0.70 as the published aspirational floor.
|
|
23
|
+
|
|
24
|
+
See ``docs/benchmark-hardening-analysis.md`` §5.4 for full context.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import json
|
|
30
|
+
from dataclasses import dataclass
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
|
|
33
|
+
BASELINE_PROVIDERS = ("random-baseline", "majority-baseline")
|
|
34
|
+
|
|
35
|
+
# Hard CI thresholds. A baseline run whose SPS meets-or-exceeds the
|
|
36
|
+
# threshold signals upward scoring-function drift and fails CI.
|
|
37
|
+
MAJORITY_MAX_SPS = 0.85
|
|
38
|
+
RANDOM_MAX_SPS = 0.80
|
|
39
|
+
|
|
40
|
+
# Aspirational floors from Wang et al. (UC Berkeley, 2026). Published on
|
|
41
|
+
# the methodology page as "the floor any benchmark-serious model must
|
|
42
|
+
# clear." The hard CI gate above is looser because SynthBench's
|
|
43
|
+
# composite_parity rewards random agents for matching human DK/Refused
|
|
44
|
+
# rates (high p_refuse) — a known property of the scoring protocol, not
|
|
45
|
+
# a bug. If p_refuse is rebalanced in future scoring revisions, the
|
|
46
|
+
# hard gate should drop toward these Berkeley targets.
|
|
47
|
+
ASPIRATIONAL_RANDOM_MAX_SPS = 0.70
|
|
48
|
+
ASPIRATIONAL_MAJORITY_MAX_SPS = 0.85
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass(frozen=True)
|
|
52
|
+
class BaselineRun:
|
|
53
|
+
"""One null-agent submission discovered under leaderboard-results/."""
|
|
54
|
+
|
|
55
|
+
provider: str
|
|
56
|
+
dataset: str
|
|
57
|
+
sps: float
|
|
58
|
+
n_evaluated: int
|
|
59
|
+
timestamp: str
|
|
60
|
+
source_file: str
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass(frozen=True)
|
|
64
|
+
class FloorViolation:
|
|
65
|
+
"""One canonical-baseline SPS that met-or-exceeded its ceiling."""
|
|
66
|
+
|
|
67
|
+
provider: str
|
|
68
|
+
dataset: str
|
|
69
|
+
sps: float
|
|
70
|
+
threshold: float
|
|
71
|
+
source_file: str
|
|
72
|
+
|
|
73
|
+
def format(self) -> str:
|
|
74
|
+
return (
|
|
75
|
+
f"{self.provider} on {self.dataset}: SPS={self.sps:.4f} "
|
|
76
|
+
f">= threshold {self.threshold:.2f} (source: {self.source_file})"
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _load_result(path: Path) -> dict | None:
|
|
81
|
+
try:
|
|
82
|
+
data = json.loads(path.read_text())
|
|
83
|
+
except (json.JSONDecodeError, OSError):
|
|
84
|
+
return None
|
|
85
|
+
if not isinstance(data, dict):
|
|
86
|
+
return None
|
|
87
|
+
cfg = data.get("config") or {}
|
|
88
|
+
if cfg.get("provider") not in BASELINE_PROVIDERS:
|
|
89
|
+
return None
|
|
90
|
+
agg = data.get("aggregate") or {}
|
|
91
|
+
scores = data.get("scores") or {}
|
|
92
|
+
sps = scores.get("sps")
|
|
93
|
+
if sps is None:
|
|
94
|
+
sps = agg.get("composite_parity")
|
|
95
|
+
if not isinstance(sps, (int, float)):
|
|
96
|
+
return None
|
|
97
|
+
return {
|
|
98
|
+
"provider": cfg["provider"],
|
|
99
|
+
"dataset": cfg.get("dataset", "unknown"),
|
|
100
|
+
"sps": float(sps),
|
|
101
|
+
"n_evaluated": int(cfg.get("n_evaluated") or agg.get("n_questions") or 0),
|
|
102
|
+
"timestamp": str(data.get("timestamp", "")),
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def discover_baseline_runs(results_dir: Path | str) -> list[BaselineRun]:
|
|
107
|
+
"""Enumerate every baseline submission under ``results_dir``.
|
|
108
|
+
|
|
109
|
+
Non-baseline submissions, malformed files, and files without an
|
|
110
|
+
SPS are silently skipped. Sorted by timestamp ascending to make
|
|
111
|
+
the output stable and usable as a drift log.
|
|
112
|
+
"""
|
|
113
|
+
root = Path(results_dir)
|
|
114
|
+
runs: list[BaselineRun] = []
|
|
115
|
+
for path in sorted(root.glob("*.json")):
|
|
116
|
+
rec = _load_result(path)
|
|
117
|
+
if rec is None:
|
|
118
|
+
continue
|
|
119
|
+
runs.append(
|
|
120
|
+
BaselineRun(
|
|
121
|
+
provider=rec["provider"],
|
|
122
|
+
dataset=rec["dataset"],
|
|
123
|
+
sps=rec["sps"],
|
|
124
|
+
n_evaluated=rec["n_evaluated"],
|
|
125
|
+
timestamp=rec["timestamp"],
|
|
126
|
+
source_file=path.name,
|
|
127
|
+
)
|
|
128
|
+
)
|
|
129
|
+
runs.sort(key=lambda r: (r.provider, r.dataset, r.timestamp))
|
|
130
|
+
return runs
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def canonical_baselines(
|
|
134
|
+
runs: list[BaselineRun],
|
|
135
|
+
) -> dict[tuple[str, str], BaselineRun]:
|
|
136
|
+
"""Pick the canonical (max-SPS) run per (provider, dataset).
|
|
137
|
+
|
|
138
|
+
Mirrors ``leaderboard.build_baseline_scores``: the value shown to
|
|
139
|
+
users as "the random baseline" on the leaderboard is the max
|
|
140
|
+
composite_parity across all runs for that provider/dataset. That
|
|
141
|
+
is the number a new submission must beat, so that is what we gate
|
|
142
|
+
on.
|
|
143
|
+
"""
|
|
144
|
+
best: dict[tuple[str, str], BaselineRun] = {}
|
|
145
|
+
for run in runs:
|
|
146
|
+
key = (run.provider, run.dataset)
|
|
147
|
+
if key not in best or run.sps > best[key].sps:
|
|
148
|
+
best[key] = run
|
|
149
|
+
return best
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def threshold_for(provider: str) -> float:
|
|
153
|
+
if provider == "random-baseline":
|
|
154
|
+
return RANDOM_MAX_SPS
|
|
155
|
+
if provider == "majority-baseline":
|
|
156
|
+
return MAJORITY_MAX_SPS
|
|
157
|
+
raise ValueError(f"unknown baseline provider: {provider!r}")
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def check_floors(
|
|
161
|
+
canonicals: dict[tuple[str, str], BaselineRun],
|
|
162
|
+
) -> list[FloorViolation]:
|
|
163
|
+
"""Return any canonical-baseline entry whose SPS >= its threshold."""
|
|
164
|
+
violations: list[FloorViolation] = []
|
|
165
|
+
for (provider, dataset), run in sorted(canonicals.items()):
|
|
166
|
+
t = threshold_for(provider)
|
|
167
|
+
if run.sps >= t:
|
|
168
|
+
violations.append(
|
|
169
|
+
FloorViolation(
|
|
170
|
+
provider=provider,
|
|
171
|
+
dataset=dataset,
|
|
172
|
+
sps=run.sps,
|
|
173
|
+
threshold=t,
|
|
174
|
+
source_file=run.source_file,
|
|
175
|
+
)
|
|
176
|
+
)
|
|
177
|
+
return violations
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def summary_report(
|
|
181
|
+
canonicals: dict[tuple[str, str], BaselineRun],
|
|
182
|
+
) -> str:
|
|
183
|
+
"""Render a human-readable floor summary. Used by tests and scripts."""
|
|
184
|
+
lines = ["Null-agent baseline floors (max SPS per provider/dataset):"]
|
|
185
|
+
by_provider: dict[str, list[BaselineRun]] = {}
|
|
186
|
+
for run in canonicals.values():
|
|
187
|
+
by_provider.setdefault(run.provider, []).append(run)
|
|
188
|
+
for provider in sorted(by_provider):
|
|
189
|
+
t = threshold_for(provider)
|
|
190
|
+
lines.append(f" {provider} (CI threshold: SPS < {t:.2f})")
|
|
191
|
+
for run in sorted(by_provider[provider], key=lambda r: r.dataset):
|
|
192
|
+
status = "OK " if run.sps < t else "FAIL"
|
|
193
|
+
lines.append(
|
|
194
|
+
f" [{status}] {run.dataset:20s} SPS={run.sps:.4f} "
|
|
195
|
+
f"n={run.n_evaluated:4d} ({run.source_file})"
|
|
196
|
+
)
|
|
197
|
+
return "\n".join(lines)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def history_report(runs: list[BaselineRun]) -> str:
|
|
201
|
+
"""Render every observed baseline run in time order.
|
|
202
|
+
|
|
203
|
+
This is the persisted drift log: every baseline submission ever
|
|
204
|
+
checked into leaderboard-results/ is listed with its SPS. Grep
|
|
205
|
+
this output to watch a specific (provider, dataset) over time.
|
|
206
|
+
"""
|
|
207
|
+
lines = ["Null-agent baseline SPS history (ascending by time):"]
|
|
208
|
+
for run in sorted(runs, key=lambda r: (r.provider, r.dataset, r.timestamp)):
|
|
209
|
+
lines.append(
|
|
210
|
+
f" {run.timestamp:30s} {run.provider:20s} "
|
|
211
|
+
f"{run.dataset:20s} SPS={run.sps:.4f} n={run.n_evaluated:4d}"
|
|
212
|
+
)
|
|
213
|
+
return "\n".join(lines)
|