synthbench-eval 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. synthbench/__init__.py +16 -0
  2. synthbench/__main__.py +5 -0
  3. synthbench/adapter.py +131 -0
  4. synthbench/anomaly.py +503 -0
  5. synthbench/baseline_floors.py +213 -0
  6. synthbench/baselines.py +902 -0
  7. synthbench/cli.py +2929 -0
  8. synthbench/config_id.py +424 -0
  9. synthbench/contamination.py +783 -0
  10. synthbench/convergence/__init__.py +67 -0
  11. synthbench/convergence/baseline.py +172 -0
  12. synthbench/convergence/bootstrap.py +60 -0
  13. synthbench/convergence/cli_report.py +556 -0
  14. synthbench/convergence/curves.py +111 -0
  15. synthbench/convergence/real_sampling.py +146 -0
  16. synthbench/convergence/thresholds.py +49 -0
  17. synthbench/datasets/__init__.py +37 -0
  18. synthbench/datasets/base.py +147 -0
  19. synthbench/datasets/eurobarometer.py +335 -0
  20. synthbench/datasets/globalopinionqa.py +252 -0
  21. synthbench/datasets/gss.py +323 -0
  22. synthbench/datasets/michigan.py +505 -0
  23. synthbench/datasets/ntia.py +357 -0
  24. synthbench/datasets/opinionsqa.py +412 -0
  25. synthbench/datasets/pewtech.py +334 -0
  26. synthbench/datasets/policy.py +142 -0
  27. synthbench/datasets/subpop.py +302 -0
  28. synthbench/datasets/wvs.py +229 -0
  29. synthbench/findings.py +988 -0
  30. synthbench/holdout.py +347 -0
  31. synthbench/human_distributions.py +243 -0
  32. synthbench/leaderboard.py +715 -0
  33. synthbench/leaderboard_pr.py +274 -0
  34. synthbench/metrics/__init__.py +38 -0
  35. synthbench/metrics/composite.py +72 -0
  36. synthbench/metrics/conditioning.py +39 -0
  37. synthbench/metrics/distributional.py +41 -0
  38. synthbench/metrics/ranking.py +37 -0
  39. synthbench/metrics/refusal.py +270 -0
  40. synthbench/metrics/subgroup.py +58 -0
  41. synthbench/private_holdout.py +240 -0
  42. synthbench/providers/__init__.py +43 -0
  43. synthbench/providers/_parsing.py +150 -0
  44. synthbench/providers/_retry.py +107 -0
  45. synthbench/providers/base.py +212 -0
  46. synthbench/providers/http.py +108 -0
  47. synthbench/providers/majority_baseline.py +23 -0
  48. synthbench/providers/ollama.py +109 -0
  49. synthbench/providers/openrouter.py +144 -0
  50. synthbench/providers/population_baseline.py +87 -0
  51. synthbench/providers/random_baseline.py +24 -0
  52. synthbench/providers/raw_anthropic.py +149 -0
  53. synthbench/providers/raw_gemini.py +145 -0
  54. synthbench/providers/raw_openai.py +139 -0
  55. synthbench/providers/synthpanel.py +978 -0
  56. synthbench/publish.py +2772 -0
  57. synthbench/r2_upload.py +178 -0
  58. synthbench/recompute.py +235 -0
  59. synthbench/report.py +493 -0
  60. synthbench/run_hash.py +111 -0
  61. synthbench/run_validity.py +232 -0
  62. synthbench/runner.py +834 -0
  63. synthbench/stats.py +1492 -0
  64. synthbench/submission.py +284 -0
  65. synthbench/submission_pr.py +436 -0
  66. synthbench/submit_adapter.py +748 -0
  67. synthbench/suite.py +424 -0
  68. synthbench/suites/__init__.py +94 -0
  69. synthbench/topics.py +256 -0
  70. synthbench/user_config.py +409 -0
  71. synthbench/validation.py +1556 -0
  72. synthbench/visualize.py +455 -0
  73. synthbench_eval-0.4.0.dist-info/METADATA +279 -0
  74. synthbench_eval-0.4.0.dist-info/RECORD +78 -0
  75. synthbench_eval-0.4.0.dist-info/WHEEL +5 -0
  76. synthbench_eval-0.4.0.dist-info/entry_points.txt +2 -0
  77. synthbench_eval-0.4.0.dist-info/licenses/LICENSE +21 -0
  78. synthbench_eval-0.4.0.dist-info/top_level.txt +1 -0
synthbench/__init__.py ADDED
@@ -0,0 +1,16 @@
1
+ """SynthBench — open benchmark harness for synthetic survey respondent quality."""
2
+
3
+ from synthbench.convergence.baseline import (
4
+ BaselineGatedError,
5
+ BaselineUnavailable,
6
+ load_convergence_baseline,
7
+ )
8
+
9
+ __version__ = "0.4.0"
10
+
11
+ __all__ = [
12
+ "BaselineGatedError",
13
+ "BaselineUnavailable",
14
+ "load_convergence_baseline",
15
+ "__version__",
16
+ ]
synthbench/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ """Allow running as `python -m synthbench`."""
2
+
3
+ from synthbench.cli import main
4
+
5
+ main()
synthbench/adapter.py ADDED
@@ -0,0 +1,131 @@
1
+ """Vendor-adapter interface for `synthbench submit-adapter` (refs #256).
2
+
3
+ The adapter is the contract between SynthBench and a third-party synthetic-
4
+ respondent vendor. Vendors implement a subclass of :class:`Adapter` in their
5
+ own Python module, then point `synthbench submit-adapter --adapter` at the
6
+ module path. The submit pipeline imports the module, instantiates the
7
+ adapter, runs the standard suite against it, and produces a submission
8
+ artifact suitable for opening a PR against the leaderboard repo.
9
+
10
+ Adapters are intentionally narrow: vendors expose **one** async method
11
+ (:meth:`Adapter.respond`) that returns a string answer to a single question
12
+ under a single persona condition. Everything upstream of that — distribution
13
+ estimation, scoring, run-hash content addressing — is owned by SynthBench so
14
+ that vendor implementations stay simple and the evaluation surface stays
15
+ honest.
16
+
17
+ NOTE (refs #256): the full evaluation pipeline that drives `Adapter.respond`
18
+ through the `core` suite is still scaffold-only as of this PR. The interface
19
+ is frozen here so vendors can start writing adapters in parallel with the
20
+ pipeline build-out tracked in follow-up issues.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import random
26
+ from abc import ABC, abstractmethod
27
+ from typing import Any
28
+
29
+
30
+ class Adapter(ABC):
31
+ """Base class every vendor adapter must subclass.
32
+
33
+ Implementations should be:
34
+ * **Stateless across questions** — the harness may call ``respond``
35
+ from many coroutines concurrently. Any per-instance state must be
36
+ safe under ``asyncio.gather``.
37
+ * **Side-effect free** — no writing to disk, no global mutation. The
38
+ harness is responsible for persistence.
39
+ * **Deterministic given the same inputs**, where possible. Sampling
40
+ randomness is expected (models are stochastic); orchestration
41
+ randomness is not.
42
+
43
+ Two metadata properties are required so the harness can stamp the run
44
+ artifact with vendor identity without re-reading CLI flags:
45
+ """
46
+
47
+ @property
48
+ @abstractmethod
49
+ def name(self) -> str:
50
+ """Human-readable vendor/model identifier (e.g. ``"acme/gpt-4o"``).
51
+
52
+ This appears in the leaderboard row and the submission filename, so
53
+ prefer short, stable strings. Avoid embedding timestamps or git
54
+ SHAs here — use :attr:`version` for that.
55
+ """
56
+
57
+ @property
58
+ @abstractmethod
59
+ def version(self) -> str:
60
+ """Adapter version string (e.g. ``"2026.05.14"`` or a semver tag).
61
+
62
+ Bumped whenever the prompt, sampling parameters, or model snapshot
63
+ materially changes. Two runs with the same ``(name, version)`` are
64
+ expected to be comparable; differing versions are not.
65
+ """
66
+
67
+ @abstractmethod
68
+ async def respond(
69
+ self,
70
+ *,
71
+ question: str,
72
+ persona: dict[str, Any],
73
+ context: dict[str, Any] | None = None,
74
+ ) -> str:
75
+ """Answer a single question under a single persona condition.
76
+
77
+ Args:
78
+ question: The prompt the harness wants a respondent answer to.
79
+ Already includes options where applicable; the adapter
80
+ should not re-format it.
81
+ persona: Demographic + attitudinal conditioning for this
82
+ respondent. Keys follow the
83
+ :class:`synthbench.providers.base.PersonaSpec` shape (e.g.
84
+ ``{"age": "30-44", "party": "Democrat", ...}``). May be
85
+ empty if the suite calls for unconditioned answers.
86
+ context: Optional suite-specific metadata (e.g. question id,
87
+ survey wave). Adapters should ignore unknown keys.
88
+
89
+ Returns:
90
+ The raw answer string. The harness handles parsing this into
91
+ an option index or Likert number, so adapters should return
92
+ whatever shape feels native to their model (a single token, a
93
+ short phrase, or a number).
94
+ """
95
+
96
+
97
+ class RandomAdapter(Adapter):
98
+ """Reference adapter that emits trivial answers, for wiring up tests.
99
+
100
+ Returns ``"yes"`` / ``"no"`` for boolean-flavored questions and a
101
+ uniformly random Likert integer (1-5) otherwise. Seeded so the test
102
+ suite stays deterministic.
103
+
104
+ This adapter exists so vendors (and SynthBench CI) have a known-good
105
+ target to point ``synthbench submit-adapter --adapter`` at while
106
+ debugging their setup. It is **not** a baseline — its scores are
107
+ expected to be terrible.
108
+ """
109
+
110
+ def __init__(self, seed: int = 42) -> None:
111
+ self._rng = random.Random(seed)
112
+
113
+ @property
114
+ def name(self) -> str:
115
+ return "synthbench/random-adapter"
116
+
117
+ @property
118
+ def version(self) -> str:
119
+ return "0.1.0"
120
+
121
+ async def respond(
122
+ self,
123
+ *,
124
+ question: str,
125
+ persona: dict[str, Any],
126
+ context: dict[str, Any] | None = None,
127
+ ) -> str:
128
+ q = question.lower()
129
+ if "yes or no" in q or q.strip().endswith("?") and "agree" not in q:
130
+ return self._rng.choice(["yes", "no"])
131
+ return str(self._rng.randint(1, 5))
synthbench/anomaly.py ADDED
@@ -0,0 +1,503 @@
1
+ """Tier-3 statistical anomaly detection for SynthBench submissions.
2
+
3
+ Tier 1 (schema) and tier 2 (arithmetic recomputation) both trust the
4
+ per-question distributions the submitter attached. A sufficiently careful
5
+ fabricator can defeat them by reverse-engineering distributions that are
6
+ arithmetically self-consistent. Tier 3 adds cheap statistical plausibility
7
+ checks that catch the most common lazy attacks: "copy the answer key",
8
+ "zero-out refusals", "match one peer run exactly".
9
+
10
+ Each detector returns an :class:`~synthbench.validation.Issue` with
11
+ ``severity = WARNING``. We deliberately keep tier 3 soft in its first
12
+ iteration — once we've observed the distribution of legitimate
13
+ submissions in practice, specific detectors can be graduated to ERROR.
14
+
15
+ Detectors:
16
+
17
+ * ``check_suspicious_perfection`` — per-question JSD near zero with near-zero
18
+ variance implies the submitter copied the human distribution verbatim.
19
+ * ``check_missing_refusals`` — a submission with model_refusal_rate ≡ 0
20
+ across a dataset where real humans refuse is suspicious: real LLMs do
21
+ refuse, and refusal-free runs are almost always fabricated or
22
+ miscalibrated post-processing.
23
+ * ``check_peer_distribution_outlier`` — soft signal comparing claimed
24
+ per-question distributions against peer submissions on the same family.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import math
30
+ from typing import Any, Iterable, Mapping, Sequence
31
+
32
+ from synthbench.private_holdout import is_holdout_enabled, is_private_holdout
33
+ from synthbench.validation import Issue, Severity
34
+
35
+
36
+ # Detector thresholds. Tuned against real submissions in leaderboard-results/
37
+ # where mean_jsd lives in ~[0.05, 0.55] with std ~[0.10, 0.30]. Fabrication
38
+ # that copies the answer key produces mean ~0 with std ~0; the threshold sits
39
+ # an order of magnitude below the noise floor of any real run we've seen.
40
+ SUSPICIOUS_MEAN_JSD = 0.005
41
+ SUSPICIOUS_STD_JSD = 0.005
42
+
43
+ # Minimum JSD sample size at which ANOMALY_PERFECTION is promoted to ERROR.
44
+ # Below this, the detector stays WARNING so small debug/test fixtures are not
45
+ # hard-rejected. At n >= 25 the thresholds are so far below the real-run noise
46
+ # floor (mean_jsd in [0.05, 0.55], std_jsd in [0.10, 0.30]) that tripping them
47
+ # is a statistically reliable fabrication signal — see Wang et al. (Berkeley,
48
+ # 2026) and docs/benchmark-hardening-analysis.md §2.
49
+ ANOMALY_PERFECTION_ERROR_MIN_N = 25
50
+
51
+ # Share of dataset questions that must have a human refusal rate above this
52
+ # cutoff before a zero-refusal submission is flagged. We require at least
53
+ # one visibly-refusing question in the dataset to avoid false positives on
54
+ # datasets where humans rarely refuse.
55
+ HUMAN_REFUSAL_CUTOFF = 0.05
56
+ HUMAN_REFUSAL_MIN_QUESTIONS = 3
57
+
58
+ # Peer-outlier detector: two submissions are "same-family" if they share
59
+ # the model family (e.g. both claim claude-haiku-4-5) and the dataset.
60
+ # We flag when the mean absolute delta between submission and peer JSD
61
+ # on overlapping questions exceeds :data:`PEER_OUTLIER_DELTA`. Real
62
+ # same-family runs differ by <=~0.05 on average; 0.15 is well outside
63
+ # observed noise and inside the range produced by answer-key attacks.
64
+ PEER_OUTLIER_DELTA = 0.15
65
+ PEER_MIN_OVERLAP = 5
66
+
67
+ # Near-copy-public detector: the attack is "scrape the public human
68
+ # distribution, add tiny noise, submit." The fingerprint is per-question JSD
69
+ # on the public subset sitting far below what any real LLM can produce.
70
+ # Empirical floor from leaderboard-results/: the best real submission's
71
+ # public-subset mean_jsd ~0.09 with std ~0.08 (claude-sonnet on
72
+ # globalopinionqa). A noise-floor attack with ε ~ U(-0.02, 0.02) yields
73
+ # mean ~0.006 and std ~0.004. Thresholds 0.02 / 0.03 leave ~4x headroom on
74
+ # both sides. Minimum n_public=50 keeps small debug fixtures from tripping.
75
+ NEAR_COPY_MEAN_JSD = 0.02
76
+ NEAR_COPY_STD_JSD = 0.03
77
+ NEAR_COPY_MIN_PUBLIC = 50
78
+
79
+ # Constant-offset fingerprint. A deterministic monotonic transformation of
80
+ # the human distribution (``model = normalize(human + c)`` is the canonical
81
+ # example from docs/benchmark-hardening-analysis.md §3.3; scaling and power
82
+ # transforms produce the same shape) preserves rank order exactly on every
83
+ # question — per-question ``kendall_tau == 1.0`` uniformly. Real LLM
84
+ # sampling flips minor ranks on nearly-tied options: across every real
85
+ # submission in leaderboard-results/ the share of questions with perfect
86
+ # tau tops out around 26% (claude-haiku, ensemble) and is typically 10–20%.
87
+ # A 95% threshold over n >= 25 sits >3x above the real-run ceiling and
88
+ # catches the fabricated ``constant_offset.json`` fixture (100% perfect tau).
89
+ CONSTANT_OFFSET_PERFECT_TAU_FRACTION = 0.95
90
+ CONSTANT_OFFSET_TAU_EPSILON = 1e-4
91
+ CONSTANT_OFFSET_MIN_N = 25
92
+
93
+
94
+ def _mean(values: Sequence[float]) -> float:
95
+ return sum(values) / len(values) if values else 0.0
96
+
97
+
98
+ def _std(values: Sequence[float]) -> float:
99
+ if len(values) < 2:
100
+ return 0.0
101
+ mu = _mean(values)
102
+ variance = sum((v - mu) ** 2 for v in values) / (len(values) - 1)
103
+ return math.sqrt(variance)
104
+
105
+
106
+ def check_suspicious_perfection(
107
+ per_question: Sequence[Mapping[str, Any]],
108
+ ) -> Issue | None:
109
+ """Flag submissions whose per-question JSD is implausibly perfect.
110
+
111
+ Real LLMs produce per-question JSD with a non-trivial spread. A
112
+ submission where either the mean OR the standard deviation of JSD
113
+ sits below :data:`SUSPICIOUS_MEAN_JSD` / :data:`SUSPICIOUS_STD_JSD`
114
+ is almost certainly copied from the answer key. We use OR because
115
+ either condition alone is enough: mean-near-zero implies uniformly
116
+ near-perfect matches (impossible for a real model), and std-near-zero
117
+ implies the per-question distances are all the same, which a real
118
+ sampling pipeline cannot produce.
119
+
120
+ Severity is ERROR when the JSD sample size reaches
121
+ :data:`ANOMALY_PERFECTION_ERROR_MIN_N`; smaller samples stay WARNING
122
+ so debug fixtures aren't hard-rejected on borderline numerics.
123
+ """
124
+ jsd_values = [
125
+ float(q["jsd"]) for q in per_question if isinstance(q.get("jsd"), (int, float))
126
+ ]
127
+ if len(jsd_values) < 5:
128
+ return None
129
+
130
+ mean_jsd = _mean(jsd_values)
131
+ std_jsd = _std(jsd_values)
132
+
133
+ if mean_jsd < SUSPICIOUS_MEAN_JSD or std_jsd < SUSPICIOUS_STD_JSD:
134
+ severity = (
135
+ Severity.ERROR
136
+ if len(jsd_values) >= ANOMALY_PERFECTION_ERROR_MIN_N
137
+ else Severity.WARNING
138
+ )
139
+ return Issue(
140
+ code="ANOMALY_PERFECTION",
141
+ severity=severity,
142
+ message=(
143
+ f"per-question JSD has mean={mean_jsd:.6f}, "
144
+ f"std={std_jsd:.6f} across n={len(jsd_values)} questions "
145
+ f"— implausibly perfect match to human distribution "
146
+ f"(thresholds: mean<{SUSPICIOUS_MEAN_JSD}, "
147
+ f"std<{SUSPICIOUS_STD_JSD}). Likely answer-key copy."
148
+ ),
149
+ path="per_question",
150
+ )
151
+ return None
152
+
153
+
154
+ def check_missing_refusals(
155
+ per_question: Sequence[Mapping[str, Any]],
156
+ ) -> Issue | None:
157
+ """Flag submissions that never refuse despite humans refusing.
158
+
159
+ Some datasets have questions humans refuse on (e.g. political
160
+ identification, touchy religious questions). Real LLMs refuse those
161
+ too — sometimes at similar rates, sometimes higher. A submission
162
+ that reports ``model_refusal_rate = 0`` across every question in a
163
+ dataset where humans clearly refuse is either fabricated or has a
164
+ silent bug that drops refusals. Either way, worth surfacing.
165
+ """
166
+ pq_list = [q for q in per_question if isinstance(q, dict)]
167
+ if not pq_list:
168
+ return None
169
+
170
+ refusing_questions = [
171
+ q
172
+ for q in pq_list
173
+ if isinstance(q.get("human_refusal_rate"), (int, float))
174
+ and float(q["human_refusal_rate"]) >= HUMAN_REFUSAL_CUTOFF
175
+ ]
176
+ if len(refusing_questions) < HUMAN_REFUSAL_MIN_QUESTIONS:
177
+ # Dataset doesn't have enough "refuseable" questions to reason about.
178
+ return None
179
+
180
+ model_refusal_values = [float(q.get("model_refusal_rate") or 0.0) for q in pq_list]
181
+ if any(v > 0.0 for v in model_refusal_values):
182
+ return None
183
+
184
+ return Issue(
185
+ code="ANOMALY_NO_REFUSAL",
186
+ severity=Severity.WARNING,
187
+ message=(
188
+ f"submission reports model_refusal_rate=0 on every question, "
189
+ f"but {len(refusing_questions)} of {len(pq_list)} questions "
190
+ f"have human_refusal_rate >= {HUMAN_REFUSAL_CUTOFF}. Real "
191
+ f"LLMs refuse sometimes — check for fabricated or post-"
192
+ f"processed refusals."
193
+ ),
194
+ path="per_question",
195
+ )
196
+
197
+
198
+ def _same_family(submission_provider: str, peer_provider: str) -> bool:
199
+ """True when two provider strings plausibly share a model family.
200
+
201
+ Heuristic: take the last path segment (usually the model id) and
202
+ compare. ``openrouter/anthropic/claude-haiku-4-5`` and
203
+ ``anthropic/claude-haiku-4-5`` both resolve to ``claude-haiku-4-5``.
204
+ Baselines and ensembles are never "same-family".
205
+ """
206
+ if not submission_provider or not peer_provider:
207
+ return False
208
+ if "baseline" in submission_provider or "baseline" in peer_provider:
209
+ return False
210
+ if submission_provider.startswith("ensemble/") or peer_provider.startswith(
211
+ "ensemble/"
212
+ ):
213
+ return False
214
+
215
+ def _model_tail(name: str) -> str:
216
+ return name.rsplit("/", 1)[-1].lower()
217
+
218
+ return _model_tail(submission_provider) == _model_tail(peer_provider)
219
+
220
+
221
+ def check_peer_distribution_outlier(
222
+ submission: Mapping[str, Any],
223
+ peers: Iterable[Mapping[str, Any]],
224
+ ) -> Issue | None:
225
+ """Soft check: claimed distribution shapes vs same-family peers.
226
+
227
+ If the submission claims model X but its per-question JSD on
228
+ overlapping questions is wildly out of line with other runs of
229
+ model X, flag it. This catches submissions that claim a particular
230
+ model but were actually generated by something else.
231
+
232
+ Not a hard reject — honest runs differ from each other (temperature,
233
+ prompt, seed). We only flag when the deviation exceeds
234
+ :data:`PEER_OUTLIER_SIGMA` standard deviations of the peer JSD
235
+ distribution on the shared question set. Returns ``None`` when
236
+ there are no same-family peers or insufficient overlap.
237
+ """
238
+ config = submission.get("config") or {}
239
+ submission_provider = str(config.get("provider", ""))
240
+ dataset = config.get("dataset")
241
+ per_question = submission.get("per_question") or []
242
+
243
+ if not submission_provider or not dataset or not per_question:
244
+ return None
245
+
246
+ submission_jsd_by_key: dict[str, float] = {}
247
+ for q in per_question:
248
+ if not isinstance(q, dict):
249
+ continue
250
+ key = q.get("key")
251
+ jsd = q.get("jsd")
252
+ if isinstance(key, str) and isinstance(jsd, (int, float)):
253
+ submission_jsd_by_key[key] = float(jsd)
254
+
255
+ # Collect peer JSD maps on same model family + same dataset.
256
+ peer_jsd_maps: list[dict[str, float]] = []
257
+ for peer in peers:
258
+ if not isinstance(peer, dict):
259
+ continue
260
+ peer_config = peer.get("config") or {}
261
+ peer_provider = str(peer_config.get("provider", ""))
262
+ peer_dataset = peer_config.get("dataset")
263
+ if peer_dataset != dataset or not _same_family(
264
+ submission_provider, peer_provider
265
+ ):
266
+ continue
267
+
268
+ peer_jsd: dict[str, float] = {}
269
+ for q in peer.get("per_question") or []:
270
+ if not isinstance(q, dict):
271
+ continue
272
+ key = q.get("key")
273
+ jsd = q.get("jsd")
274
+ if isinstance(key, str) and isinstance(jsd, (int, float)):
275
+ peer_jsd[key] = float(jsd)
276
+ if peer_jsd:
277
+ peer_jsd_maps.append(peer_jsd)
278
+
279
+ if not peer_jsd_maps:
280
+ return None
281
+
282
+ # Compute per-key peer mean JSD using every peer that covered that key.
283
+ per_key_peer_values: dict[str, list[float]] = {}
284
+ for peer_map in peer_jsd_maps:
285
+ for key, jsd in peer_map.items():
286
+ if key in submission_jsd_by_key:
287
+ per_key_peer_values.setdefault(key, []).append(jsd)
288
+
289
+ overlap = [
290
+ (submission_jsd_by_key[k], _mean(vals))
291
+ for k, vals in per_key_peer_values.items()
292
+ if vals
293
+ ]
294
+ if len(overlap) < PEER_MIN_OVERLAP:
295
+ return None
296
+
297
+ deltas = [sub - peer_mean for sub, peer_mean in overlap]
298
+ mu = _mean(deltas)
299
+ if abs(mu) < PEER_OUTLIER_DELTA:
300
+ return None
301
+
302
+ direction = "lower" if mu < 0 else "higher"
303
+ return Issue(
304
+ code="ANOMALY_PEER_OUTLIER",
305
+ severity=Severity.WARNING,
306
+ message=(
307
+ f"per-question JSD runs {direction} than same-family peers "
308
+ f"({len(overlap)} shared questions, mean delta={mu:.4f} > "
309
+ f"threshold {PEER_OUTLIER_DELTA}). Investigate whether the "
310
+ f"claimed model matches the submission."
311
+ ),
312
+ path="per_question",
313
+ )
314
+
315
+
316
+ def check_near_copy_public(
317
+ data: Mapping[str, Any],
318
+ ) -> Issue | None:
319
+ """Flag submissions whose public-subset JSD is implausibly near-zero.
320
+
321
+ The attack this closes: scrape published ``human_distribution`` values
322
+ from the public leaderboard, add small noise, submit as
323
+ ``model_distribution``. Every tier-1/2 check passes (distributions are
324
+ valid, aggregates reconcile), and ``ANOMALY_PERFECTION`` can miss it
325
+ because the attacker's fabricated private rows dilute the full-dataset
326
+ JSD. Restricting the statistic to the public subset removes that
327
+ dilution — the attack's fingerprint is uniformly tight JSD across
328
+ every public question.
329
+
330
+ Fires only on holdout-enabled datasets. When the dataset is not
331
+ partitioned, we have no principled public/private split to restrict
332
+ to. Returns ``None`` below :data:`NEAR_COPY_MIN_PUBLIC` public rows.
333
+
334
+ The detector requires BOTH mean < :data:`NEAR_COPY_MEAN_JSD` AND std <
335
+ :data:`NEAR_COPY_STD_JSD` to flag. Unlike ``ANOMALY_PERFECTION`` (OR),
336
+ this AND guard reduces false positives on genuinely strong real
337
+ models — a real model can have either tight variance or a low mean,
338
+ but not both simultaneously on 50+ public questions, in every run
339
+ we've observed.
340
+ """
341
+ config = data.get("config") or {}
342
+ dataset = config.get("dataset")
343
+ if not isinstance(dataset, str) or not is_holdout_enabled(dataset):
344
+ return None
345
+
346
+ per_question = data.get("per_question") or []
347
+ if not isinstance(per_question, list):
348
+ return None
349
+
350
+ public_jsd: list[float] = []
351
+ for q in per_question:
352
+ if not isinstance(q, dict):
353
+ continue
354
+ key = q.get("key")
355
+ jsd = q.get("jsd")
356
+ if not isinstance(key, str) or not isinstance(jsd, (int, float)):
357
+ continue
358
+ if is_private_holdout(dataset, key):
359
+ continue
360
+ public_jsd.append(float(jsd))
361
+
362
+ if len(public_jsd) < NEAR_COPY_MIN_PUBLIC:
363
+ return None
364
+
365
+ mean_jsd = _mean(public_jsd)
366
+ std_jsd = _std(public_jsd)
367
+
368
+ if mean_jsd >= NEAR_COPY_MEAN_JSD or std_jsd >= NEAR_COPY_STD_JSD:
369
+ return None
370
+
371
+ return Issue(
372
+ code="ANOMALY_NEAR_COPY_PUBLIC",
373
+ severity=Severity.ERROR,
374
+ message=(
375
+ f"public-subset JSD has mean={mean_jsd:.6f}, std={std_jsd:.6f} "
376
+ f"over {len(public_jsd)} public questions (thresholds: "
377
+ f"mean<{NEAR_COPY_MEAN_JSD}, std<{NEAR_COPY_STD_JSD}). This is "
378
+ f"the fingerprint of a submission that copied the published "
379
+ f"human_distribution and added small noise. Real LLM runs "
380
+ f"produce public mean_jsd >= ~0.09 on every holdout dataset "
381
+ f"we have on file."
382
+ ),
383
+ path="per_question",
384
+ )
385
+
386
+
387
+ def check_constant_offset(
388
+ per_question: Sequence[Mapping[str, Any]],
389
+ ) -> Issue | None:
390
+ """Flag deterministic monotonic-transformation attacks.
391
+
392
+ The canonical attack from ``docs/benchmark-hardening-analysis.md §3.3``
393
+ is ``model[opt] = human[opt] + c`` renormalized. Because adding a
394
+ constant to every option is a strictly monotonic transformation, the
395
+ option rank order is preserved exactly on every question — so
396
+ per-question ``kendall_tau`` equals 1.0 uniformly. The same fingerprint
397
+ appears for any deterministic monotonic transformation the attacker
398
+ might apply to the answer key (scaling, power transform, etc.).
399
+
400
+ Real LLM sampling flips minor rank orders on nearly-tied options: across
401
+ the full leaderboard the share of questions with ``tau == 1.0`` tops
402
+ out near 26% and sits at 10–20% for typical strong models. A
403
+ submission with ``>= 95%`` of questions at near-unity tau over n >= 25
404
+ is therefore not a plausible real run.
405
+
406
+ The detector is independent of :func:`check_suspicious_perfection` and
407
+ :func:`check_near_copy_public` — those fire on distribution-distance
408
+ signals, which a cleverly-tuned multiplicative/offset attack could
409
+ drive out of range by choosing a large constant. The tau fingerprint
410
+ survives any monotonic f, which is why it's the right cross-check.
411
+ """
412
+ tau_values = [
413
+ float(q["kendall_tau"])
414
+ for q in per_question
415
+ if isinstance(q.get("kendall_tau"), (int, float))
416
+ ]
417
+ if len(tau_values) < CONSTANT_OFFSET_MIN_N:
418
+ return None
419
+
420
+ perfect = sum(1 for t in tau_values if t >= 1.0 - CONSTANT_OFFSET_TAU_EPSILON)
421
+ fraction = perfect / len(tau_values)
422
+ if fraction < CONSTANT_OFFSET_PERFECT_TAU_FRACTION:
423
+ return None
424
+
425
+ return Issue(
426
+ code="ANOMALY_CONSTANT_OFFSET",
427
+ severity=Severity.ERROR,
428
+ message=(
429
+ f"per-question kendall_tau is near-unity on "
430
+ f"{perfect}/{len(tau_values)} questions ({fraction:.1%}, "
431
+ f"threshold >= {CONSTANT_OFFSET_PERFECT_TAU_FRACTION:.0%}). "
432
+ f"Real LLM sampling flips minor rank orders on nearly-tied "
433
+ f"options — uniform rank preservation is the fingerprint of a "
434
+ f"deterministic monotonic transformation of the human "
435
+ f"distribution (e.g. model = normalize(human + c)). See "
436
+ f"docs/benchmark-hardening-analysis.md §3.3."
437
+ ),
438
+ path="per_question",
439
+ )
440
+
441
+
442
+ def tier3_checks(
443
+ data: Mapping[str, Any],
444
+ *,
445
+ peers: Iterable[Mapping[str, Any]] = (),
446
+ ) -> list[Issue]:
447
+ """Run every tier-3 anomaly detector and return the list of issues.
448
+
449
+ Note: ``check_missing_refusals`` is intentionally *not* called from
450
+ the default dispatch. Every current provider prompt in
451
+ ``src/synthbench/providers/`` ends with ``"Respond with ONLY the
452
+ letter of your choice"`` and gives the model no way to refuse — the
453
+ harness architecturally produces ``model_refusal_rate == 0`` on
454
+ every question, so the detector flagged every legitimate submission
455
+ (sb-a613). The function is kept exported so callers can invoke it
456
+ directly once a refusal-capable prompt variant exists.
457
+ """
458
+ per_question = data.get("per_question") or []
459
+ if not isinstance(per_question, list):
460
+ return []
461
+
462
+ issues: list[Issue] = []
463
+
464
+ perfection = check_suspicious_perfection(per_question)
465
+ if perfection is not None:
466
+ issues.append(perfection)
467
+
468
+ near_copy = check_near_copy_public(data)
469
+ if near_copy is not None:
470
+ issues.append(near_copy)
471
+
472
+ constant_offset = check_constant_offset(per_question)
473
+ if constant_offset is not None:
474
+ issues.append(constant_offset)
475
+
476
+ peer_outlier = check_peer_distribution_outlier(data, peers)
477
+ if peer_outlier is not None:
478
+ issues.append(peer_outlier)
479
+
480
+ return issues
481
+
482
+
483
+ __all__ = [
484
+ "SUSPICIOUS_MEAN_JSD",
485
+ "SUSPICIOUS_STD_JSD",
486
+ "ANOMALY_PERFECTION_ERROR_MIN_N",
487
+ "HUMAN_REFUSAL_CUTOFF",
488
+ "HUMAN_REFUSAL_MIN_QUESTIONS",
489
+ "PEER_OUTLIER_DELTA",
490
+ "PEER_MIN_OVERLAP",
491
+ "NEAR_COPY_MEAN_JSD",
492
+ "NEAR_COPY_STD_JSD",
493
+ "NEAR_COPY_MIN_PUBLIC",
494
+ "CONSTANT_OFFSET_PERFECT_TAU_FRACTION",
495
+ "CONSTANT_OFFSET_TAU_EPSILON",
496
+ "CONSTANT_OFFSET_MIN_N",
497
+ "check_suspicious_perfection",
498
+ "check_missing_refusals",
499
+ "check_peer_distribution_outlier",
500
+ "check_near_copy_public",
501
+ "check_constant_offset",
502
+ "tier3_checks",
503
+ ]