synthbench-eval 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- synthbench/__init__.py +16 -0
- synthbench/__main__.py +5 -0
- synthbench/adapter.py +131 -0
- synthbench/anomaly.py +503 -0
- synthbench/baseline_floors.py +213 -0
- synthbench/baselines.py +902 -0
- synthbench/cli.py +2929 -0
- synthbench/config_id.py +424 -0
- synthbench/contamination.py +783 -0
- synthbench/convergence/__init__.py +67 -0
- synthbench/convergence/baseline.py +172 -0
- synthbench/convergence/bootstrap.py +60 -0
- synthbench/convergence/cli_report.py +556 -0
- synthbench/convergence/curves.py +111 -0
- synthbench/convergence/real_sampling.py +146 -0
- synthbench/convergence/thresholds.py +49 -0
- synthbench/datasets/__init__.py +37 -0
- synthbench/datasets/base.py +147 -0
- synthbench/datasets/eurobarometer.py +335 -0
- synthbench/datasets/globalopinionqa.py +252 -0
- synthbench/datasets/gss.py +323 -0
- synthbench/datasets/michigan.py +505 -0
- synthbench/datasets/ntia.py +357 -0
- synthbench/datasets/opinionsqa.py +412 -0
- synthbench/datasets/pewtech.py +334 -0
- synthbench/datasets/policy.py +142 -0
- synthbench/datasets/subpop.py +302 -0
- synthbench/datasets/wvs.py +229 -0
- synthbench/findings.py +988 -0
- synthbench/holdout.py +347 -0
- synthbench/human_distributions.py +243 -0
- synthbench/leaderboard.py +715 -0
- synthbench/leaderboard_pr.py +274 -0
- synthbench/metrics/__init__.py +38 -0
- synthbench/metrics/composite.py +72 -0
- synthbench/metrics/conditioning.py +39 -0
- synthbench/metrics/distributional.py +41 -0
- synthbench/metrics/ranking.py +37 -0
- synthbench/metrics/refusal.py +270 -0
- synthbench/metrics/subgroup.py +58 -0
- synthbench/private_holdout.py +240 -0
- synthbench/providers/__init__.py +43 -0
- synthbench/providers/_parsing.py +150 -0
- synthbench/providers/_retry.py +107 -0
- synthbench/providers/base.py +212 -0
- synthbench/providers/http.py +108 -0
- synthbench/providers/majority_baseline.py +23 -0
- synthbench/providers/ollama.py +109 -0
- synthbench/providers/openrouter.py +144 -0
- synthbench/providers/population_baseline.py +87 -0
- synthbench/providers/random_baseline.py +24 -0
- synthbench/providers/raw_anthropic.py +149 -0
- synthbench/providers/raw_gemini.py +145 -0
- synthbench/providers/raw_openai.py +139 -0
- synthbench/providers/synthpanel.py +978 -0
- synthbench/publish.py +2772 -0
- synthbench/r2_upload.py +178 -0
- synthbench/recompute.py +235 -0
- synthbench/report.py +493 -0
- synthbench/run_hash.py +111 -0
- synthbench/run_validity.py +232 -0
- synthbench/runner.py +834 -0
- synthbench/stats.py +1492 -0
- synthbench/submission.py +284 -0
- synthbench/submission_pr.py +436 -0
- synthbench/submit_adapter.py +748 -0
- synthbench/suite.py +424 -0
- synthbench/suites/__init__.py +94 -0
- synthbench/topics.py +256 -0
- synthbench/user_config.py +409 -0
- synthbench/validation.py +1556 -0
- synthbench/visualize.py +455 -0
- synthbench_eval-0.4.0.dist-info/METADATA +279 -0
- synthbench_eval-0.4.0.dist-info/RECORD +78 -0
- synthbench_eval-0.4.0.dist-info/WHEEL +5 -0
- synthbench_eval-0.4.0.dist-info/entry_points.txt +2 -0
- synthbench_eval-0.4.0.dist-info/licenses/LICENSE +21 -0
- synthbench_eval-0.4.0.dist-info/top_level.txt +1 -0
synthbench/__init__.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""SynthBench — open benchmark harness for synthetic survey respondent quality."""
|
|
2
|
+
|
|
3
|
+
from synthbench.convergence.baseline import (
|
|
4
|
+
BaselineGatedError,
|
|
5
|
+
BaselineUnavailable,
|
|
6
|
+
load_convergence_baseline,
|
|
7
|
+
)
|
|
8
|
+
|
|
9
|
+
__version__ = "0.4.0"
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"BaselineGatedError",
|
|
13
|
+
"BaselineUnavailable",
|
|
14
|
+
"load_convergence_baseline",
|
|
15
|
+
"__version__",
|
|
16
|
+
]
|
synthbench/__main__.py
ADDED
synthbench/adapter.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Vendor-adapter interface for `synthbench submit-adapter` (refs #256).
|
|
2
|
+
|
|
3
|
+
The adapter is the contract between SynthBench and a third-party synthetic-
|
|
4
|
+
respondent vendor. Vendors implement a subclass of :class:`Adapter` in their
|
|
5
|
+
own Python module, then point `synthbench submit-adapter --adapter` at the
|
|
6
|
+
module path. The submit pipeline imports the module, instantiates the
|
|
7
|
+
adapter, runs the standard suite against it, and produces a submission
|
|
8
|
+
artifact suitable for opening a PR against the leaderboard repo.
|
|
9
|
+
|
|
10
|
+
Adapters are intentionally narrow: vendors expose **one** async method
|
|
11
|
+
(:meth:`Adapter.respond`) that returns a string answer to a single question
|
|
12
|
+
under a single persona condition. Everything upstream of that — distribution
|
|
13
|
+
estimation, scoring, run-hash content addressing — is owned by SynthBench so
|
|
14
|
+
that vendor implementations stay simple and the evaluation surface stays
|
|
15
|
+
honest.
|
|
16
|
+
|
|
17
|
+
NOTE (refs #256): the full evaluation pipeline that drives `Adapter.respond`
|
|
18
|
+
through the `core` suite is still scaffold-only as of this PR. The interface
|
|
19
|
+
is frozen here so vendors can start writing adapters in parallel with the
|
|
20
|
+
pipeline build-out tracked in follow-up issues.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import random
|
|
26
|
+
from abc import ABC, abstractmethod
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class Adapter(ABC):
|
|
31
|
+
"""Base class every vendor adapter must subclass.
|
|
32
|
+
|
|
33
|
+
Implementations should be:
|
|
34
|
+
* **Stateless across questions** — the harness may call ``respond``
|
|
35
|
+
from many coroutines concurrently. Any per-instance state must be
|
|
36
|
+
safe under ``asyncio.gather``.
|
|
37
|
+
* **Side-effect free** — no writing to disk, no global mutation. The
|
|
38
|
+
harness is responsible for persistence.
|
|
39
|
+
* **Deterministic given the same inputs**, where possible. Sampling
|
|
40
|
+
randomness is expected (models are stochastic); orchestration
|
|
41
|
+
randomness is not.
|
|
42
|
+
|
|
43
|
+
Two metadata properties are required so the harness can stamp the run
|
|
44
|
+
artifact with vendor identity without re-reading CLI flags:
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
@property
|
|
48
|
+
@abstractmethod
|
|
49
|
+
def name(self) -> str:
|
|
50
|
+
"""Human-readable vendor/model identifier (e.g. ``"acme/gpt-4o"``).
|
|
51
|
+
|
|
52
|
+
This appears in the leaderboard row and the submission filename, so
|
|
53
|
+
prefer short, stable strings. Avoid embedding timestamps or git
|
|
54
|
+
SHAs here — use :attr:`version` for that.
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
@abstractmethod
|
|
59
|
+
def version(self) -> str:
|
|
60
|
+
"""Adapter version string (e.g. ``"2026.05.14"`` or a semver tag).
|
|
61
|
+
|
|
62
|
+
Bumped whenever the prompt, sampling parameters, or model snapshot
|
|
63
|
+
materially changes. Two runs with the same ``(name, version)`` are
|
|
64
|
+
expected to be comparable; differing versions are not.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
@abstractmethod
|
|
68
|
+
async def respond(
|
|
69
|
+
self,
|
|
70
|
+
*,
|
|
71
|
+
question: str,
|
|
72
|
+
persona: dict[str, Any],
|
|
73
|
+
context: dict[str, Any] | None = None,
|
|
74
|
+
) -> str:
|
|
75
|
+
"""Answer a single question under a single persona condition.
|
|
76
|
+
|
|
77
|
+
Args:
|
|
78
|
+
question: The prompt the harness wants a respondent answer to.
|
|
79
|
+
Already includes options where applicable; the adapter
|
|
80
|
+
should not re-format it.
|
|
81
|
+
persona: Demographic + attitudinal conditioning for this
|
|
82
|
+
respondent. Keys follow the
|
|
83
|
+
:class:`synthbench.providers.base.PersonaSpec` shape (e.g.
|
|
84
|
+
``{"age": "30-44", "party": "Democrat", ...}``). May be
|
|
85
|
+
empty if the suite calls for unconditioned answers.
|
|
86
|
+
context: Optional suite-specific metadata (e.g. question id,
|
|
87
|
+
survey wave). Adapters should ignore unknown keys.
|
|
88
|
+
|
|
89
|
+
Returns:
|
|
90
|
+
The raw answer string. The harness handles parsing this into
|
|
91
|
+
an option index or Likert number, so adapters should return
|
|
92
|
+
whatever shape feels native to their model (a single token, a
|
|
93
|
+
short phrase, or a number).
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class RandomAdapter(Adapter):
|
|
98
|
+
"""Reference adapter that emits trivial answers, for wiring up tests.
|
|
99
|
+
|
|
100
|
+
Returns ``"yes"`` / ``"no"`` for boolean-flavored questions and a
|
|
101
|
+
uniformly random Likert integer (1-5) otherwise. Seeded so the test
|
|
102
|
+
suite stays deterministic.
|
|
103
|
+
|
|
104
|
+
This adapter exists so vendors (and SynthBench CI) have a known-good
|
|
105
|
+
target to point ``synthbench submit-adapter --adapter`` at while
|
|
106
|
+
debugging their setup. It is **not** a baseline — its scores are
|
|
107
|
+
expected to be terrible.
|
|
108
|
+
"""
|
|
109
|
+
|
|
110
|
+
def __init__(self, seed: int = 42) -> None:
|
|
111
|
+
self._rng = random.Random(seed)
|
|
112
|
+
|
|
113
|
+
@property
|
|
114
|
+
def name(self) -> str:
|
|
115
|
+
return "synthbench/random-adapter"
|
|
116
|
+
|
|
117
|
+
@property
|
|
118
|
+
def version(self) -> str:
|
|
119
|
+
return "0.1.0"
|
|
120
|
+
|
|
121
|
+
async def respond(
|
|
122
|
+
self,
|
|
123
|
+
*,
|
|
124
|
+
question: str,
|
|
125
|
+
persona: dict[str, Any],
|
|
126
|
+
context: dict[str, Any] | None = None,
|
|
127
|
+
) -> str:
|
|
128
|
+
q = question.lower()
|
|
129
|
+
if "yes or no" in q or q.strip().endswith("?") and "agree" not in q:
|
|
130
|
+
return self._rng.choice(["yes", "no"])
|
|
131
|
+
return str(self._rng.randint(1, 5))
|
synthbench/anomaly.py
ADDED
|
@@ -0,0 +1,503 @@
|
|
|
1
|
+
"""Tier-3 statistical anomaly detection for SynthBench submissions.
|
|
2
|
+
|
|
3
|
+
Tier 1 (schema) and tier 2 (arithmetic recomputation) both trust the
|
|
4
|
+
per-question distributions the submitter attached. A sufficiently careful
|
|
5
|
+
fabricator can defeat them by reverse-engineering distributions that are
|
|
6
|
+
arithmetically self-consistent. Tier 3 adds cheap statistical plausibility
|
|
7
|
+
checks that catch the most common lazy attacks: "copy the answer key",
|
|
8
|
+
"zero-out refusals", "match one peer run exactly".
|
|
9
|
+
|
|
10
|
+
Each detector returns an :class:`~synthbench.validation.Issue` with
|
|
11
|
+
``severity = WARNING``. We deliberately keep tier 3 soft in its first
|
|
12
|
+
iteration — once we've observed the distribution of legitimate
|
|
13
|
+
submissions in practice, specific detectors can be graduated to ERROR.
|
|
14
|
+
|
|
15
|
+
Detectors:
|
|
16
|
+
|
|
17
|
+
* ``check_suspicious_perfection`` — per-question JSD near zero with near-zero
|
|
18
|
+
variance implies the submitter copied the human distribution verbatim.
|
|
19
|
+
* ``check_missing_refusals`` — a submission with model_refusal_rate ≡ 0
|
|
20
|
+
across a dataset where real humans refuse is suspicious: real LLMs do
|
|
21
|
+
refuse, and refusal-free runs are almost always fabricated or
|
|
22
|
+
miscalibrated post-processing.
|
|
23
|
+
* ``check_peer_distribution_outlier`` — soft signal comparing claimed
|
|
24
|
+
per-question distributions against peer submissions on the same family.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import math
|
|
30
|
+
from typing import Any, Iterable, Mapping, Sequence
|
|
31
|
+
|
|
32
|
+
from synthbench.private_holdout import is_holdout_enabled, is_private_holdout
|
|
33
|
+
from synthbench.validation import Issue, Severity
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# Detector thresholds. Tuned against real submissions in leaderboard-results/
|
|
37
|
+
# where mean_jsd lives in ~[0.05, 0.55] with std ~[0.10, 0.30]. Fabrication
|
|
38
|
+
# that copies the answer key produces mean ~0 with std ~0; the threshold sits
|
|
39
|
+
# an order of magnitude below the noise floor of any real run we've seen.
|
|
40
|
+
SUSPICIOUS_MEAN_JSD = 0.005
|
|
41
|
+
SUSPICIOUS_STD_JSD = 0.005
|
|
42
|
+
|
|
43
|
+
# Minimum JSD sample size at which ANOMALY_PERFECTION is promoted to ERROR.
|
|
44
|
+
# Below this, the detector stays WARNING so small debug/test fixtures are not
|
|
45
|
+
# hard-rejected. At n >= 25 the thresholds are so far below the real-run noise
|
|
46
|
+
# floor (mean_jsd in [0.05, 0.55], std_jsd in [0.10, 0.30]) that tripping them
|
|
47
|
+
# is a statistically reliable fabrication signal — see Wang et al. (Berkeley,
|
|
48
|
+
# 2026) and docs/benchmark-hardening-analysis.md §2.
|
|
49
|
+
ANOMALY_PERFECTION_ERROR_MIN_N = 25
|
|
50
|
+
|
|
51
|
+
# Share of dataset questions that must have a human refusal rate above this
|
|
52
|
+
# cutoff before a zero-refusal submission is flagged. We require at least
|
|
53
|
+
# one visibly-refusing question in the dataset to avoid false positives on
|
|
54
|
+
# datasets where humans rarely refuse.
|
|
55
|
+
HUMAN_REFUSAL_CUTOFF = 0.05
|
|
56
|
+
HUMAN_REFUSAL_MIN_QUESTIONS = 3
|
|
57
|
+
|
|
58
|
+
# Peer-outlier detector: two submissions are "same-family" if they share
|
|
59
|
+
# the model family (e.g. both claim claude-haiku-4-5) and the dataset.
|
|
60
|
+
# We flag when the mean absolute delta between submission and peer JSD
|
|
61
|
+
# on overlapping questions exceeds :data:`PEER_OUTLIER_DELTA`. Real
|
|
62
|
+
# same-family runs differ by <=~0.05 on average; 0.15 is well outside
|
|
63
|
+
# observed noise and inside the range produced by answer-key attacks.
|
|
64
|
+
PEER_OUTLIER_DELTA = 0.15
|
|
65
|
+
PEER_MIN_OVERLAP = 5
|
|
66
|
+
|
|
67
|
+
# Near-copy-public detector: the attack is "scrape the public human
|
|
68
|
+
# distribution, add tiny noise, submit." The fingerprint is per-question JSD
|
|
69
|
+
# on the public subset sitting far below what any real LLM can produce.
|
|
70
|
+
# Empirical floor from leaderboard-results/: the best real submission's
|
|
71
|
+
# public-subset mean_jsd ~0.09 with std ~0.08 (claude-sonnet on
|
|
72
|
+
# globalopinionqa). A noise-floor attack with ε ~ U(-0.02, 0.02) yields
|
|
73
|
+
# mean ~0.006 and std ~0.004. Thresholds 0.02 / 0.03 leave ~4x headroom on
|
|
74
|
+
# both sides. Minimum n_public=50 keeps small debug fixtures from tripping.
|
|
75
|
+
NEAR_COPY_MEAN_JSD = 0.02
|
|
76
|
+
NEAR_COPY_STD_JSD = 0.03
|
|
77
|
+
NEAR_COPY_MIN_PUBLIC = 50
|
|
78
|
+
|
|
79
|
+
# Constant-offset fingerprint. A deterministic monotonic transformation of
|
|
80
|
+
# the human distribution (``model = normalize(human + c)`` is the canonical
|
|
81
|
+
# example from docs/benchmark-hardening-analysis.md §3.3; scaling and power
|
|
82
|
+
# transforms produce the same shape) preserves rank order exactly on every
|
|
83
|
+
# question — per-question ``kendall_tau == 1.0`` uniformly. Real LLM
|
|
84
|
+
# sampling flips minor ranks on nearly-tied options: across every real
|
|
85
|
+
# submission in leaderboard-results/ the share of questions with perfect
|
|
86
|
+
# tau tops out around 26% (claude-haiku, ensemble) and is typically 10–20%.
|
|
87
|
+
# A 95% threshold over n >= 25 sits >3x above the real-run ceiling and
|
|
88
|
+
# catches the fabricated ``constant_offset.json`` fixture (100% perfect tau).
|
|
89
|
+
CONSTANT_OFFSET_PERFECT_TAU_FRACTION = 0.95
|
|
90
|
+
CONSTANT_OFFSET_TAU_EPSILON = 1e-4
|
|
91
|
+
CONSTANT_OFFSET_MIN_N = 25
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _mean(values: Sequence[float]) -> float:
|
|
95
|
+
return sum(values) / len(values) if values else 0.0
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _std(values: Sequence[float]) -> float:
|
|
99
|
+
if len(values) < 2:
|
|
100
|
+
return 0.0
|
|
101
|
+
mu = _mean(values)
|
|
102
|
+
variance = sum((v - mu) ** 2 for v in values) / (len(values) - 1)
|
|
103
|
+
return math.sqrt(variance)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def check_suspicious_perfection(
|
|
107
|
+
per_question: Sequence[Mapping[str, Any]],
|
|
108
|
+
) -> Issue | None:
|
|
109
|
+
"""Flag submissions whose per-question JSD is implausibly perfect.
|
|
110
|
+
|
|
111
|
+
Real LLMs produce per-question JSD with a non-trivial spread. A
|
|
112
|
+
submission where either the mean OR the standard deviation of JSD
|
|
113
|
+
sits below :data:`SUSPICIOUS_MEAN_JSD` / :data:`SUSPICIOUS_STD_JSD`
|
|
114
|
+
is almost certainly copied from the answer key. We use OR because
|
|
115
|
+
either condition alone is enough: mean-near-zero implies uniformly
|
|
116
|
+
near-perfect matches (impossible for a real model), and std-near-zero
|
|
117
|
+
implies the per-question distances are all the same, which a real
|
|
118
|
+
sampling pipeline cannot produce.
|
|
119
|
+
|
|
120
|
+
Severity is ERROR when the JSD sample size reaches
|
|
121
|
+
:data:`ANOMALY_PERFECTION_ERROR_MIN_N`; smaller samples stay WARNING
|
|
122
|
+
so debug fixtures aren't hard-rejected on borderline numerics.
|
|
123
|
+
"""
|
|
124
|
+
jsd_values = [
|
|
125
|
+
float(q["jsd"]) for q in per_question if isinstance(q.get("jsd"), (int, float))
|
|
126
|
+
]
|
|
127
|
+
if len(jsd_values) < 5:
|
|
128
|
+
return None
|
|
129
|
+
|
|
130
|
+
mean_jsd = _mean(jsd_values)
|
|
131
|
+
std_jsd = _std(jsd_values)
|
|
132
|
+
|
|
133
|
+
if mean_jsd < SUSPICIOUS_MEAN_JSD or std_jsd < SUSPICIOUS_STD_JSD:
|
|
134
|
+
severity = (
|
|
135
|
+
Severity.ERROR
|
|
136
|
+
if len(jsd_values) >= ANOMALY_PERFECTION_ERROR_MIN_N
|
|
137
|
+
else Severity.WARNING
|
|
138
|
+
)
|
|
139
|
+
return Issue(
|
|
140
|
+
code="ANOMALY_PERFECTION",
|
|
141
|
+
severity=severity,
|
|
142
|
+
message=(
|
|
143
|
+
f"per-question JSD has mean={mean_jsd:.6f}, "
|
|
144
|
+
f"std={std_jsd:.6f} across n={len(jsd_values)} questions "
|
|
145
|
+
f"— implausibly perfect match to human distribution "
|
|
146
|
+
f"(thresholds: mean<{SUSPICIOUS_MEAN_JSD}, "
|
|
147
|
+
f"std<{SUSPICIOUS_STD_JSD}). Likely answer-key copy."
|
|
148
|
+
),
|
|
149
|
+
path="per_question",
|
|
150
|
+
)
|
|
151
|
+
return None
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def check_missing_refusals(
|
|
155
|
+
per_question: Sequence[Mapping[str, Any]],
|
|
156
|
+
) -> Issue | None:
|
|
157
|
+
"""Flag submissions that never refuse despite humans refusing.
|
|
158
|
+
|
|
159
|
+
Some datasets have questions humans refuse on (e.g. political
|
|
160
|
+
identification, touchy religious questions). Real LLMs refuse those
|
|
161
|
+
too — sometimes at similar rates, sometimes higher. A submission
|
|
162
|
+
that reports ``model_refusal_rate = 0`` across every question in a
|
|
163
|
+
dataset where humans clearly refuse is either fabricated or has a
|
|
164
|
+
silent bug that drops refusals. Either way, worth surfacing.
|
|
165
|
+
"""
|
|
166
|
+
pq_list = [q for q in per_question if isinstance(q, dict)]
|
|
167
|
+
if not pq_list:
|
|
168
|
+
return None
|
|
169
|
+
|
|
170
|
+
refusing_questions = [
|
|
171
|
+
q
|
|
172
|
+
for q in pq_list
|
|
173
|
+
if isinstance(q.get("human_refusal_rate"), (int, float))
|
|
174
|
+
and float(q["human_refusal_rate"]) >= HUMAN_REFUSAL_CUTOFF
|
|
175
|
+
]
|
|
176
|
+
if len(refusing_questions) < HUMAN_REFUSAL_MIN_QUESTIONS:
|
|
177
|
+
# Dataset doesn't have enough "refuseable" questions to reason about.
|
|
178
|
+
return None
|
|
179
|
+
|
|
180
|
+
model_refusal_values = [float(q.get("model_refusal_rate") or 0.0) for q in pq_list]
|
|
181
|
+
if any(v > 0.0 for v in model_refusal_values):
|
|
182
|
+
return None
|
|
183
|
+
|
|
184
|
+
return Issue(
|
|
185
|
+
code="ANOMALY_NO_REFUSAL",
|
|
186
|
+
severity=Severity.WARNING,
|
|
187
|
+
message=(
|
|
188
|
+
f"submission reports model_refusal_rate=0 on every question, "
|
|
189
|
+
f"but {len(refusing_questions)} of {len(pq_list)} questions "
|
|
190
|
+
f"have human_refusal_rate >= {HUMAN_REFUSAL_CUTOFF}. Real "
|
|
191
|
+
f"LLMs refuse sometimes — check for fabricated or post-"
|
|
192
|
+
f"processed refusals."
|
|
193
|
+
),
|
|
194
|
+
path="per_question",
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _same_family(submission_provider: str, peer_provider: str) -> bool:
|
|
199
|
+
"""True when two provider strings plausibly share a model family.
|
|
200
|
+
|
|
201
|
+
Heuristic: take the last path segment (usually the model id) and
|
|
202
|
+
compare. ``openrouter/anthropic/claude-haiku-4-5`` and
|
|
203
|
+
``anthropic/claude-haiku-4-5`` both resolve to ``claude-haiku-4-5``.
|
|
204
|
+
Baselines and ensembles are never "same-family".
|
|
205
|
+
"""
|
|
206
|
+
if not submission_provider or not peer_provider:
|
|
207
|
+
return False
|
|
208
|
+
if "baseline" in submission_provider or "baseline" in peer_provider:
|
|
209
|
+
return False
|
|
210
|
+
if submission_provider.startswith("ensemble/") or peer_provider.startswith(
|
|
211
|
+
"ensemble/"
|
|
212
|
+
):
|
|
213
|
+
return False
|
|
214
|
+
|
|
215
|
+
def _model_tail(name: str) -> str:
|
|
216
|
+
return name.rsplit("/", 1)[-1].lower()
|
|
217
|
+
|
|
218
|
+
return _model_tail(submission_provider) == _model_tail(peer_provider)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def check_peer_distribution_outlier(
|
|
222
|
+
submission: Mapping[str, Any],
|
|
223
|
+
peers: Iterable[Mapping[str, Any]],
|
|
224
|
+
) -> Issue | None:
|
|
225
|
+
"""Soft check: claimed distribution shapes vs same-family peers.
|
|
226
|
+
|
|
227
|
+
If the submission claims model X but its per-question JSD on
|
|
228
|
+
overlapping questions is wildly out of line with other runs of
|
|
229
|
+
model X, flag it. This catches submissions that claim a particular
|
|
230
|
+
model but were actually generated by something else.
|
|
231
|
+
|
|
232
|
+
Not a hard reject — honest runs differ from each other (temperature,
|
|
233
|
+
prompt, seed). We only flag when the deviation exceeds
|
|
234
|
+
:data:`PEER_OUTLIER_SIGMA` standard deviations of the peer JSD
|
|
235
|
+
distribution on the shared question set. Returns ``None`` when
|
|
236
|
+
there are no same-family peers or insufficient overlap.
|
|
237
|
+
"""
|
|
238
|
+
config = submission.get("config") or {}
|
|
239
|
+
submission_provider = str(config.get("provider", ""))
|
|
240
|
+
dataset = config.get("dataset")
|
|
241
|
+
per_question = submission.get("per_question") or []
|
|
242
|
+
|
|
243
|
+
if not submission_provider or not dataset or not per_question:
|
|
244
|
+
return None
|
|
245
|
+
|
|
246
|
+
submission_jsd_by_key: dict[str, float] = {}
|
|
247
|
+
for q in per_question:
|
|
248
|
+
if not isinstance(q, dict):
|
|
249
|
+
continue
|
|
250
|
+
key = q.get("key")
|
|
251
|
+
jsd = q.get("jsd")
|
|
252
|
+
if isinstance(key, str) and isinstance(jsd, (int, float)):
|
|
253
|
+
submission_jsd_by_key[key] = float(jsd)
|
|
254
|
+
|
|
255
|
+
# Collect peer JSD maps on same model family + same dataset.
|
|
256
|
+
peer_jsd_maps: list[dict[str, float]] = []
|
|
257
|
+
for peer in peers:
|
|
258
|
+
if not isinstance(peer, dict):
|
|
259
|
+
continue
|
|
260
|
+
peer_config = peer.get("config") or {}
|
|
261
|
+
peer_provider = str(peer_config.get("provider", ""))
|
|
262
|
+
peer_dataset = peer_config.get("dataset")
|
|
263
|
+
if peer_dataset != dataset or not _same_family(
|
|
264
|
+
submission_provider, peer_provider
|
|
265
|
+
):
|
|
266
|
+
continue
|
|
267
|
+
|
|
268
|
+
peer_jsd: dict[str, float] = {}
|
|
269
|
+
for q in peer.get("per_question") or []:
|
|
270
|
+
if not isinstance(q, dict):
|
|
271
|
+
continue
|
|
272
|
+
key = q.get("key")
|
|
273
|
+
jsd = q.get("jsd")
|
|
274
|
+
if isinstance(key, str) and isinstance(jsd, (int, float)):
|
|
275
|
+
peer_jsd[key] = float(jsd)
|
|
276
|
+
if peer_jsd:
|
|
277
|
+
peer_jsd_maps.append(peer_jsd)
|
|
278
|
+
|
|
279
|
+
if not peer_jsd_maps:
|
|
280
|
+
return None
|
|
281
|
+
|
|
282
|
+
# Compute per-key peer mean JSD using every peer that covered that key.
|
|
283
|
+
per_key_peer_values: dict[str, list[float]] = {}
|
|
284
|
+
for peer_map in peer_jsd_maps:
|
|
285
|
+
for key, jsd in peer_map.items():
|
|
286
|
+
if key in submission_jsd_by_key:
|
|
287
|
+
per_key_peer_values.setdefault(key, []).append(jsd)
|
|
288
|
+
|
|
289
|
+
overlap = [
|
|
290
|
+
(submission_jsd_by_key[k], _mean(vals))
|
|
291
|
+
for k, vals in per_key_peer_values.items()
|
|
292
|
+
if vals
|
|
293
|
+
]
|
|
294
|
+
if len(overlap) < PEER_MIN_OVERLAP:
|
|
295
|
+
return None
|
|
296
|
+
|
|
297
|
+
deltas = [sub - peer_mean for sub, peer_mean in overlap]
|
|
298
|
+
mu = _mean(deltas)
|
|
299
|
+
if abs(mu) < PEER_OUTLIER_DELTA:
|
|
300
|
+
return None
|
|
301
|
+
|
|
302
|
+
direction = "lower" if mu < 0 else "higher"
|
|
303
|
+
return Issue(
|
|
304
|
+
code="ANOMALY_PEER_OUTLIER",
|
|
305
|
+
severity=Severity.WARNING,
|
|
306
|
+
message=(
|
|
307
|
+
f"per-question JSD runs {direction} than same-family peers "
|
|
308
|
+
f"({len(overlap)} shared questions, mean delta={mu:.4f} > "
|
|
309
|
+
f"threshold {PEER_OUTLIER_DELTA}). Investigate whether the "
|
|
310
|
+
f"claimed model matches the submission."
|
|
311
|
+
),
|
|
312
|
+
path="per_question",
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def check_near_copy_public(
|
|
317
|
+
data: Mapping[str, Any],
|
|
318
|
+
) -> Issue | None:
|
|
319
|
+
"""Flag submissions whose public-subset JSD is implausibly near-zero.
|
|
320
|
+
|
|
321
|
+
The attack this closes: scrape published ``human_distribution`` values
|
|
322
|
+
from the public leaderboard, add small noise, submit as
|
|
323
|
+
``model_distribution``. Every tier-1/2 check passes (distributions are
|
|
324
|
+
valid, aggregates reconcile), and ``ANOMALY_PERFECTION`` can miss it
|
|
325
|
+
because the attacker's fabricated private rows dilute the full-dataset
|
|
326
|
+
JSD. Restricting the statistic to the public subset removes that
|
|
327
|
+
dilution — the attack's fingerprint is uniformly tight JSD across
|
|
328
|
+
every public question.
|
|
329
|
+
|
|
330
|
+
Fires only on holdout-enabled datasets. When the dataset is not
|
|
331
|
+
partitioned, we have no principled public/private split to restrict
|
|
332
|
+
to. Returns ``None`` below :data:`NEAR_COPY_MIN_PUBLIC` public rows.
|
|
333
|
+
|
|
334
|
+
The detector requires BOTH mean < :data:`NEAR_COPY_MEAN_JSD` AND std <
|
|
335
|
+
:data:`NEAR_COPY_STD_JSD` to flag. Unlike ``ANOMALY_PERFECTION`` (OR),
|
|
336
|
+
this AND guard reduces false positives on genuinely strong real
|
|
337
|
+
models — a real model can have either tight variance or a low mean,
|
|
338
|
+
but not both simultaneously on 50+ public questions, in every run
|
|
339
|
+
we've observed.
|
|
340
|
+
"""
|
|
341
|
+
config = data.get("config") or {}
|
|
342
|
+
dataset = config.get("dataset")
|
|
343
|
+
if not isinstance(dataset, str) or not is_holdout_enabled(dataset):
|
|
344
|
+
return None
|
|
345
|
+
|
|
346
|
+
per_question = data.get("per_question") or []
|
|
347
|
+
if not isinstance(per_question, list):
|
|
348
|
+
return None
|
|
349
|
+
|
|
350
|
+
public_jsd: list[float] = []
|
|
351
|
+
for q in per_question:
|
|
352
|
+
if not isinstance(q, dict):
|
|
353
|
+
continue
|
|
354
|
+
key = q.get("key")
|
|
355
|
+
jsd = q.get("jsd")
|
|
356
|
+
if not isinstance(key, str) or not isinstance(jsd, (int, float)):
|
|
357
|
+
continue
|
|
358
|
+
if is_private_holdout(dataset, key):
|
|
359
|
+
continue
|
|
360
|
+
public_jsd.append(float(jsd))
|
|
361
|
+
|
|
362
|
+
if len(public_jsd) < NEAR_COPY_MIN_PUBLIC:
|
|
363
|
+
return None
|
|
364
|
+
|
|
365
|
+
mean_jsd = _mean(public_jsd)
|
|
366
|
+
std_jsd = _std(public_jsd)
|
|
367
|
+
|
|
368
|
+
if mean_jsd >= NEAR_COPY_MEAN_JSD or std_jsd >= NEAR_COPY_STD_JSD:
|
|
369
|
+
return None
|
|
370
|
+
|
|
371
|
+
return Issue(
|
|
372
|
+
code="ANOMALY_NEAR_COPY_PUBLIC",
|
|
373
|
+
severity=Severity.ERROR,
|
|
374
|
+
message=(
|
|
375
|
+
f"public-subset JSD has mean={mean_jsd:.6f}, std={std_jsd:.6f} "
|
|
376
|
+
f"over {len(public_jsd)} public questions (thresholds: "
|
|
377
|
+
f"mean<{NEAR_COPY_MEAN_JSD}, std<{NEAR_COPY_STD_JSD}). This is "
|
|
378
|
+
f"the fingerprint of a submission that copied the published "
|
|
379
|
+
f"human_distribution and added small noise. Real LLM runs "
|
|
380
|
+
f"produce public mean_jsd >= ~0.09 on every holdout dataset "
|
|
381
|
+
f"we have on file."
|
|
382
|
+
),
|
|
383
|
+
path="per_question",
|
|
384
|
+
)
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def check_constant_offset(
|
|
388
|
+
per_question: Sequence[Mapping[str, Any]],
|
|
389
|
+
) -> Issue | None:
|
|
390
|
+
"""Flag deterministic monotonic-transformation attacks.
|
|
391
|
+
|
|
392
|
+
The canonical attack from ``docs/benchmark-hardening-analysis.md §3.3``
|
|
393
|
+
is ``model[opt] = human[opt] + c`` renormalized. Because adding a
|
|
394
|
+
constant to every option is a strictly monotonic transformation, the
|
|
395
|
+
option rank order is preserved exactly on every question — so
|
|
396
|
+
per-question ``kendall_tau`` equals 1.0 uniformly. The same fingerprint
|
|
397
|
+
appears for any deterministic monotonic transformation the attacker
|
|
398
|
+
might apply to the answer key (scaling, power transform, etc.).
|
|
399
|
+
|
|
400
|
+
Real LLM sampling flips minor rank orders on nearly-tied options: across
|
|
401
|
+
the full leaderboard the share of questions with ``tau == 1.0`` tops
|
|
402
|
+
out near 26% and sits at 10–20% for typical strong models. A
|
|
403
|
+
submission with ``>= 95%`` of questions at near-unity tau over n >= 25
|
|
404
|
+
is therefore not a plausible real run.
|
|
405
|
+
|
|
406
|
+
The detector is independent of :func:`check_suspicious_perfection` and
|
|
407
|
+
:func:`check_near_copy_public` — those fire on distribution-distance
|
|
408
|
+
signals, which a cleverly-tuned multiplicative/offset attack could
|
|
409
|
+
drive out of range by choosing a large constant. The tau fingerprint
|
|
410
|
+
survives any monotonic f, which is why it's the right cross-check.
|
|
411
|
+
"""
|
|
412
|
+
tau_values = [
|
|
413
|
+
float(q["kendall_tau"])
|
|
414
|
+
for q in per_question
|
|
415
|
+
if isinstance(q.get("kendall_tau"), (int, float))
|
|
416
|
+
]
|
|
417
|
+
if len(tau_values) < CONSTANT_OFFSET_MIN_N:
|
|
418
|
+
return None
|
|
419
|
+
|
|
420
|
+
perfect = sum(1 for t in tau_values if t >= 1.0 - CONSTANT_OFFSET_TAU_EPSILON)
|
|
421
|
+
fraction = perfect / len(tau_values)
|
|
422
|
+
if fraction < CONSTANT_OFFSET_PERFECT_TAU_FRACTION:
|
|
423
|
+
return None
|
|
424
|
+
|
|
425
|
+
return Issue(
|
|
426
|
+
code="ANOMALY_CONSTANT_OFFSET",
|
|
427
|
+
severity=Severity.ERROR,
|
|
428
|
+
message=(
|
|
429
|
+
f"per-question kendall_tau is near-unity on "
|
|
430
|
+
f"{perfect}/{len(tau_values)} questions ({fraction:.1%}, "
|
|
431
|
+
f"threshold >= {CONSTANT_OFFSET_PERFECT_TAU_FRACTION:.0%}). "
|
|
432
|
+
f"Real LLM sampling flips minor rank orders on nearly-tied "
|
|
433
|
+
f"options — uniform rank preservation is the fingerprint of a "
|
|
434
|
+
f"deterministic monotonic transformation of the human "
|
|
435
|
+
f"distribution (e.g. model = normalize(human + c)). See "
|
|
436
|
+
f"docs/benchmark-hardening-analysis.md §3.3."
|
|
437
|
+
),
|
|
438
|
+
path="per_question",
|
|
439
|
+
)
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def tier3_checks(
|
|
443
|
+
data: Mapping[str, Any],
|
|
444
|
+
*,
|
|
445
|
+
peers: Iterable[Mapping[str, Any]] = (),
|
|
446
|
+
) -> list[Issue]:
|
|
447
|
+
"""Run every tier-3 anomaly detector and return the list of issues.
|
|
448
|
+
|
|
449
|
+
Note: ``check_missing_refusals`` is intentionally *not* called from
|
|
450
|
+
the default dispatch. Every current provider prompt in
|
|
451
|
+
``src/synthbench/providers/`` ends with ``"Respond with ONLY the
|
|
452
|
+
letter of your choice"`` and gives the model no way to refuse — the
|
|
453
|
+
harness architecturally produces ``model_refusal_rate == 0`` on
|
|
454
|
+
every question, so the detector flagged every legitimate submission
|
|
455
|
+
(sb-a613). The function is kept exported so callers can invoke it
|
|
456
|
+
directly once a refusal-capable prompt variant exists.
|
|
457
|
+
"""
|
|
458
|
+
per_question = data.get("per_question") or []
|
|
459
|
+
if not isinstance(per_question, list):
|
|
460
|
+
return []
|
|
461
|
+
|
|
462
|
+
issues: list[Issue] = []
|
|
463
|
+
|
|
464
|
+
perfection = check_suspicious_perfection(per_question)
|
|
465
|
+
if perfection is not None:
|
|
466
|
+
issues.append(perfection)
|
|
467
|
+
|
|
468
|
+
near_copy = check_near_copy_public(data)
|
|
469
|
+
if near_copy is not None:
|
|
470
|
+
issues.append(near_copy)
|
|
471
|
+
|
|
472
|
+
constant_offset = check_constant_offset(per_question)
|
|
473
|
+
if constant_offset is not None:
|
|
474
|
+
issues.append(constant_offset)
|
|
475
|
+
|
|
476
|
+
peer_outlier = check_peer_distribution_outlier(data, peers)
|
|
477
|
+
if peer_outlier is not None:
|
|
478
|
+
issues.append(peer_outlier)
|
|
479
|
+
|
|
480
|
+
return issues
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
__all__ = [
|
|
484
|
+
"SUSPICIOUS_MEAN_JSD",
|
|
485
|
+
"SUSPICIOUS_STD_JSD",
|
|
486
|
+
"ANOMALY_PERFECTION_ERROR_MIN_N",
|
|
487
|
+
"HUMAN_REFUSAL_CUTOFF",
|
|
488
|
+
"HUMAN_REFUSAL_MIN_QUESTIONS",
|
|
489
|
+
"PEER_OUTLIER_DELTA",
|
|
490
|
+
"PEER_MIN_OVERLAP",
|
|
491
|
+
"NEAR_COPY_MEAN_JSD",
|
|
492
|
+
"NEAR_COPY_STD_JSD",
|
|
493
|
+
"NEAR_COPY_MIN_PUBLIC",
|
|
494
|
+
"CONSTANT_OFFSET_PERFECT_TAU_FRACTION",
|
|
495
|
+
"CONSTANT_OFFSET_TAU_EPSILON",
|
|
496
|
+
"CONSTANT_OFFSET_MIN_N",
|
|
497
|
+
"check_suspicious_perfection",
|
|
498
|
+
"check_missing_refusals",
|
|
499
|
+
"check_peer_distribution_outlier",
|
|
500
|
+
"check_near_copy_public",
|
|
501
|
+
"check_constant_offset",
|
|
502
|
+
"tier3_checks",
|
|
503
|
+
]
|