synthbench-eval 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (149) hide show
  1. synthbench_eval-0.4.0/LICENSE +21 -0
  2. synthbench_eval-0.4.0/PKG-INFO +279 -0
  3. synthbench_eval-0.4.0/README.md +241 -0
  4. synthbench_eval-0.4.0/pyproject.toml +79 -0
  5. synthbench_eval-0.4.0/setup.cfg +4 -0
  6. synthbench_eval-0.4.0/src/synthbench/__init__.py +16 -0
  7. synthbench_eval-0.4.0/src/synthbench/__main__.py +5 -0
  8. synthbench_eval-0.4.0/src/synthbench/adapter.py +131 -0
  9. synthbench_eval-0.4.0/src/synthbench/anomaly.py +503 -0
  10. synthbench_eval-0.4.0/src/synthbench/baseline_floors.py +213 -0
  11. synthbench_eval-0.4.0/src/synthbench/baselines.py +902 -0
  12. synthbench_eval-0.4.0/src/synthbench/cli.py +2929 -0
  13. synthbench_eval-0.4.0/src/synthbench/config_id.py +424 -0
  14. synthbench_eval-0.4.0/src/synthbench/contamination.py +783 -0
  15. synthbench_eval-0.4.0/src/synthbench/convergence/__init__.py +67 -0
  16. synthbench_eval-0.4.0/src/synthbench/convergence/baseline.py +172 -0
  17. synthbench_eval-0.4.0/src/synthbench/convergence/bootstrap.py +60 -0
  18. synthbench_eval-0.4.0/src/synthbench/convergence/cli_report.py +556 -0
  19. synthbench_eval-0.4.0/src/synthbench/convergence/curves.py +111 -0
  20. synthbench_eval-0.4.0/src/synthbench/convergence/real_sampling.py +146 -0
  21. synthbench_eval-0.4.0/src/synthbench/convergence/thresholds.py +49 -0
  22. synthbench_eval-0.4.0/src/synthbench/datasets/__init__.py +37 -0
  23. synthbench_eval-0.4.0/src/synthbench/datasets/base.py +147 -0
  24. synthbench_eval-0.4.0/src/synthbench/datasets/eurobarometer.py +335 -0
  25. synthbench_eval-0.4.0/src/synthbench/datasets/globalopinionqa.py +252 -0
  26. synthbench_eval-0.4.0/src/synthbench/datasets/gss.py +323 -0
  27. synthbench_eval-0.4.0/src/synthbench/datasets/michigan.py +505 -0
  28. synthbench_eval-0.4.0/src/synthbench/datasets/ntia.py +357 -0
  29. synthbench_eval-0.4.0/src/synthbench/datasets/opinionsqa.py +412 -0
  30. synthbench_eval-0.4.0/src/synthbench/datasets/pewtech.py +334 -0
  31. synthbench_eval-0.4.0/src/synthbench/datasets/policy.py +142 -0
  32. synthbench_eval-0.4.0/src/synthbench/datasets/subpop.py +302 -0
  33. synthbench_eval-0.4.0/src/synthbench/datasets/wvs.py +229 -0
  34. synthbench_eval-0.4.0/src/synthbench/findings.py +988 -0
  35. synthbench_eval-0.4.0/src/synthbench/holdout.py +347 -0
  36. synthbench_eval-0.4.0/src/synthbench/human_distributions.py +243 -0
  37. synthbench_eval-0.4.0/src/synthbench/leaderboard.py +715 -0
  38. synthbench_eval-0.4.0/src/synthbench/leaderboard_pr.py +274 -0
  39. synthbench_eval-0.4.0/src/synthbench/metrics/__init__.py +38 -0
  40. synthbench_eval-0.4.0/src/synthbench/metrics/composite.py +72 -0
  41. synthbench_eval-0.4.0/src/synthbench/metrics/conditioning.py +39 -0
  42. synthbench_eval-0.4.0/src/synthbench/metrics/distributional.py +41 -0
  43. synthbench_eval-0.4.0/src/synthbench/metrics/ranking.py +37 -0
  44. synthbench_eval-0.4.0/src/synthbench/metrics/refusal.py +270 -0
  45. synthbench_eval-0.4.0/src/synthbench/metrics/subgroup.py +58 -0
  46. synthbench_eval-0.4.0/src/synthbench/private_holdout.py +240 -0
  47. synthbench_eval-0.4.0/src/synthbench/providers/__init__.py +43 -0
  48. synthbench_eval-0.4.0/src/synthbench/providers/_parsing.py +150 -0
  49. synthbench_eval-0.4.0/src/synthbench/providers/_retry.py +107 -0
  50. synthbench_eval-0.4.0/src/synthbench/providers/base.py +212 -0
  51. synthbench_eval-0.4.0/src/synthbench/providers/http.py +108 -0
  52. synthbench_eval-0.4.0/src/synthbench/providers/majority_baseline.py +23 -0
  53. synthbench_eval-0.4.0/src/synthbench/providers/ollama.py +109 -0
  54. synthbench_eval-0.4.0/src/synthbench/providers/openrouter.py +144 -0
  55. synthbench_eval-0.4.0/src/synthbench/providers/population_baseline.py +87 -0
  56. synthbench_eval-0.4.0/src/synthbench/providers/random_baseline.py +24 -0
  57. synthbench_eval-0.4.0/src/synthbench/providers/raw_anthropic.py +149 -0
  58. synthbench_eval-0.4.0/src/synthbench/providers/raw_gemini.py +145 -0
  59. synthbench_eval-0.4.0/src/synthbench/providers/raw_openai.py +139 -0
  60. synthbench_eval-0.4.0/src/synthbench/providers/synthpanel.py +978 -0
  61. synthbench_eval-0.4.0/src/synthbench/publish.py +2772 -0
  62. synthbench_eval-0.4.0/src/synthbench/r2_upload.py +178 -0
  63. synthbench_eval-0.4.0/src/synthbench/recompute.py +235 -0
  64. synthbench_eval-0.4.0/src/synthbench/report.py +493 -0
  65. synthbench_eval-0.4.0/src/synthbench/run_hash.py +111 -0
  66. synthbench_eval-0.4.0/src/synthbench/run_validity.py +232 -0
  67. synthbench_eval-0.4.0/src/synthbench/runner.py +834 -0
  68. synthbench_eval-0.4.0/src/synthbench/stats.py +1492 -0
  69. synthbench_eval-0.4.0/src/synthbench/submission.py +284 -0
  70. synthbench_eval-0.4.0/src/synthbench/submission_pr.py +436 -0
  71. synthbench_eval-0.4.0/src/synthbench/submit_adapter.py +748 -0
  72. synthbench_eval-0.4.0/src/synthbench/suite.py +424 -0
  73. synthbench_eval-0.4.0/src/synthbench/suites/__init__.py +94 -0
  74. synthbench_eval-0.4.0/src/synthbench/topics.py +256 -0
  75. synthbench_eval-0.4.0/src/synthbench/user_config.py +409 -0
  76. synthbench_eval-0.4.0/src/synthbench/validation.py +1556 -0
  77. synthbench_eval-0.4.0/src/synthbench/visualize.py +455 -0
  78. synthbench_eval-0.4.0/src/synthbench_eval.egg-info/PKG-INFO +279 -0
  79. synthbench_eval-0.4.0/src/synthbench_eval.egg-info/SOURCES.txt +147 -0
  80. synthbench_eval-0.4.0/src/synthbench_eval.egg-info/dependency_links.txt +1 -0
  81. synthbench_eval-0.4.0/src/synthbench_eval.egg-info/entry_points.txt +2 -0
  82. synthbench_eval-0.4.0/src/synthbench_eval.egg-info/requires.txt +35 -0
  83. synthbench_eval-0.4.0/src/synthbench_eval.egg-info/top_level.txt +1 -0
  84. synthbench_eval-0.4.0/tests/test_anomaly.py +369 -0
  85. synthbench_eval-0.4.0/tests/test_baseline_floors.py +356 -0
  86. synthbench_eval-0.4.0/tests/test_baselines.py +572 -0
  87. synthbench_eval-0.4.0/tests/test_cli_run_submit.py +415 -0
  88. synthbench_eval-0.4.0/tests/test_cli_submit.py +223 -0
  89. synthbench_eval-0.4.0/tests/test_cli_submit_adapter.py +543 -0
  90. synthbench_eval-0.4.0/tests/test_config_id.py +396 -0
  91. synthbench_eval-0.4.0/tests/test_contamination.py +694 -0
  92. synthbench_eval-0.4.0/tests/test_convergence_baseline.py +245 -0
  93. synthbench_eval-0.4.0/tests/test_convergence_bootstrap.py +385 -0
  94. synthbench_eval-0.4.0/tests/test_datasets_eurobarometer.py +201 -0
  95. synthbench_eval-0.4.0/tests/test_datasets_globalopinionqa.py +99 -0
  96. synthbench_eval-0.4.0/tests/test_datasets_gss.py +246 -0
  97. synthbench_eval-0.4.0/tests/test_datasets_michigan.py +321 -0
  98. synthbench_eval-0.4.0/tests/test_datasets_ntia.py +400 -0
  99. synthbench_eval-0.4.0/tests/test_datasets_policy.py +219 -0
  100. synthbench_eval-0.4.0/tests/test_datasets_wvs.py +283 -0
  101. synthbench_eval-0.4.0/tests/test_effort.py +465 -0
  102. synthbench_eval-0.4.0/tests/test_findings_drift.py +114 -0
  103. synthbench_eval-0.4.0/tests/test_findings_elicitation.py +140 -0
  104. synthbench_eval-0.4.0/tests/test_findings_nonresponse.py +108 -0
  105. synthbench_eval-0.4.0/tests/test_holdout.py +294 -0
  106. synthbench_eval-0.4.0/tests/test_human_distributions.py +87 -0
  107. synthbench_eval-0.4.0/tests/test_integration_tokens.py +140 -0
  108. synthbench_eval-0.4.0/tests/test_metrics.py +400 -0
  109. synthbench_eval-0.4.0/tests/test_microdata.py +376 -0
  110. synthbench_eval-0.4.0/tests/test_private_holdout.py +202 -0
  111. synthbench_eval-0.4.0/tests/test_provider_parsing.py +939 -0
  112. synthbench_eval-0.4.0/tests/test_providers.py +214 -0
  113. synthbench_eval-0.4.0/tests/test_publish_config_id_consistency.py +240 -0
  114. synthbench_eval-0.4.0/tests/test_publish_cost.py +697 -0
  115. synthbench_eval-0.4.0/tests/test_publish_cross_provider_jsd.py +246 -0
  116. synthbench_eval-0.4.0/tests/test_publish_demographic_scorecard.py +137 -0
  117. synthbench_eval-0.4.0/tests/test_publish_gated_fail_closed.py +277 -0
  118. synthbench_eval-0.4.0/tests/test_publish_holdout.py +290 -0
  119. synthbench_eval-0.4.0/tests/test_publish_invalid_runs.py +232 -0
  120. synthbench_eval-0.4.0/tests/test_publish_normalized.py +161 -0
  121. synthbench_eval-0.4.0/tests/test_publish_policy.py +245 -0
  122. synthbench_eval-0.4.0/tests/test_publish_questions.py +683 -0
  123. synthbench_eval-0.4.0/tests/test_publish_r2_routing.py +317 -0
  124. synthbench_eval-0.4.0/tests/test_publish_rehydration.py +165 -0
  125. synthbench_eval-0.4.0/tests/test_publish_run_counts.py +157 -0
  126. synthbench_eval-0.4.0/tests/test_publish_runnable_ids.py +153 -0
  127. synthbench_eval-0.4.0/tests/test_publish_text_rehydration.py +101 -0
  128. synthbench_eval-0.4.0/tests/test_publish_topic_metrics.py +109 -0
  129. synthbench_eval-0.4.0/tests/test_r2_upload.py +189 -0
  130. synthbench_eval-0.4.0/tests/test_recompute_integrity.py +488 -0
  131. synthbench_eval-0.4.0/tests/test_report.py +162 -0
  132. synthbench_eval-0.4.0/tests/test_run_hash.py +271 -0
  133. synthbench_eval-0.4.0/tests/test_run_validity.py +210 -0
  134. synthbench_eval-0.4.0/tests/test_runner.py +662 -0
  135. synthbench_eval-0.4.0/tests/test_site_dataset_cards.py +100 -0
  136. synthbench_eval-0.4.0/tests/test_stats.py +150 -0
  137. synthbench_eval-0.4.0/tests/test_stats_golden.py +139 -0
  138. synthbench_eval-0.4.0/tests/test_strip_gated_guard.py +113 -0
  139. synthbench_eval-0.4.0/tests/test_submission.py +364 -0
  140. synthbench_eval-0.4.0/tests/test_submission_pr.py +367 -0
  141. synthbench_eval-0.4.0/tests/test_suite.py +325 -0
  142. synthbench_eval-0.4.0/tests/test_suites.py +96 -0
  143. synthbench_eval-0.4.0/tests/test_suites_novel_products.py +127 -0
  144. synthbench_eval-0.4.0/tests/test_topics.py +184 -0
  145. synthbench_eval-0.4.0/tests/test_user_config.py +503 -0
  146. synthbench_eval-0.4.0/tests/test_validation.py +991 -0
  147. synthbench_eval-0.4.0/tests/test_validation_holdout.py +124 -0
  148. synthbench_eval-0.4.0/tests/test_validation_stripped.py +184 -0
  149. synthbench_eval-0.4.0/tests/test_verify_publish_integrity.py +157 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 DataViking Tech
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,279 @@
1
+ Metadata-Version: 2.4
2
+ Name: synthbench-eval
3
+ Version: 0.4.0
4
+ Summary: Open benchmark harness for synthetic survey respondent quality
5
+ License-Expression: MIT
6
+ Keywords: benchmark,survey,synthetic-data,llm,evaluation,opinionsqa,globalopinionqa,gss,convergence,synthpanel
7
+ Requires-Python: >=3.9
8
+ Description-Content-Type: text/markdown
9
+ License-File: LICENSE
10
+ Requires-Dist: click>=8.0
11
+ Requires-Dist: httpx>=0.24
12
+ Requires-Dist: numpy>=1.24
13
+ Requires-Dist: pyyaml>=6.0
14
+ Requires-Dist: scipy>=1.10
15
+ Requires-Dist: synthpanel>=0.2.0
16
+ Provides-Extra: anthropic
17
+ Requires-Dist: anthropic>=0.18; extra == "anthropic"
18
+ Provides-Extra: openai
19
+ Requires-Dist: openai>=1.0; extra == "openai"
20
+ Provides-Extra: viz
21
+ Requires-Dist: matplotlib>=3.7; extra == "viz"
22
+ Provides-Extra: hf
23
+ Requires-Dist: datasets>=2.14; extra == "hf"
24
+ Provides-Extra: subpop
25
+ Requires-Dist: datasets>=2.14; extra == "subpop"
26
+ Provides-Extra: r2
27
+ Requires-Dist: boto3>=1.34; extra == "r2"
28
+ Provides-Extra: all
29
+ Requires-Dist: anthropic>=0.18; extra == "all"
30
+ Requires-Dist: openai>=1.0; extra == "all"
31
+ Requires-Dist: matplotlib>=3.7; extra == "all"
32
+ Requires-Dist: datasets>=2.14; extra == "all"
33
+ Requires-Dist: boto3>=1.34; extra == "all"
34
+ Provides-Extra: dev
35
+ Requires-Dist: pytest>=7.0; extra == "dev"
36
+ Requires-Dist: pytest-asyncio>=0.21; extra == "dev"
37
+ Dynamic: license-file
38
+
39
+ # SynthBench
40
+
41
+ [![CI](https://github.com/DataViking-Tech/synthbench/actions/workflows/ci.yml/badge.svg)](https://github.com/DataViking-Tech/synthbench/actions/workflows/ci.yml)
42
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
43
+ [![Python](https://img.shields.io/badge/python-3.10%20%7C%203.11%20%7C%203.12%20%7C%203.13-blue.svg)](pyproject.toml)
44
+ [![Leaderboard](https://img.shields.io/badge/leaderboard-live-success)](https://synthbench.org/)
45
+
46
+ Open benchmark harness for synthetic survey respondent quality.
47
+
48
+ **The MLPerf of synthetic UXR.**
49
+
50
+ SynthBench measures how well synthetic respondent systems (like [synthpanel](https://github.com/DataViking-Tech/SynthPanel), Ditto, Synthetic Users, or raw ChatGPT prompting) reproduce real human survey response patterns against real Pew American Trends Panel and GlobalOpinionQA ground truth — so "it sounds plausible" gets replaced with a measurable similarity score.
51
+
52
+ ## Quick Start
53
+
54
+ Run your first benchmark in 3 commands:
55
+
56
+ ```bash
57
+ pip install synthbench-eval
58
+ synthbench run --provider random --suite smoke --output results/
59
+ synthbench leaderboard --results-dir results/
60
+ ```
61
+
62
+ > **Note:** the distribution is named `synthbench-eval` — the bare `synthbench`
63
+ > name on PyPI belongs to an unrelated project. The import package and CLI are
64
+ > still `synthbench`. For development, clone this repo and `pip install -e .`.
65
+ > API-backed providers need an extra, e.g. `pip install "synthbench-eval[openai]"`
66
+ > for the `openrouter` / `raw-openai` / `raw-gemini` / `ollama` providers.
67
+
68
+ Try with a real model (requires API key):
69
+
70
+ ```bash
71
+ export OPENROUTER_API_KEY=your-key
72
+ synthbench run --provider openrouter --model openai/gpt-4o-mini --suite core --samples 50
73
+ ```
74
+
75
+ See [`notebooks/getting_started.ipynb`](notebooks/getting_started.ipynb) for a guided walkthrough.
76
+
77
+ ## Leaderboard
78
+
79
+ **[View the live leaderboard](https://synthbench.org/)** — see also the
80
+ [methodology](https://synthbench.org/methodology/) and
81
+ [findings](https://synthbench.org/findings/) pages.
82
+
83
+ Regenerate leaderboard data for the Astro site:
84
+ ```bash
85
+ synthbench publish-data --results-dir ./leaderboard-results --output site/src/data/leaderboard.json
86
+ ```
87
+
88
+ ### Contributor note: gated data publishing
89
+
90
+ Most contributors do **not** need to publish gated artifacts. If you're running
91
+ benchmarks locally or contributing via PR, focus on `synthbench run`,
92
+ `synthbench validate`, and result submission.
93
+
94
+ The gated data publication path is maintainer infrastructure and is handled by
95
+ project deployment workflows.
96
+
97
+ ## Development
98
+
99
+ After cloning, enable the repo-tracked git hooks so pushes that would fail CI's
100
+ `ruff format --check` are caught locally:
101
+
102
+ ```bash
103
+ ./scripts/install-hooks.sh # one-time: wires .githooks/ via core.hooksPath
104
+ ```
105
+
106
+ The `pre-push` hook only checks Python files changed in the commits being
107
+ pushed, so already-formatted branches add no meaningful overhead. Run
108
+ `./scripts/format-check.sh` anytime to mirror the full CI lint job. Emergency
109
+ bypass: `git push --no-verify`.
110
+
111
+ For contribution workflow and PR expectations, see
112
+ [`CONTRIBUTING.md`](CONTRIBUTING.md).
113
+
114
+ ## Submit Results
115
+
116
+ Three ways to land a run on the leaderboard, in order of friction:
117
+
118
+ ### 1. CLI (recommended for repeat submissions)
119
+
120
+ Mint an API key at [synthbench.org/account](https://synthbench.org/account/),
121
+ then:
122
+
123
+ ```bash
124
+ export SYNTHBENCH_API_KEY=sb_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx
125
+ synthbench run --provider openrouter --model gpt-4o-mini --suite full -o results/
126
+ synthbench submit results/openrouter_gpt-4o-mini_opinionsqa.json
127
+ ```
128
+
129
+ #### End-to-end: run + submit in one command (`--submit`)
130
+
131
+ Collapse the two steps above into a single invocation. The CLI saves the
132
+ result JSON locally (so a validation rejection doesn't lose your run) and
133
+ then POSTs it to the leaderboard. With `--wait`, the process blocks until
134
+ the validator reaches a terminal state and the exit code mirrors the
135
+ outcome — suitable for dropping into a CI pipeline that gates on
136
+ leaderboard publication:
137
+
138
+ ```bash
139
+ export SYNTHBENCH_API_KEY=sb_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx
140
+ synthbench run \
141
+ --provider raw-anthropic --model claude-haiku-4-5 \
142
+ --dataset globalopinionqa --samples 30 -n 100 \
143
+ --submit --wait \
144
+ --submit-message "first pass with new prompt template"
145
+ ```
146
+
147
+ Exit codes with `--wait`:
148
+
149
+ | Code | Meaning |
150
+ |-----:|---------|
151
+ | 0 | Published — result is live on the leaderboard |
152
+ | 1 | Rejected by the validator OR hard error (bad key, 5xx, etc.) |
153
+ | 2 | Poll deadline exceeded (validation is still running server-side; check [/account/submissions/](https://synthbench.org/account/submissions/)) |
154
+
155
+ Without `--wait`, the upload exits 0 as soon as the Worker accepts the
156
+ submission (status = `validating`) — useful for fire-and-forget runs
157
+ where you'll check the web dashboard later.
158
+
159
+ `--submit-message` is optional; it's stored alongside the uploaded JSON
160
+ so you can label experiments (e.g. "v2 prompt", "temp=1.0 sweep") without
161
+ touching your `config` fields and perturbing the `config_id` hash.
162
+
163
+ The Worker validates the submission, stages it to R2, and dispatches the
164
+ GitHub Actions pipeline. Successful runs publish within ~5 minutes. Keys
165
+ are rate-limited to 60 submissions/hour. See
166
+ [SUBMISSIONS.md → API key flow](SUBMISSIONS.md#api-key-flow-cli-submission).
167
+
168
+ ### 2. Web upload
169
+
170
+ Sign in at [/account](https://synthbench.org/account/) and drag your result
171
+ JSON into [/submit/upload](https://synthbench.org/submit/upload/). Same
172
+ validation pipeline, no key required.
173
+
174
+ ### 3. GitHub PR (power-user path)
175
+
176
+ 1. **Fork** this repo.
177
+ 2. **Run** SynthBench with your provider:
178
+ ```bash
179
+ synthbench run --provider <your-provider> --model <your-model> --suite full --output results/
180
+ ```
181
+ 3. **Validate locally** before opening a PR:
182
+ ```bash
183
+ synthbench validate results/<your-result>.json
184
+ ```
185
+ 4. **Copy** the result JSON into `leaderboard-results/`.
186
+ 5. **Open a PR** against this repo.
187
+ 6. **CI validates** schema, bounds, distributions, and recomputes every metric against the per-question data. Fabricated or inconsistent submissions are rejected.
188
+ 7. **Maintainers review and merge** — your results appear on the leaderboard.
189
+
190
+ See [SUBMISSIONS.md](SUBMISSIONS.md) for the full list of integrity checks and common failure modes.
191
+
192
+ ## Key Research Findings
193
+
194
+ Our benchmarking experiments across 3 models, 3 datasets, and 200+ runs reveal:
195
+
196
+ | Finding | Impact |
197
+ |---------|--------|
198
+ | **3-model ensemble hits SPS 0.90** | Equal-weight average of Haiku + Gemini + GPT-4o-mini beats any single model by +5-7 pts |
199
+ | **Temperature is model-specific** | Gemini benefits from high temp (+4.5%), Haiku is insensitive, GPT-4o-mini mild |
200
+ | **Demographic conditioning quantifies LLM bias** | Republican conditioning 2.4x stronger than Democrat — model defaults approximate liberal responses |
201
+ | **Persona template matters** | Default template beats stripped/broken templates by +11 SPS pts |
202
+
203
+ See [FINDINGS.md](FINDINGS.md) for the full experimental report with methodology, replications, and per-metric breakdowns.
204
+
205
+ ## Status
206
+
207
+ Phase 2 complete: Multi-model benchmarking, ensemble blending, temperature sweeps, and demographic conditioning analysis across OpinionsQA, SubPOP, and GlobalOpinionQA.
208
+
209
+ ## Ground Truth
210
+
211
+ Built on nine registered survey datasets. Each adapter declares a
212
+ redistribution policy; `full` ships `human_distribution` publicly, `gated`
213
+ routes per-question artifacts to a JWT-authenticated Cloudflare R2 origin,
214
+ and `aggregates_only` / `citation_only` contribute to leaderboard aggregates
215
+ only. Canonical source of truth is the `redistribution_policy` attribute on
216
+ each adapter in `src/synthbench/datasets/` (see
217
+ [`src/synthbench/datasets/policy.py`](src/synthbench/datasets/policy.py)).
218
+
219
+ | Dataset | Tier | Source |
220
+ |---------|------|--------|
221
+ | [OpinionsQA](https://github.com/tatsu-lab/opinions_qa) (Santurkar et al., ICML 2023) | gated | Pew American Trends Panel, 1,498 questions |
222
+ | [GlobalOpinionQA](https://arxiv.org/abs/2306.16388) (Durmus et al., 2024) | gated | Pew Global Attitudes, 138 countries |
223
+ | GSS (General Social Survey) | full | NORC, microdata-capable |
224
+ | NTIA Internet Use Supplement | full | US Census / NTIA |
225
+ | SubPOP | gated | 22 US subpopulations, 3,362 questions |
226
+ | WVS (World Values Survey) | gated | WVSA, cross-national |
227
+ | Eurobarometer | gated | European Commission |
228
+ | Michigan (Surveys of Consumers) | gated | U. of Michigan |
229
+ | Pew Technology | gated | Pew Research |
230
+
231
+ GSS and NTIA ship with full per-question distributions; the remaining seven
232
+ require a signed-in account to reach per-question payloads.
233
+
234
+ ## Cost tracking
235
+
236
+ The leaderboard JSON carries per-row cost fields and a top-level
237
+ `pricing_snapshot` object:
238
+
239
+ - Each row exposes `cost_usd`, `cost_per_100q`, `cost_per_sps_point`, and
240
+ `is_cost_estimated`. Ensemble rows sum `cost_usd` across constituent
241
+ runs listed in `config.ensemble_sources`.
242
+ - `pricing_snapshot` records the per-model `input_per_1m` / `output_per_1m`
243
+ rates used for this publish run, the `snapshot_date` anchor comment, and
244
+ the installed `synth_panel_version` that produced the rates.
245
+
246
+ This lets downstream consumers audit which pricing table produced which
247
+ `cost_usd` and reconcile against provider-reported billing without guessing
248
+ at rate drift. See [sb-x8t] and `src/synthbench/publish.py::_build_pricing_snapshot`.
249
+
250
+ ## Convergence analysis
251
+
252
+ `synthbench convergence bootstrap` computes theoretical ~1/√n convergence
253
+ curves for every question in a dataset by multinomial resampling from the
254
+ aggregate `human_distribution`. `synthbench convergence real` runs the same
255
+ curve shape over individual-level microdata (GSS today; WVS / Eurobarometer
256
+ microdata adapters are a follow-on). `synthbench convergence compare` emits
257
+ both curves side-by-side.
258
+
259
+ See [`docs/convergence-analysis.md`](docs/convergence-analysis.md) for the
260
+ JSON schema, CLI flags, and the `load_convergence_baseline` integration
261
+ surface that synthpanel's `--calibrate-against DATASET:QUESTION` flag
262
+ consumes.
263
+
264
+ ## Citation
265
+
266
+ If you use SynthBench in your research, please cite:
267
+
268
+ ```bibtex
269
+ @misc{synthbench2026,
270
+ title={SynthBench: Open Benchmark for Synthetic Survey Respondent Quality},
271
+ author={DataViking-Tech},
272
+ year={2026},
273
+ url={https://github.com/DataViking-Tech/synthbench}
274
+ }
275
+ ```
276
+
277
+ ## License
278
+
279
+ MIT
@@ -0,0 +1,241 @@
1
+ # SynthBench
2
+
3
+ [![CI](https://github.com/DataViking-Tech/synthbench/actions/workflows/ci.yml/badge.svg)](https://github.com/DataViking-Tech/synthbench/actions/workflows/ci.yml)
4
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
5
+ [![Python](https://img.shields.io/badge/python-3.10%20%7C%203.11%20%7C%203.12%20%7C%203.13-blue.svg)](pyproject.toml)
6
+ [![Leaderboard](https://img.shields.io/badge/leaderboard-live-success)](https://synthbench.org/)
7
+
8
+ Open benchmark harness for synthetic survey respondent quality.
9
+
10
+ **The MLPerf of synthetic UXR.**
11
+
12
+ SynthBench measures how well synthetic respondent systems (like [synthpanel](https://github.com/DataViking-Tech/SynthPanel), Ditto, Synthetic Users, or raw ChatGPT prompting) reproduce real human survey response patterns against real Pew American Trends Panel and GlobalOpinionQA ground truth — so "it sounds plausible" gets replaced with a measurable similarity score.
13
+
14
+ ## Quick Start
15
+
16
+ Run your first benchmark in 3 commands:
17
+
18
+ ```bash
19
+ pip install synthbench-eval
20
+ synthbench run --provider random --suite smoke --output results/
21
+ synthbench leaderboard --results-dir results/
22
+ ```
23
+
24
+ > **Note:** the distribution is named `synthbench-eval` — the bare `synthbench`
25
+ > name on PyPI belongs to an unrelated project. The import package and CLI are
26
+ > still `synthbench`. For development, clone this repo and `pip install -e .`.
27
+ > API-backed providers need an extra, e.g. `pip install "synthbench-eval[openai]"`
28
+ > for the `openrouter` / `raw-openai` / `raw-gemini` / `ollama` providers.
29
+
30
+ Try with a real model (requires API key):
31
+
32
+ ```bash
33
+ export OPENROUTER_API_KEY=your-key
34
+ synthbench run --provider openrouter --model openai/gpt-4o-mini --suite core --samples 50
35
+ ```
36
+
37
+ See [`notebooks/getting_started.ipynb`](notebooks/getting_started.ipynb) for a guided walkthrough.
38
+
39
+ ## Leaderboard
40
+
41
+ **[View the live leaderboard](https://synthbench.org/)** — see also the
42
+ [methodology](https://synthbench.org/methodology/) and
43
+ [findings](https://synthbench.org/findings/) pages.
44
+
45
+ Regenerate leaderboard data for the Astro site:
46
+ ```bash
47
+ synthbench publish-data --results-dir ./leaderboard-results --output site/src/data/leaderboard.json
48
+ ```
49
+
50
+ ### Contributor note: gated data publishing
51
+
52
+ Most contributors do **not** need to publish gated artifacts. If you're running
53
+ benchmarks locally or contributing via PR, focus on `synthbench run`,
54
+ `synthbench validate`, and result submission.
55
+
56
+ The gated data publication path is maintainer infrastructure and is handled by
57
+ project deployment workflows.
58
+
59
+ ## Development
60
+
61
+ After cloning, enable the repo-tracked git hooks so pushes that would fail CI's
62
+ `ruff format --check` are caught locally:
63
+
64
+ ```bash
65
+ ./scripts/install-hooks.sh # one-time: wires .githooks/ via core.hooksPath
66
+ ```
67
+
68
+ The `pre-push` hook only checks Python files changed in the commits being
69
+ pushed, so already-formatted branches add no meaningful overhead. Run
70
+ `./scripts/format-check.sh` anytime to mirror the full CI lint job. Emergency
71
+ bypass: `git push --no-verify`.
72
+
73
+ For contribution workflow and PR expectations, see
74
+ [`CONTRIBUTING.md`](CONTRIBUTING.md).
75
+
76
+ ## Submit Results
77
+
78
+ Three ways to land a run on the leaderboard, in order of friction:
79
+
80
+ ### 1. CLI (recommended for repeat submissions)
81
+
82
+ Mint an API key at [synthbench.org/account](https://synthbench.org/account/),
83
+ then:
84
+
85
+ ```bash
86
+ export SYNTHBENCH_API_KEY=sb_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx
87
+ synthbench run --provider openrouter --model gpt-4o-mini --suite full -o results/
88
+ synthbench submit results/openrouter_gpt-4o-mini_opinionsqa.json
89
+ ```
90
+
91
+ #### End-to-end: run + submit in one command (`--submit`)
92
+
93
+ Collapse the two steps above into a single invocation. The CLI saves the
94
+ result JSON locally (so a validation rejection doesn't lose your run) and
95
+ then POSTs it to the leaderboard. With `--wait`, the process blocks until
96
+ the validator reaches a terminal state and the exit code mirrors the
97
+ outcome — suitable for dropping into a CI pipeline that gates on
98
+ leaderboard publication:
99
+
100
+ ```bash
101
+ export SYNTHBENCH_API_KEY=sb_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx
102
+ synthbench run \
103
+ --provider raw-anthropic --model claude-haiku-4-5 \
104
+ --dataset globalopinionqa --samples 30 -n 100 \
105
+ --submit --wait \
106
+ --submit-message "first pass with new prompt template"
107
+ ```
108
+
109
+ Exit codes with `--wait`:
110
+
111
+ | Code | Meaning |
112
+ |-----:|---------|
113
+ | 0 | Published — result is live on the leaderboard |
114
+ | 1 | Rejected by the validator OR hard error (bad key, 5xx, etc.) |
115
+ | 2 | Poll deadline exceeded (validation is still running server-side; check [/account/submissions/](https://synthbench.org/account/submissions/)) |
116
+
117
+ Without `--wait`, the upload exits 0 as soon as the Worker accepts the
118
+ submission (status = `validating`) — useful for fire-and-forget runs
119
+ where you'll check the web dashboard later.
120
+
121
+ `--submit-message` is optional; it's stored alongside the uploaded JSON
122
+ so you can label experiments (e.g. "v2 prompt", "temp=1.0 sweep") without
123
+ touching your `config` fields and perturbing the `config_id` hash.
124
+
125
+ The Worker validates the submission, stages it to R2, and dispatches the
126
+ GitHub Actions pipeline. Successful runs publish within ~5 minutes. Keys
127
+ are rate-limited to 60 submissions/hour. See
128
+ [SUBMISSIONS.md → API key flow](SUBMISSIONS.md#api-key-flow-cli-submission).
129
+
130
+ ### 2. Web upload
131
+
132
+ Sign in at [/account](https://synthbench.org/account/) and drag your result
133
+ JSON into [/submit/upload](https://synthbench.org/submit/upload/). Same
134
+ validation pipeline, no key required.
135
+
136
+ ### 3. GitHub PR (power-user path)
137
+
138
+ 1. **Fork** this repo.
139
+ 2. **Run** SynthBench with your provider:
140
+ ```bash
141
+ synthbench run --provider <your-provider> --model <your-model> --suite full --output results/
142
+ ```
143
+ 3. **Validate locally** before opening a PR:
144
+ ```bash
145
+ synthbench validate results/<your-result>.json
146
+ ```
147
+ 4. **Copy** the result JSON into `leaderboard-results/`.
148
+ 5. **Open a PR** against this repo.
149
+ 6. **CI validates** schema, bounds, distributions, and recomputes every metric against the per-question data. Fabricated or inconsistent submissions are rejected.
150
+ 7. **Maintainers review and merge** — your results appear on the leaderboard.
151
+
152
+ See [SUBMISSIONS.md](SUBMISSIONS.md) for the full list of integrity checks and common failure modes.
153
+
154
+ ## Key Research Findings
155
+
156
+ Our benchmarking experiments across 3 models, 3 datasets, and 200+ runs reveal:
157
+
158
+ | Finding | Impact |
159
+ |---------|--------|
160
+ | **3-model ensemble hits SPS 0.90** | Equal-weight average of Haiku + Gemini + GPT-4o-mini beats any single model by +5-7 pts |
161
+ | **Temperature is model-specific** | Gemini benefits from high temp (+4.5%), Haiku is insensitive, GPT-4o-mini mild |
162
+ | **Demographic conditioning quantifies LLM bias** | Republican conditioning 2.4x stronger than Democrat — model defaults approximate liberal responses |
163
+ | **Persona template matters** | Default template beats stripped/broken templates by +11 SPS pts |
164
+
165
+ See [FINDINGS.md](FINDINGS.md) for the full experimental report with methodology, replications, and per-metric breakdowns.
166
+
167
+ ## Status
168
+
169
+ Phase 2 complete: Multi-model benchmarking, ensemble blending, temperature sweeps, and demographic conditioning analysis across OpinionsQA, SubPOP, and GlobalOpinionQA.
170
+
171
+ ## Ground Truth
172
+
173
+ Built on nine registered survey datasets. Each adapter declares a
174
+ redistribution policy; `full` ships `human_distribution` publicly, `gated`
175
+ routes per-question artifacts to a JWT-authenticated Cloudflare R2 origin,
176
+ and `aggregates_only` / `citation_only` contribute to leaderboard aggregates
177
+ only. Canonical source of truth is the `redistribution_policy` attribute on
178
+ each adapter in `src/synthbench/datasets/` (see
179
+ [`src/synthbench/datasets/policy.py`](src/synthbench/datasets/policy.py)).
180
+
181
+ | Dataset | Tier | Source |
182
+ |---------|------|--------|
183
+ | [OpinionsQA](https://github.com/tatsu-lab/opinions_qa) (Santurkar et al., ICML 2023) | gated | Pew American Trends Panel, 1,498 questions |
184
+ | [GlobalOpinionQA](https://arxiv.org/abs/2306.16388) (Durmus et al., 2024) | gated | Pew Global Attitudes, 138 countries |
185
+ | GSS (General Social Survey) | full | NORC, microdata-capable |
186
+ | NTIA Internet Use Supplement | full | US Census / NTIA |
187
+ | SubPOP | gated | 22 US subpopulations, 3,362 questions |
188
+ | WVS (World Values Survey) | gated | WVSA, cross-national |
189
+ | Eurobarometer | gated | European Commission |
190
+ | Michigan (Surveys of Consumers) | gated | U. of Michigan |
191
+ | Pew Technology | gated | Pew Research |
192
+
193
+ GSS and NTIA ship with full per-question distributions; the remaining seven
194
+ require a signed-in account to reach per-question payloads.
195
+
196
+ ## Cost tracking
197
+
198
+ The leaderboard JSON carries per-row cost fields and a top-level
199
+ `pricing_snapshot` object:
200
+
201
+ - Each row exposes `cost_usd`, `cost_per_100q`, `cost_per_sps_point`, and
202
+ `is_cost_estimated`. Ensemble rows sum `cost_usd` across constituent
203
+ runs listed in `config.ensemble_sources`.
204
+ - `pricing_snapshot` records the per-model `input_per_1m` / `output_per_1m`
205
+ rates used for this publish run, the `snapshot_date` anchor comment, and
206
+ the installed `synth_panel_version` that produced the rates.
207
+
208
+ This lets downstream consumers audit which pricing table produced which
209
+ `cost_usd` and reconcile against provider-reported billing without guessing
210
+ at rate drift. See [sb-x8t] and `src/synthbench/publish.py::_build_pricing_snapshot`.
211
+
212
+ ## Convergence analysis
213
+
214
+ `synthbench convergence bootstrap` computes theoretical ~1/√n convergence
215
+ curves for every question in a dataset by multinomial resampling from the
216
+ aggregate `human_distribution`. `synthbench convergence real` runs the same
217
+ curve shape over individual-level microdata (GSS today; WVS / Eurobarometer
218
+ microdata adapters are a follow-on). `synthbench convergence compare` emits
219
+ both curves side-by-side.
220
+
221
+ See [`docs/convergence-analysis.md`](docs/convergence-analysis.md) for the
222
+ JSON schema, CLI flags, and the `load_convergence_baseline` integration
223
+ surface that synthpanel's `--calibrate-against DATASET:QUESTION` flag
224
+ consumes.
225
+
226
+ ## Citation
227
+
228
+ If you use SynthBench in your research, please cite:
229
+
230
+ ```bibtex
231
+ @misc{synthbench2026,
232
+ title={SynthBench: Open Benchmark for Synthetic Survey Respondent Quality},
233
+ author={DataViking-Tech},
234
+ year={2026},
235
+ url={https://github.com/DataViking-Tech/synthbench}
236
+ }
237
+ ```
238
+
239
+ ## License
240
+
241
+ MIT
@@ -0,0 +1,79 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ # Distribution name is synthbench-eval: the bare "synthbench" name on PyPI
7
+ # belongs to an unrelated project. Import package and CLI remain `synthbench`.
8
+ name = "synthbench-eval"
9
+ # Placeholder — the publish job stamps the real version from the release tag
10
+ # at build time (see publish-pypi in .github/workflows/auto-tag.yml).
11
+ version = "0.4.0"
12
+ description = "Open benchmark harness for synthetic survey respondent quality"
13
+ readme = "README.md"
14
+ license = "MIT"
15
+ requires-python = ">=3.9"
16
+ keywords = [
17
+ "benchmark",
18
+ "survey",
19
+ "synthetic-data",
20
+ "llm",
21
+ "evaluation",
22
+ "opinionsqa",
23
+ "globalopinionqa",
24
+ "gss",
25
+ "convergence",
26
+ "synthpanel",
27
+ ]
28
+ dependencies = [
29
+ "click>=8.0",
30
+ "httpx>=0.24",
31
+ "numpy>=1.24",
32
+ "pyyaml>=6.0",
33
+ "scipy>=1.10",
34
+ "synthpanel>=0.2.0",
35
+ ]
36
+
37
+ [project.optional-dependencies]
38
+ anthropic = ["anthropic>=0.18"]
39
+ openai = ["openai>=1.0"]
40
+ viz = ["matplotlib>=3.7"]
41
+ hf = ["datasets>=2.14"]
42
+ subpop = ["datasets>=2.14"]
43
+ # R2 publish target for gated-tier per-question/run/config artifacts (sb-sjs).
44
+ # Only required in environments that actually upload (CI publish step); local
45
+ # dev defaults to writing to site/public/data without R2.
46
+ r2 = ["boto3>=1.34"]
47
+ all = [
48
+ "anthropic>=0.18",
49
+ "openai>=1.0",
50
+ "matplotlib>=3.7",
51
+ "datasets>=2.14",
52
+ "boto3>=1.34",
53
+ ]
54
+ dev = ["pytest>=7.0", "pytest-asyncio>=0.21"]
55
+
56
+ [project.scripts]
57
+ synthbench = "synthbench.cli:main"
58
+
59
+ [tool.setuptools.packages.find]
60
+ where = ["src"]
61
+
62
+ # Holdout answer-key data (issue #259) is intentionally NOT included in
63
+ # the wheel. Raw upstream survey items are downloaded at runtime by the
64
+ # dataset adapters into the user's data directory, never from the package
65
+ # install. The public / private-holdout partition is a logical filter over
66
+ # question keys (see ``synthbench.private_holdout``), and the canonical
67
+ # private *answer key* (per-question human distributions) lives only in
68
+ # the gated R2 origin + credentialed maintainer caches. We therefore
69
+ # deliberately leave ``[tool.setuptools.package-data]`` unset for
70
+ # synthbench — the only files shipped are the Python sources resolved by
71
+ # ``packages.find``. Adding raw data globs here would break the holdout
72
+ # integrity model. See ``docs/held-out.md``.
73
+
74
+ [tool.pytest.ini_options]
75
+ asyncio_mode = "auto"
76
+ testpaths = ["tests"]
77
+ # Prepend local src/ to sys.path so the in-tree synthbench package shadows any
78
+ # stale `pip install -e .` entries from sibling worktrees (sb-3bn).
79
+ pythonpath = ["src"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,16 @@
1
+ """SynthBench — open benchmark harness for synthetic survey respondent quality."""
2
+
3
+ from synthbench.convergence.baseline import (
4
+ BaselineGatedError,
5
+ BaselineUnavailable,
6
+ load_convergence_baseline,
7
+ )
8
+
9
+ __version__ = "0.4.0"
10
+
11
+ __all__ = [
12
+ "BaselineGatedError",
13
+ "BaselineUnavailable",
14
+ "load_convergence_baseline",
15
+ "__version__",
16
+ ]
@@ -0,0 +1,5 @@
1
+ """Allow running as `python -m synthbench`."""
2
+
3
+ from synthbench.cli import main
4
+
5
+ main()