synthbench-eval 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- synthbench_eval-0.4.0/LICENSE +21 -0
- synthbench_eval-0.4.0/PKG-INFO +279 -0
- synthbench_eval-0.4.0/README.md +241 -0
- synthbench_eval-0.4.0/pyproject.toml +79 -0
- synthbench_eval-0.4.0/setup.cfg +4 -0
- synthbench_eval-0.4.0/src/synthbench/__init__.py +16 -0
- synthbench_eval-0.4.0/src/synthbench/__main__.py +5 -0
- synthbench_eval-0.4.0/src/synthbench/adapter.py +131 -0
- synthbench_eval-0.4.0/src/synthbench/anomaly.py +503 -0
- synthbench_eval-0.4.0/src/synthbench/baseline_floors.py +213 -0
- synthbench_eval-0.4.0/src/synthbench/baselines.py +902 -0
- synthbench_eval-0.4.0/src/synthbench/cli.py +2929 -0
- synthbench_eval-0.4.0/src/synthbench/config_id.py +424 -0
- synthbench_eval-0.4.0/src/synthbench/contamination.py +783 -0
- synthbench_eval-0.4.0/src/synthbench/convergence/__init__.py +67 -0
- synthbench_eval-0.4.0/src/synthbench/convergence/baseline.py +172 -0
- synthbench_eval-0.4.0/src/synthbench/convergence/bootstrap.py +60 -0
- synthbench_eval-0.4.0/src/synthbench/convergence/cli_report.py +556 -0
- synthbench_eval-0.4.0/src/synthbench/convergence/curves.py +111 -0
- synthbench_eval-0.4.0/src/synthbench/convergence/real_sampling.py +146 -0
- synthbench_eval-0.4.0/src/synthbench/convergence/thresholds.py +49 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/__init__.py +37 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/base.py +147 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/eurobarometer.py +335 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/globalopinionqa.py +252 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/gss.py +323 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/michigan.py +505 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/ntia.py +357 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/opinionsqa.py +412 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/pewtech.py +334 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/policy.py +142 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/subpop.py +302 -0
- synthbench_eval-0.4.0/src/synthbench/datasets/wvs.py +229 -0
- synthbench_eval-0.4.0/src/synthbench/findings.py +988 -0
- synthbench_eval-0.4.0/src/synthbench/holdout.py +347 -0
- synthbench_eval-0.4.0/src/synthbench/human_distributions.py +243 -0
- synthbench_eval-0.4.0/src/synthbench/leaderboard.py +715 -0
- synthbench_eval-0.4.0/src/synthbench/leaderboard_pr.py +274 -0
- synthbench_eval-0.4.0/src/synthbench/metrics/__init__.py +38 -0
- synthbench_eval-0.4.0/src/synthbench/metrics/composite.py +72 -0
- synthbench_eval-0.4.0/src/synthbench/metrics/conditioning.py +39 -0
- synthbench_eval-0.4.0/src/synthbench/metrics/distributional.py +41 -0
- synthbench_eval-0.4.0/src/synthbench/metrics/ranking.py +37 -0
- synthbench_eval-0.4.0/src/synthbench/metrics/refusal.py +270 -0
- synthbench_eval-0.4.0/src/synthbench/metrics/subgroup.py +58 -0
- synthbench_eval-0.4.0/src/synthbench/private_holdout.py +240 -0
- synthbench_eval-0.4.0/src/synthbench/providers/__init__.py +43 -0
- synthbench_eval-0.4.0/src/synthbench/providers/_parsing.py +150 -0
- synthbench_eval-0.4.0/src/synthbench/providers/_retry.py +107 -0
- synthbench_eval-0.4.0/src/synthbench/providers/base.py +212 -0
- synthbench_eval-0.4.0/src/synthbench/providers/http.py +108 -0
- synthbench_eval-0.4.0/src/synthbench/providers/majority_baseline.py +23 -0
- synthbench_eval-0.4.0/src/synthbench/providers/ollama.py +109 -0
- synthbench_eval-0.4.0/src/synthbench/providers/openrouter.py +144 -0
- synthbench_eval-0.4.0/src/synthbench/providers/population_baseline.py +87 -0
- synthbench_eval-0.4.0/src/synthbench/providers/random_baseline.py +24 -0
- synthbench_eval-0.4.0/src/synthbench/providers/raw_anthropic.py +149 -0
- synthbench_eval-0.4.0/src/synthbench/providers/raw_gemini.py +145 -0
- synthbench_eval-0.4.0/src/synthbench/providers/raw_openai.py +139 -0
- synthbench_eval-0.4.0/src/synthbench/providers/synthpanel.py +978 -0
- synthbench_eval-0.4.0/src/synthbench/publish.py +2772 -0
- synthbench_eval-0.4.0/src/synthbench/r2_upload.py +178 -0
- synthbench_eval-0.4.0/src/synthbench/recompute.py +235 -0
- synthbench_eval-0.4.0/src/synthbench/report.py +493 -0
- synthbench_eval-0.4.0/src/synthbench/run_hash.py +111 -0
- synthbench_eval-0.4.0/src/synthbench/run_validity.py +232 -0
- synthbench_eval-0.4.0/src/synthbench/runner.py +834 -0
- synthbench_eval-0.4.0/src/synthbench/stats.py +1492 -0
- synthbench_eval-0.4.0/src/synthbench/submission.py +284 -0
- synthbench_eval-0.4.0/src/synthbench/submission_pr.py +436 -0
- synthbench_eval-0.4.0/src/synthbench/submit_adapter.py +748 -0
- synthbench_eval-0.4.0/src/synthbench/suite.py +424 -0
- synthbench_eval-0.4.0/src/synthbench/suites/__init__.py +94 -0
- synthbench_eval-0.4.0/src/synthbench/topics.py +256 -0
- synthbench_eval-0.4.0/src/synthbench/user_config.py +409 -0
- synthbench_eval-0.4.0/src/synthbench/validation.py +1556 -0
- synthbench_eval-0.4.0/src/synthbench/visualize.py +455 -0
- synthbench_eval-0.4.0/src/synthbench_eval.egg-info/PKG-INFO +279 -0
- synthbench_eval-0.4.0/src/synthbench_eval.egg-info/SOURCES.txt +147 -0
- synthbench_eval-0.4.0/src/synthbench_eval.egg-info/dependency_links.txt +1 -0
- synthbench_eval-0.4.0/src/synthbench_eval.egg-info/entry_points.txt +2 -0
- synthbench_eval-0.4.0/src/synthbench_eval.egg-info/requires.txt +35 -0
- synthbench_eval-0.4.0/src/synthbench_eval.egg-info/top_level.txt +1 -0
- synthbench_eval-0.4.0/tests/test_anomaly.py +369 -0
- synthbench_eval-0.4.0/tests/test_baseline_floors.py +356 -0
- synthbench_eval-0.4.0/tests/test_baselines.py +572 -0
- synthbench_eval-0.4.0/tests/test_cli_run_submit.py +415 -0
- synthbench_eval-0.4.0/tests/test_cli_submit.py +223 -0
- synthbench_eval-0.4.0/tests/test_cli_submit_adapter.py +543 -0
- synthbench_eval-0.4.0/tests/test_config_id.py +396 -0
- synthbench_eval-0.4.0/tests/test_contamination.py +694 -0
- synthbench_eval-0.4.0/tests/test_convergence_baseline.py +245 -0
- synthbench_eval-0.4.0/tests/test_convergence_bootstrap.py +385 -0
- synthbench_eval-0.4.0/tests/test_datasets_eurobarometer.py +201 -0
- synthbench_eval-0.4.0/tests/test_datasets_globalopinionqa.py +99 -0
- synthbench_eval-0.4.0/tests/test_datasets_gss.py +246 -0
- synthbench_eval-0.4.0/tests/test_datasets_michigan.py +321 -0
- synthbench_eval-0.4.0/tests/test_datasets_ntia.py +400 -0
- synthbench_eval-0.4.0/tests/test_datasets_policy.py +219 -0
- synthbench_eval-0.4.0/tests/test_datasets_wvs.py +283 -0
- synthbench_eval-0.4.0/tests/test_effort.py +465 -0
- synthbench_eval-0.4.0/tests/test_findings_drift.py +114 -0
- synthbench_eval-0.4.0/tests/test_findings_elicitation.py +140 -0
- synthbench_eval-0.4.0/tests/test_findings_nonresponse.py +108 -0
- synthbench_eval-0.4.0/tests/test_holdout.py +294 -0
- synthbench_eval-0.4.0/tests/test_human_distributions.py +87 -0
- synthbench_eval-0.4.0/tests/test_integration_tokens.py +140 -0
- synthbench_eval-0.4.0/tests/test_metrics.py +400 -0
- synthbench_eval-0.4.0/tests/test_microdata.py +376 -0
- synthbench_eval-0.4.0/tests/test_private_holdout.py +202 -0
- synthbench_eval-0.4.0/tests/test_provider_parsing.py +939 -0
- synthbench_eval-0.4.0/tests/test_providers.py +214 -0
- synthbench_eval-0.4.0/tests/test_publish_config_id_consistency.py +240 -0
- synthbench_eval-0.4.0/tests/test_publish_cost.py +697 -0
- synthbench_eval-0.4.0/tests/test_publish_cross_provider_jsd.py +246 -0
- synthbench_eval-0.4.0/tests/test_publish_demographic_scorecard.py +137 -0
- synthbench_eval-0.4.0/tests/test_publish_gated_fail_closed.py +277 -0
- synthbench_eval-0.4.0/tests/test_publish_holdout.py +290 -0
- synthbench_eval-0.4.0/tests/test_publish_invalid_runs.py +232 -0
- synthbench_eval-0.4.0/tests/test_publish_normalized.py +161 -0
- synthbench_eval-0.4.0/tests/test_publish_policy.py +245 -0
- synthbench_eval-0.4.0/tests/test_publish_questions.py +683 -0
- synthbench_eval-0.4.0/tests/test_publish_r2_routing.py +317 -0
- synthbench_eval-0.4.0/tests/test_publish_rehydration.py +165 -0
- synthbench_eval-0.4.0/tests/test_publish_run_counts.py +157 -0
- synthbench_eval-0.4.0/tests/test_publish_runnable_ids.py +153 -0
- synthbench_eval-0.4.0/tests/test_publish_text_rehydration.py +101 -0
- synthbench_eval-0.4.0/tests/test_publish_topic_metrics.py +109 -0
- synthbench_eval-0.4.0/tests/test_r2_upload.py +189 -0
- synthbench_eval-0.4.0/tests/test_recompute_integrity.py +488 -0
- synthbench_eval-0.4.0/tests/test_report.py +162 -0
- synthbench_eval-0.4.0/tests/test_run_hash.py +271 -0
- synthbench_eval-0.4.0/tests/test_run_validity.py +210 -0
- synthbench_eval-0.4.0/tests/test_runner.py +662 -0
- synthbench_eval-0.4.0/tests/test_site_dataset_cards.py +100 -0
- synthbench_eval-0.4.0/tests/test_stats.py +150 -0
- synthbench_eval-0.4.0/tests/test_stats_golden.py +139 -0
- synthbench_eval-0.4.0/tests/test_strip_gated_guard.py +113 -0
- synthbench_eval-0.4.0/tests/test_submission.py +364 -0
- synthbench_eval-0.4.0/tests/test_submission_pr.py +367 -0
- synthbench_eval-0.4.0/tests/test_suite.py +325 -0
- synthbench_eval-0.4.0/tests/test_suites.py +96 -0
- synthbench_eval-0.4.0/tests/test_suites_novel_products.py +127 -0
- synthbench_eval-0.4.0/tests/test_topics.py +184 -0
- synthbench_eval-0.4.0/tests/test_user_config.py +503 -0
- synthbench_eval-0.4.0/tests/test_validation.py +991 -0
- synthbench_eval-0.4.0/tests/test_validation_holdout.py +124 -0
- synthbench_eval-0.4.0/tests/test_validation_stripped.py +184 -0
- synthbench_eval-0.4.0/tests/test_verify_publish_integrity.py +157 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 DataViking Tech
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: synthbench-eval
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Open benchmark harness for synthetic survey respondent quality
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Keywords: benchmark,survey,synthetic-data,llm,evaluation,opinionsqa,globalopinionqa,gss,convergence,synthpanel
|
|
7
|
+
Requires-Python: >=3.9
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: click>=8.0
|
|
11
|
+
Requires-Dist: httpx>=0.24
|
|
12
|
+
Requires-Dist: numpy>=1.24
|
|
13
|
+
Requires-Dist: pyyaml>=6.0
|
|
14
|
+
Requires-Dist: scipy>=1.10
|
|
15
|
+
Requires-Dist: synthpanel>=0.2.0
|
|
16
|
+
Provides-Extra: anthropic
|
|
17
|
+
Requires-Dist: anthropic>=0.18; extra == "anthropic"
|
|
18
|
+
Provides-Extra: openai
|
|
19
|
+
Requires-Dist: openai>=1.0; extra == "openai"
|
|
20
|
+
Provides-Extra: viz
|
|
21
|
+
Requires-Dist: matplotlib>=3.7; extra == "viz"
|
|
22
|
+
Provides-Extra: hf
|
|
23
|
+
Requires-Dist: datasets>=2.14; extra == "hf"
|
|
24
|
+
Provides-Extra: subpop
|
|
25
|
+
Requires-Dist: datasets>=2.14; extra == "subpop"
|
|
26
|
+
Provides-Extra: r2
|
|
27
|
+
Requires-Dist: boto3>=1.34; extra == "r2"
|
|
28
|
+
Provides-Extra: all
|
|
29
|
+
Requires-Dist: anthropic>=0.18; extra == "all"
|
|
30
|
+
Requires-Dist: openai>=1.0; extra == "all"
|
|
31
|
+
Requires-Dist: matplotlib>=3.7; extra == "all"
|
|
32
|
+
Requires-Dist: datasets>=2.14; extra == "all"
|
|
33
|
+
Requires-Dist: boto3>=1.34; extra == "all"
|
|
34
|
+
Provides-Extra: dev
|
|
35
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
36
|
+
Requires-Dist: pytest-asyncio>=0.21; extra == "dev"
|
|
37
|
+
Dynamic: license-file
|
|
38
|
+
|
|
39
|
+
# SynthBench
|
|
40
|
+
|
|
41
|
+
[](https://github.com/DataViking-Tech/synthbench/actions/workflows/ci.yml)
|
|
42
|
+
[](LICENSE)
|
|
43
|
+
[](pyproject.toml)
|
|
44
|
+
[](https://synthbench.org/)
|
|
45
|
+
|
|
46
|
+
Open benchmark harness for synthetic survey respondent quality.
|
|
47
|
+
|
|
48
|
+
**The MLPerf of synthetic UXR.**
|
|
49
|
+
|
|
50
|
+
SynthBench measures how well synthetic respondent systems (like [synthpanel](https://github.com/DataViking-Tech/SynthPanel), Ditto, Synthetic Users, or raw ChatGPT prompting) reproduce real human survey response patterns against real Pew American Trends Panel and GlobalOpinionQA ground truth — so "it sounds plausible" gets replaced with a measurable similarity score.
|
|
51
|
+
|
|
52
|
+
## Quick Start
|
|
53
|
+
|
|
54
|
+
Run your first benchmark in 3 commands:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
pip install synthbench-eval
|
|
58
|
+
synthbench run --provider random --suite smoke --output results/
|
|
59
|
+
synthbench leaderboard --results-dir results/
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
> **Note:** the distribution is named `synthbench-eval` — the bare `synthbench`
|
|
63
|
+
> name on PyPI belongs to an unrelated project. The import package and CLI are
|
|
64
|
+
> still `synthbench`. For development, clone this repo and `pip install -e .`.
|
|
65
|
+
> API-backed providers need an extra, e.g. `pip install "synthbench-eval[openai]"`
|
|
66
|
+
> for the `openrouter` / `raw-openai` / `raw-gemini` / `ollama` providers.
|
|
67
|
+
|
|
68
|
+
Try with a real model (requires API key):
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
export OPENROUTER_API_KEY=your-key
|
|
72
|
+
synthbench run --provider openrouter --model openai/gpt-4o-mini --suite core --samples 50
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
See [`notebooks/getting_started.ipynb`](notebooks/getting_started.ipynb) for a guided walkthrough.
|
|
76
|
+
|
|
77
|
+
## Leaderboard
|
|
78
|
+
|
|
79
|
+
**[View the live leaderboard](https://synthbench.org/)** — see also the
|
|
80
|
+
[methodology](https://synthbench.org/methodology/) and
|
|
81
|
+
[findings](https://synthbench.org/findings/) pages.
|
|
82
|
+
|
|
83
|
+
Regenerate leaderboard data for the Astro site:
|
|
84
|
+
```bash
|
|
85
|
+
synthbench publish-data --results-dir ./leaderboard-results --output site/src/data/leaderboard.json
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
### Contributor note: gated data publishing
|
|
89
|
+
|
|
90
|
+
Most contributors do **not** need to publish gated artifacts. If you're running
|
|
91
|
+
benchmarks locally or contributing via PR, focus on `synthbench run`,
|
|
92
|
+
`synthbench validate`, and result submission.
|
|
93
|
+
|
|
94
|
+
The gated data publication path is maintainer infrastructure and is handled by
|
|
95
|
+
project deployment workflows.
|
|
96
|
+
|
|
97
|
+
## Development
|
|
98
|
+
|
|
99
|
+
After cloning, enable the repo-tracked git hooks so pushes that would fail CI's
|
|
100
|
+
`ruff format --check` are caught locally:
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
./scripts/install-hooks.sh # one-time: wires .githooks/ via core.hooksPath
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
The `pre-push` hook only checks Python files changed in the commits being
|
|
107
|
+
pushed, so already-formatted branches add no meaningful overhead. Run
|
|
108
|
+
`./scripts/format-check.sh` anytime to mirror the full CI lint job. Emergency
|
|
109
|
+
bypass: `git push --no-verify`.
|
|
110
|
+
|
|
111
|
+
For contribution workflow and PR expectations, see
|
|
112
|
+
[`CONTRIBUTING.md`](CONTRIBUTING.md).
|
|
113
|
+
|
|
114
|
+
## Submit Results
|
|
115
|
+
|
|
116
|
+
Three ways to land a run on the leaderboard, in order of friction:
|
|
117
|
+
|
|
118
|
+
### 1. CLI (recommended for repeat submissions)
|
|
119
|
+
|
|
120
|
+
Mint an API key at [synthbench.org/account](https://synthbench.org/account/),
|
|
121
|
+
then:
|
|
122
|
+
|
|
123
|
+
```bash
|
|
124
|
+
export SYNTHBENCH_API_KEY=sb_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx
|
|
125
|
+
synthbench run --provider openrouter --model gpt-4o-mini --suite full -o results/
|
|
126
|
+
synthbench submit results/openrouter_gpt-4o-mini_opinionsqa.json
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
#### End-to-end: run + submit in one command (`--submit`)
|
|
130
|
+
|
|
131
|
+
Collapse the two steps above into a single invocation. The CLI saves the
|
|
132
|
+
result JSON locally (so a validation rejection doesn't lose your run) and
|
|
133
|
+
then POSTs it to the leaderboard. With `--wait`, the process blocks until
|
|
134
|
+
the validator reaches a terminal state and the exit code mirrors the
|
|
135
|
+
outcome — suitable for dropping into a CI pipeline that gates on
|
|
136
|
+
leaderboard publication:
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
export SYNTHBENCH_API_KEY=sb_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx
|
|
140
|
+
synthbench run \
|
|
141
|
+
--provider raw-anthropic --model claude-haiku-4-5 \
|
|
142
|
+
--dataset globalopinionqa --samples 30 -n 100 \
|
|
143
|
+
--submit --wait \
|
|
144
|
+
--submit-message "first pass with new prompt template"
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
Exit codes with `--wait`:
|
|
148
|
+
|
|
149
|
+
| Code | Meaning |
|
|
150
|
+
|-----:|---------|
|
|
151
|
+
| 0 | Published — result is live on the leaderboard |
|
|
152
|
+
| 1 | Rejected by the validator OR hard error (bad key, 5xx, etc.) |
|
|
153
|
+
| 2 | Poll deadline exceeded (validation is still running server-side; check [/account/submissions/](https://synthbench.org/account/submissions/)) |
|
|
154
|
+
|
|
155
|
+
Without `--wait`, the upload exits 0 as soon as the Worker accepts the
|
|
156
|
+
submission (status = `validating`) — useful for fire-and-forget runs
|
|
157
|
+
where you'll check the web dashboard later.
|
|
158
|
+
|
|
159
|
+
`--submit-message` is optional; it's stored alongside the uploaded JSON
|
|
160
|
+
so you can label experiments (e.g. "v2 prompt", "temp=1.0 sweep") without
|
|
161
|
+
touching your `config` fields and perturbing the `config_id` hash.
|
|
162
|
+
|
|
163
|
+
The Worker validates the submission, stages it to R2, and dispatches the
|
|
164
|
+
GitHub Actions pipeline. Successful runs publish within ~5 minutes. Keys
|
|
165
|
+
are rate-limited to 60 submissions/hour. See
|
|
166
|
+
[SUBMISSIONS.md → API key flow](SUBMISSIONS.md#api-key-flow-cli-submission).
|
|
167
|
+
|
|
168
|
+
### 2. Web upload
|
|
169
|
+
|
|
170
|
+
Sign in at [/account](https://synthbench.org/account/) and drag your result
|
|
171
|
+
JSON into [/submit/upload](https://synthbench.org/submit/upload/). Same
|
|
172
|
+
validation pipeline, no key required.
|
|
173
|
+
|
|
174
|
+
### 3. GitHub PR (power-user path)
|
|
175
|
+
|
|
176
|
+
1. **Fork** this repo.
|
|
177
|
+
2. **Run** SynthBench with your provider:
|
|
178
|
+
```bash
|
|
179
|
+
synthbench run --provider <your-provider> --model <your-model> --suite full --output results/
|
|
180
|
+
```
|
|
181
|
+
3. **Validate locally** before opening a PR:
|
|
182
|
+
```bash
|
|
183
|
+
synthbench validate results/<your-result>.json
|
|
184
|
+
```
|
|
185
|
+
4. **Copy** the result JSON into `leaderboard-results/`.
|
|
186
|
+
5. **Open a PR** against this repo.
|
|
187
|
+
6. **CI validates** schema, bounds, distributions, and recomputes every metric against the per-question data. Fabricated or inconsistent submissions are rejected.
|
|
188
|
+
7. **Maintainers review and merge** — your results appear on the leaderboard.
|
|
189
|
+
|
|
190
|
+
See [SUBMISSIONS.md](SUBMISSIONS.md) for the full list of integrity checks and common failure modes.
|
|
191
|
+
|
|
192
|
+
## Key Research Findings
|
|
193
|
+
|
|
194
|
+
Our benchmarking experiments across 3 models, 3 datasets, and 200+ runs reveal:
|
|
195
|
+
|
|
196
|
+
| Finding | Impact |
|
|
197
|
+
|---------|--------|
|
|
198
|
+
| **3-model ensemble hits SPS 0.90** | Equal-weight average of Haiku + Gemini + GPT-4o-mini beats any single model by +5-7 pts |
|
|
199
|
+
| **Temperature is model-specific** | Gemini benefits from high temp (+4.5%), Haiku is insensitive, GPT-4o-mini mild |
|
|
200
|
+
| **Demographic conditioning quantifies LLM bias** | Republican conditioning 2.4x stronger than Democrat — model defaults approximate liberal responses |
|
|
201
|
+
| **Persona template matters** | Default template beats stripped/broken templates by +11 SPS pts |
|
|
202
|
+
|
|
203
|
+
See [FINDINGS.md](FINDINGS.md) for the full experimental report with methodology, replications, and per-metric breakdowns.
|
|
204
|
+
|
|
205
|
+
## Status
|
|
206
|
+
|
|
207
|
+
Phase 2 complete: Multi-model benchmarking, ensemble blending, temperature sweeps, and demographic conditioning analysis across OpinionsQA, SubPOP, and GlobalOpinionQA.
|
|
208
|
+
|
|
209
|
+
## Ground Truth
|
|
210
|
+
|
|
211
|
+
Built on nine registered survey datasets. Each adapter declares a
|
|
212
|
+
redistribution policy; `full` ships `human_distribution` publicly, `gated`
|
|
213
|
+
routes per-question artifacts to a JWT-authenticated Cloudflare R2 origin,
|
|
214
|
+
and `aggregates_only` / `citation_only` contribute to leaderboard aggregates
|
|
215
|
+
only. Canonical source of truth is the `redistribution_policy` attribute on
|
|
216
|
+
each adapter in `src/synthbench/datasets/` (see
|
|
217
|
+
[`src/synthbench/datasets/policy.py`](src/synthbench/datasets/policy.py)).
|
|
218
|
+
|
|
219
|
+
| Dataset | Tier | Source |
|
|
220
|
+
|---------|------|--------|
|
|
221
|
+
| [OpinionsQA](https://github.com/tatsu-lab/opinions_qa) (Santurkar et al., ICML 2023) | gated | Pew American Trends Panel, 1,498 questions |
|
|
222
|
+
| [GlobalOpinionQA](https://arxiv.org/abs/2306.16388) (Durmus et al., 2024) | gated | Pew Global Attitudes, 138 countries |
|
|
223
|
+
| GSS (General Social Survey) | full | NORC, microdata-capable |
|
|
224
|
+
| NTIA Internet Use Supplement | full | US Census / NTIA |
|
|
225
|
+
| SubPOP | gated | 22 US subpopulations, 3,362 questions |
|
|
226
|
+
| WVS (World Values Survey) | gated | WVSA, cross-national |
|
|
227
|
+
| Eurobarometer | gated | European Commission |
|
|
228
|
+
| Michigan (Surveys of Consumers) | gated | U. of Michigan |
|
|
229
|
+
| Pew Technology | gated | Pew Research |
|
|
230
|
+
|
|
231
|
+
GSS and NTIA ship with full per-question distributions; the remaining seven
|
|
232
|
+
require a signed-in account to reach per-question payloads.
|
|
233
|
+
|
|
234
|
+
## Cost tracking
|
|
235
|
+
|
|
236
|
+
The leaderboard JSON carries per-row cost fields and a top-level
|
|
237
|
+
`pricing_snapshot` object:
|
|
238
|
+
|
|
239
|
+
- Each row exposes `cost_usd`, `cost_per_100q`, `cost_per_sps_point`, and
|
|
240
|
+
`is_cost_estimated`. Ensemble rows sum `cost_usd` across constituent
|
|
241
|
+
runs listed in `config.ensemble_sources`.
|
|
242
|
+
- `pricing_snapshot` records the per-model `input_per_1m` / `output_per_1m`
|
|
243
|
+
rates used for this publish run, the `snapshot_date` anchor comment, and
|
|
244
|
+
the installed `synth_panel_version` that produced the rates.
|
|
245
|
+
|
|
246
|
+
This lets downstream consumers audit which pricing table produced which
|
|
247
|
+
`cost_usd` and reconcile against provider-reported billing without guessing
|
|
248
|
+
at rate drift. See [sb-x8t] and `src/synthbench/publish.py::_build_pricing_snapshot`.
|
|
249
|
+
|
|
250
|
+
## Convergence analysis
|
|
251
|
+
|
|
252
|
+
`synthbench convergence bootstrap` computes theoretical ~1/√n convergence
|
|
253
|
+
curves for every question in a dataset by multinomial resampling from the
|
|
254
|
+
aggregate `human_distribution`. `synthbench convergence real` runs the same
|
|
255
|
+
curve shape over individual-level microdata (GSS today; WVS / Eurobarometer
|
|
256
|
+
microdata adapters are a follow-on). `synthbench convergence compare` emits
|
|
257
|
+
both curves side-by-side.
|
|
258
|
+
|
|
259
|
+
See [`docs/convergence-analysis.md`](docs/convergence-analysis.md) for the
|
|
260
|
+
JSON schema, CLI flags, and the `load_convergence_baseline` integration
|
|
261
|
+
surface that synthpanel's `--calibrate-against DATASET:QUESTION` flag
|
|
262
|
+
consumes.
|
|
263
|
+
|
|
264
|
+
## Citation
|
|
265
|
+
|
|
266
|
+
If you use SynthBench in your research, please cite:
|
|
267
|
+
|
|
268
|
+
```bibtex
|
|
269
|
+
@misc{synthbench2026,
|
|
270
|
+
title={SynthBench: Open Benchmark for Synthetic Survey Respondent Quality},
|
|
271
|
+
author={DataViking-Tech},
|
|
272
|
+
year={2026},
|
|
273
|
+
url={https://github.com/DataViking-Tech/synthbench}
|
|
274
|
+
}
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
## License
|
|
278
|
+
|
|
279
|
+
MIT
|
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
# SynthBench
|
|
2
|
+
|
|
3
|
+
[](https://github.com/DataViking-Tech/synthbench/actions/workflows/ci.yml)
|
|
4
|
+
[](LICENSE)
|
|
5
|
+
[](pyproject.toml)
|
|
6
|
+
[](https://synthbench.org/)
|
|
7
|
+
|
|
8
|
+
Open benchmark harness for synthetic survey respondent quality.
|
|
9
|
+
|
|
10
|
+
**The MLPerf of synthetic UXR.**
|
|
11
|
+
|
|
12
|
+
SynthBench measures how well synthetic respondent systems (like [synthpanel](https://github.com/DataViking-Tech/SynthPanel), Ditto, Synthetic Users, or raw ChatGPT prompting) reproduce real human survey response patterns against real Pew American Trends Panel and GlobalOpinionQA ground truth — so "it sounds plausible" gets replaced with a measurable similarity score.
|
|
13
|
+
|
|
14
|
+
## Quick Start
|
|
15
|
+
|
|
16
|
+
Run your first benchmark in 3 commands:
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
pip install synthbench-eval
|
|
20
|
+
synthbench run --provider random --suite smoke --output results/
|
|
21
|
+
synthbench leaderboard --results-dir results/
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
> **Note:** the distribution is named `synthbench-eval` — the bare `synthbench`
|
|
25
|
+
> name on PyPI belongs to an unrelated project. The import package and CLI are
|
|
26
|
+
> still `synthbench`. For development, clone this repo and `pip install -e .`.
|
|
27
|
+
> API-backed providers need an extra, e.g. `pip install "synthbench-eval[openai]"`
|
|
28
|
+
> for the `openrouter` / `raw-openai` / `raw-gemini` / `ollama` providers.
|
|
29
|
+
|
|
30
|
+
Try with a real model (requires API key):
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
export OPENROUTER_API_KEY=your-key
|
|
34
|
+
synthbench run --provider openrouter --model openai/gpt-4o-mini --suite core --samples 50
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
See [`notebooks/getting_started.ipynb`](notebooks/getting_started.ipynb) for a guided walkthrough.
|
|
38
|
+
|
|
39
|
+
## Leaderboard
|
|
40
|
+
|
|
41
|
+
**[View the live leaderboard](https://synthbench.org/)** — see also the
|
|
42
|
+
[methodology](https://synthbench.org/methodology/) and
|
|
43
|
+
[findings](https://synthbench.org/findings/) pages.
|
|
44
|
+
|
|
45
|
+
Regenerate leaderboard data for the Astro site:
|
|
46
|
+
```bash
|
|
47
|
+
synthbench publish-data --results-dir ./leaderboard-results --output site/src/data/leaderboard.json
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
### Contributor note: gated data publishing
|
|
51
|
+
|
|
52
|
+
Most contributors do **not** need to publish gated artifacts. If you're running
|
|
53
|
+
benchmarks locally or contributing via PR, focus on `synthbench run`,
|
|
54
|
+
`synthbench validate`, and result submission.
|
|
55
|
+
|
|
56
|
+
The gated data publication path is maintainer infrastructure and is handled by
|
|
57
|
+
project deployment workflows.
|
|
58
|
+
|
|
59
|
+
## Development
|
|
60
|
+
|
|
61
|
+
After cloning, enable the repo-tracked git hooks so pushes that would fail CI's
|
|
62
|
+
`ruff format --check` are caught locally:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
./scripts/install-hooks.sh # one-time: wires .githooks/ via core.hooksPath
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
The `pre-push` hook only checks Python files changed in the commits being
|
|
69
|
+
pushed, so already-formatted branches add no meaningful overhead. Run
|
|
70
|
+
`./scripts/format-check.sh` anytime to mirror the full CI lint job. Emergency
|
|
71
|
+
bypass: `git push --no-verify`.
|
|
72
|
+
|
|
73
|
+
For contribution workflow and PR expectations, see
|
|
74
|
+
[`CONTRIBUTING.md`](CONTRIBUTING.md).
|
|
75
|
+
|
|
76
|
+
## Submit Results
|
|
77
|
+
|
|
78
|
+
Three ways to land a run on the leaderboard, in order of friction:
|
|
79
|
+
|
|
80
|
+
### 1. CLI (recommended for repeat submissions)
|
|
81
|
+
|
|
82
|
+
Mint an API key at [synthbench.org/account](https://synthbench.org/account/),
|
|
83
|
+
then:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
export SYNTHBENCH_API_KEY=sb_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx
|
|
87
|
+
synthbench run --provider openrouter --model gpt-4o-mini --suite full -o results/
|
|
88
|
+
synthbench submit results/openrouter_gpt-4o-mini_opinionsqa.json
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
#### End-to-end: run + submit in one command (`--submit`)
|
|
92
|
+
|
|
93
|
+
Collapse the two steps above into a single invocation. The CLI saves the
|
|
94
|
+
result JSON locally (so a validation rejection doesn't lose your run) and
|
|
95
|
+
then POSTs it to the leaderboard. With `--wait`, the process blocks until
|
|
96
|
+
the validator reaches a terminal state and the exit code mirrors the
|
|
97
|
+
outcome — suitable for dropping into a CI pipeline that gates on
|
|
98
|
+
leaderboard publication:
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
export SYNTHBENCH_API_KEY=sb_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx
|
|
102
|
+
synthbench run \
|
|
103
|
+
--provider raw-anthropic --model claude-haiku-4-5 \
|
|
104
|
+
--dataset globalopinionqa --samples 30 -n 100 \
|
|
105
|
+
--submit --wait \
|
|
106
|
+
--submit-message "first pass with new prompt template"
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Exit codes with `--wait`:
|
|
110
|
+
|
|
111
|
+
| Code | Meaning |
|
|
112
|
+
|-----:|---------|
|
|
113
|
+
| 0 | Published — result is live on the leaderboard |
|
|
114
|
+
| 1 | Rejected by the validator OR hard error (bad key, 5xx, etc.) |
|
|
115
|
+
| 2 | Poll deadline exceeded (validation is still running server-side; check [/account/submissions/](https://synthbench.org/account/submissions/)) |
|
|
116
|
+
|
|
117
|
+
Without `--wait`, the upload exits 0 as soon as the Worker accepts the
|
|
118
|
+
submission (status = `validating`) — useful for fire-and-forget runs
|
|
119
|
+
where you'll check the web dashboard later.
|
|
120
|
+
|
|
121
|
+
`--submit-message` is optional; it's stored alongside the uploaded JSON
|
|
122
|
+
so you can label experiments (e.g. "v2 prompt", "temp=1.0 sweep") without
|
|
123
|
+
touching your `config` fields and perturbing the `config_id` hash.
|
|
124
|
+
|
|
125
|
+
The Worker validates the submission, stages it to R2, and dispatches the
|
|
126
|
+
GitHub Actions pipeline. Successful runs publish within ~5 minutes. Keys
|
|
127
|
+
are rate-limited to 60 submissions/hour. See
|
|
128
|
+
[SUBMISSIONS.md → API key flow](SUBMISSIONS.md#api-key-flow-cli-submission).
|
|
129
|
+
|
|
130
|
+
### 2. Web upload
|
|
131
|
+
|
|
132
|
+
Sign in at [/account](https://synthbench.org/account/) and drag your result
|
|
133
|
+
JSON into [/submit/upload](https://synthbench.org/submit/upload/). Same
|
|
134
|
+
validation pipeline, no key required.
|
|
135
|
+
|
|
136
|
+
### 3. GitHub PR (power-user path)
|
|
137
|
+
|
|
138
|
+
1. **Fork** this repo.
|
|
139
|
+
2. **Run** SynthBench with your provider:
|
|
140
|
+
```bash
|
|
141
|
+
synthbench run --provider <your-provider> --model <your-model> --suite full --output results/
|
|
142
|
+
```
|
|
143
|
+
3. **Validate locally** before opening a PR:
|
|
144
|
+
```bash
|
|
145
|
+
synthbench validate results/<your-result>.json
|
|
146
|
+
```
|
|
147
|
+
4. **Copy** the result JSON into `leaderboard-results/`.
|
|
148
|
+
5. **Open a PR** against this repo.
|
|
149
|
+
6. **CI validates** schema, bounds, distributions, and recomputes every metric against the per-question data. Fabricated or inconsistent submissions are rejected.
|
|
150
|
+
7. **Maintainers review and merge** — your results appear on the leaderboard.
|
|
151
|
+
|
|
152
|
+
See [SUBMISSIONS.md](SUBMISSIONS.md) for the full list of integrity checks and common failure modes.
|
|
153
|
+
|
|
154
|
+
## Key Research Findings
|
|
155
|
+
|
|
156
|
+
Our benchmarking experiments across 3 models, 3 datasets, and 200+ runs reveal:
|
|
157
|
+
|
|
158
|
+
| Finding | Impact |
|
|
159
|
+
|---------|--------|
|
|
160
|
+
| **3-model ensemble hits SPS 0.90** | Equal-weight average of Haiku + Gemini + GPT-4o-mini beats any single model by +5-7 pts |
|
|
161
|
+
| **Temperature is model-specific** | Gemini benefits from high temp (+4.5%), Haiku is insensitive, GPT-4o-mini mild |
|
|
162
|
+
| **Demographic conditioning quantifies LLM bias** | Republican conditioning 2.4x stronger than Democrat — model defaults approximate liberal responses |
|
|
163
|
+
| **Persona template matters** | Default template beats stripped/broken templates by +11 SPS pts |
|
|
164
|
+
|
|
165
|
+
See [FINDINGS.md](FINDINGS.md) for the full experimental report with methodology, replications, and per-metric breakdowns.
|
|
166
|
+
|
|
167
|
+
## Status
|
|
168
|
+
|
|
169
|
+
Phase 2 complete: Multi-model benchmarking, ensemble blending, temperature sweeps, and demographic conditioning analysis across OpinionsQA, SubPOP, and GlobalOpinionQA.
|
|
170
|
+
|
|
171
|
+
## Ground Truth
|
|
172
|
+
|
|
173
|
+
Built on nine registered survey datasets. Each adapter declares a
|
|
174
|
+
redistribution policy; `full` ships `human_distribution` publicly, `gated`
|
|
175
|
+
routes per-question artifacts to a JWT-authenticated Cloudflare R2 origin,
|
|
176
|
+
and `aggregates_only` / `citation_only` contribute to leaderboard aggregates
|
|
177
|
+
only. Canonical source of truth is the `redistribution_policy` attribute on
|
|
178
|
+
each adapter in `src/synthbench/datasets/` (see
|
|
179
|
+
[`src/synthbench/datasets/policy.py`](src/synthbench/datasets/policy.py)).
|
|
180
|
+
|
|
181
|
+
| Dataset | Tier | Source |
|
|
182
|
+
|---------|------|--------|
|
|
183
|
+
| [OpinionsQA](https://github.com/tatsu-lab/opinions_qa) (Santurkar et al., ICML 2023) | gated | Pew American Trends Panel, 1,498 questions |
|
|
184
|
+
| [GlobalOpinionQA](https://arxiv.org/abs/2306.16388) (Durmus et al., 2024) | gated | Pew Global Attitudes, 138 countries |
|
|
185
|
+
| GSS (General Social Survey) | full | NORC, microdata-capable |
|
|
186
|
+
| NTIA Internet Use Supplement | full | US Census / NTIA |
|
|
187
|
+
| SubPOP | gated | 22 US subpopulations, 3,362 questions |
|
|
188
|
+
| WVS (World Values Survey) | gated | WVSA, cross-national |
|
|
189
|
+
| Eurobarometer | gated | European Commission |
|
|
190
|
+
| Michigan (Surveys of Consumers) | gated | U. of Michigan |
|
|
191
|
+
| Pew Technology | gated | Pew Research |
|
|
192
|
+
|
|
193
|
+
GSS and NTIA ship with full per-question distributions; the remaining seven
|
|
194
|
+
require a signed-in account to reach per-question payloads.
|
|
195
|
+
|
|
196
|
+
## Cost tracking
|
|
197
|
+
|
|
198
|
+
The leaderboard JSON carries per-row cost fields and a top-level
|
|
199
|
+
`pricing_snapshot` object:
|
|
200
|
+
|
|
201
|
+
- Each row exposes `cost_usd`, `cost_per_100q`, `cost_per_sps_point`, and
|
|
202
|
+
`is_cost_estimated`. Ensemble rows sum `cost_usd` across constituent
|
|
203
|
+
runs listed in `config.ensemble_sources`.
|
|
204
|
+
- `pricing_snapshot` records the per-model `input_per_1m` / `output_per_1m`
|
|
205
|
+
rates used for this publish run, the `snapshot_date` anchor comment, and
|
|
206
|
+
the installed `synth_panel_version` that produced the rates.
|
|
207
|
+
|
|
208
|
+
This lets downstream consumers audit which pricing table produced which
|
|
209
|
+
`cost_usd` and reconcile against provider-reported billing without guessing
|
|
210
|
+
at rate drift. See [sb-x8t] and `src/synthbench/publish.py::_build_pricing_snapshot`.
|
|
211
|
+
|
|
212
|
+
## Convergence analysis
|
|
213
|
+
|
|
214
|
+
`synthbench convergence bootstrap` computes theoretical ~1/√n convergence
|
|
215
|
+
curves for every question in a dataset by multinomial resampling from the
|
|
216
|
+
aggregate `human_distribution`. `synthbench convergence real` runs the same
|
|
217
|
+
curve shape over individual-level microdata (GSS today; WVS / Eurobarometer
|
|
218
|
+
microdata adapters are a follow-on). `synthbench convergence compare` emits
|
|
219
|
+
both curves side-by-side.
|
|
220
|
+
|
|
221
|
+
See [`docs/convergence-analysis.md`](docs/convergence-analysis.md) for the
|
|
222
|
+
JSON schema, CLI flags, and the `load_convergence_baseline` integration
|
|
223
|
+
surface that synthpanel's `--calibrate-against DATASET:QUESTION` flag
|
|
224
|
+
consumes.
|
|
225
|
+
|
|
226
|
+
## Citation
|
|
227
|
+
|
|
228
|
+
If you use SynthBench in your research, please cite:
|
|
229
|
+
|
|
230
|
+
```bibtex
|
|
231
|
+
@misc{synthbench2026,
|
|
232
|
+
title={SynthBench: Open Benchmark for Synthetic Survey Respondent Quality},
|
|
233
|
+
author={DataViking-Tech},
|
|
234
|
+
year={2026},
|
|
235
|
+
url={https://github.com/DataViking-Tech/synthbench}
|
|
236
|
+
}
|
|
237
|
+
```
|
|
238
|
+
|
|
239
|
+
## License
|
|
240
|
+
|
|
241
|
+
MIT
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
# Distribution name is synthbench-eval: the bare "synthbench" name on PyPI
|
|
7
|
+
# belongs to an unrelated project. Import package and CLI remain `synthbench`.
|
|
8
|
+
name = "synthbench-eval"
|
|
9
|
+
# Placeholder — the publish job stamps the real version from the release tag
|
|
10
|
+
# at build time (see publish-pypi in .github/workflows/auto-tag.yml).
|
|
11
|
+
version = "0.4.0"
|
|
12
|
+
description = "Open benchmark harness for synthetic survey respondent quality"
|
|
13
|
+
readme = "README.md"
|
|
14
|
+
license = "MIT"
|
|
15
|
+
requires-python = ">=3.9"
|
|
16
|
+
keywords = [
|
|
17
|
+
"benchmark",
|
|
18
|
+
"survey",
|
|
19
|
+
"synthetic-data",
|
|
20
|
+
"llm",
|
|
21
|
+
"evaluation",
|
|
22
|
+
"opinionsqa",
|
|
23
|
+
"globalopinionqa",
|
|
24
|
+
"gss",
|
|
25
|
+
"convergence",
|
|
26
|
+
"synthpanel",
|
|
27
|
+
]
|
|
28
|
+
dependencies = [
|
|
29
|
+
"click>=8.0",
|
|
30
|
+
"httpx>=0.24",
|
|
31
|
+
"numpy>=1.24",
|
|
32
|
+
"pyyaml>=6.0",
|
|
33
|
+
"scipy>=1.10",
|
|
34
|
+
"synthpanel>=0.2.0",
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
[project.optional-dependencies]
|
|
38
|
+
anthropic = ["anthropic>=0.18"]
|
|
39
|
+
openai = ["openai>=1.0"]
|
|
40
|
+
viz = ["matplotlib>=3.7"]
|
|
41
|
+
hf = ["datasets>=2.14"]
|
|
42
|
+
subpop = ["datasets>=2.14"]
|
|
43
|
+
# R2 publish target for gated-tier per-question/run/config artifacts (sb-sjs).
|
|
44
|
+
# Only required in environments that actually upload (CI publish step); local
|
|
45
|
+
# dev defaults to writing to site/public/data without R2.
|
|
46
|
+
r2 = ["boto3>=1.34"]
|
|
47
|
+
all = [
|
|
48
|
+
"anthropic>=0.18",
|
|
49
|
+
"openai>=1.0",
|
|
50
|
+
"matplotlib>=3.7",
|
|
51
|
+
"datasets>=2.14",
|
|
52
|
+
"boto3>=1.34",
|
|
53
|
+
]
|
|
54
|
+
dev = ["pytest>=7.0", "pytest-asyncio>=0.21"]
|
|
55
|
+
|
|
56
|
+
[project.scripts]
|
|
57
|
+
synthbench = "synthbench.cli:main"
|
|
58
|
+
|
|
59
|
+
[tool.setuptools.packages.find]
|
|
60
|
+
where = ["src"]
|
|
61
|
+
|
|
62
|
+
# Holdout answer-key data (issue #259) is intentionally NOT included in
|
|
63
|
+
# the wheel. Raw upstream survey items are downloaded at runtime by the
|
|
64
|
+
# dataset adapters into the user's data directory, never from the package
|
|
65
|
+
# install. The public / private-holdout partition is a logical filter over
|
|
66
|
+
# question keys (see ``synthbench.private_holdout``), and the canonical
|
|
67
|
+
# private *answer key* (per-question human distributions) lives only in
|
|
68
|
+
# the gated R2 origin + credentialed maintainer caches. We therefore
|
|
69
|
+
# deliberately leave ``[tool.setuptools.package-data]`` unset for
|
|
70
|
+
# synthbench — the only files shipped are the Python sources resolved by
|
|
71
|
+
# ``packages.find``. Adding raw data globs here would break the holdout
|
|
72
|
+
# integrity model. See ``docs/held-out.md``.
|
|
73
|
+
|
|
74
|
+
[tool.pytest.ini_options]
|
|
75
|
+
asyncio_mode = "auto"
|
|
76
|
+
testpaths = ["tests"]
|
|
77
|
+
# Prepend local src/ to sys.path so the in-tree synthbench package shadows any
|
|
78
|
+
# stale `pip install -e .` entries from sibling worktrees (sb-3bn).
|
|
79
|
+
pythonpath = ["src"]
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""SynthBench — open benchmark harness for synthetic survey respondent quality."""
|
|
2
|
+
|
|
3
|
+
from synthbench.convergence.baseline import (
|
|
4
|
+
BaselineGatedError,
|
|
5
|
+
BaselineUnavailable,
|
|
6
|
+
load_convergence_baseline,
|
|
7
|
+
)
|
|
8
|
+
|
|
9
|
+
__version__ = "0.4.0"
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"BaselineGatedError",
|
|
13
|
+
"BaselineUnavailable",
|
|
14
|
+
"load_convergence_baseline",
|
|
15
|
+
"__version__",
|
|
16
|
+
]
|