synthbench-eval 0.6.0__tar.gz → 0.6.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/PKG-INFO +1 -1
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/pyproject.toml +1 -1
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/__init__.py +1 -1
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/cli.py +22 -5
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/report.py +18 -4
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/runner.py +35 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/suites/__init__.py +4 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/validation.py +11 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/PKG-INFO +1 -1
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_report.py +23 -1
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_runner.py +38 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_validation.py +10 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/LICENSE +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/README.md +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/setup.cfg +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/__main__.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/adapter.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/anomaly.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/baseline_floors.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/baselines.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/config_id.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/contamination.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/__init__.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/baseline.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/bootstrap.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/cli_report.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/curves.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/real_sampling.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/thresholds.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/__init__.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/base.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/eurobarometer.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/globalopinionqa.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/gss.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/michigan.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/ntia.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/opinionsqa.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/pewtech.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/policy.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/subpop.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/wvs.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/findings.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/holdout.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/human_distributions.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/leaderboard.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/leaderboard_pr.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/__init__.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/composite.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/conditioning.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/distributional.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/ranking.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/refusal.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/subgroup.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/private_holdout.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/__init__.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/_parsing.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/_retry.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/althing.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/base.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/http.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/majority_baseline.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/ollama.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/openrouter.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/population_baseline.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/random_baseline.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/raw_anthropic.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/raw_gemini.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/raw_openai.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/publish.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/r2_upload.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/recompute.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/run_hash.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/run_validity.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/stats.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/submission.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/submission_pr.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/submit_adapter.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/suite.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/topics.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/user_config.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/visualize.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/SOURCES.txt +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/dependency_links.txt +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/entry_points.txt +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/requires.txt +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/top_level.txt +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_anomaly.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_baseline_floors.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_baselines.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_cli_run_submit.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_cli_submit.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_cli_submit_adapter.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_config_id.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_contamination.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_convergence_baseline.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_convergence_bootstrap.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_eurobarometer.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_globalopinionqa.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_gss.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_michigan.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_ntia.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_policy.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_wvs.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_effort.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_findings_drift.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_findings_elicitation.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_findings_nonresponse.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_holdout.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_human_distributions.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_integration_tokens.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_metrics.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_microdata.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_private_holdout.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_provider_parsing.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_providers.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_config_id_consistency.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_cost.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_cross_provider_jsd.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_demographic_scorecard.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_gated_fail_closed.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_holdout.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_invalid_runs.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_normalized.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_policy.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_questions.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_r2_routing.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_rehydration.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_run_counts.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_runnable_ids.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_text_rehydration.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_topic_metrics.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_r2_upload.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_recompute_integrity.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_run_hash.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_run_validity.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_site_dataset_cards.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_stats.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_stats_golden.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_strip_gated_guard.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_submission.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_submission_pr.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_suite.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_suites.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_suites_novel_products.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_topics.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_user_config.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_validation_holdout.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_validation_stripped.py +0 -0
- {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_verify_publish_integrity.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: synthbench-eval
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.2
|
|
4
4
|
Summary: Open benchmark harness for synthetic survey respondent quality
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
Keywords: benchmark,survey,synthetic-data,llm,evaluation,opinionsqa,globalopinionqa,gss,convergence,althing
|
|
@@ -8,7 +8,7 @@ build-backend = "setuptools.build_meta"
|
|
|
8
8
|
name = "synthbench-eval"
|
|
9
9
|
# Placeholder — the publish job stamps the real version from the release tag
|
|
10
10
|
# at build time (see publish-pypi in .github/workflows/auto-tag.yml).
|
|
11
|
-
version = "0.6.
|
|
11
|
+
version = "0.6.2"
|
|
12
12
|
description = "Open benchmark harness for synthetic survey respondent quality"
|
|
13
13
|
readme = "README.md"
|
|
14
14
|
license = "MIT"
|
|
@@ -390,7 +390,7 @@ def run(
|
|
|
390
390
|
)
|
|
391
391
|
sys.exit(2)
|
|
392
392
|
|
|
393
|
-
|
|
393
|
+
_run_benchmark_or_exit(
|
|
394
394
|
_run_async(
|
|
395
395
|
provider,
|
|
396
396
|
model,
|
|
@@ -423,6 +423,17 @@ def run(
|
|
|
423
423
|
)
|
|
424
424
|
|
|
425
425
|
|
|
426
|
+
def _run_benchmark_or_exit(coro) -> None:
|
|
427
|
+
"""Run a benchmark coroutine, turning an empty question set into a clean CLI error."""
|
|
428
|
+
from synthbench.runner import EmptyQuestionSetError
|
|
429
|
+
|
|
430
|
+
try:
|
|
431
|
+
asyncio.run(coro)
|
|
432
|
+
except EmptyQuestionSetError as exc:
|
|
433
|
+
click.echo(f"Error: {exc}", err=True)
|
|
434
|
+
sys.exit(2)
|
|
435
|
+
|
|
436
|
+
|
|
426
437
|
async def _run_async(
|
|
427
438
|
provider_name,
|
|
428
439
|
model,
|
|
@@ -1147,7 +1158,7 @@ def replicate(
|
|
|
1147
1158
|
Example:
|
|
1148
1159
|
synthbench replicate --provider raw-anthropic --n-runs 5 --suite core
|
|
1149
1160
|
"""
|
|
1150
|
-
|
|
1161
|
+
_run_benchmark_or_exit(
|
|
1151
1162
|
_replicate_async(
|
|
1152
1163
|
provider,
|
|
1153
1164
|
model,
|
|
@@ -1284,8 +1295,10 @@ async def _replicate_async(
|
|
|
1284
1295
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
1285
1296
|
from datetime import datetime
|
|
1286
1297
|
|
|
1298
|
+
from synthbench.report import provider_slug as report_provider_slug
|
|
1299
|
+
|
|
1287
1300
|
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
|
|
1288
|
-
provider_slug = prov.name
|
|
1301
|
+
provider_slug = report_provider_slug(prov.name)
|
|
1289
1302
|
|
|
1290
1303
|
json_data = {
|
|
1291
1304
|
"benchmark": "synthbench",
|
|
@@ -2018,8 +2031,10 @@ async def _contamination_async(
|
|
|
2018
2031
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
2019
2032
|
from datetime import datetime
|
|
2020
2033
|
|
|
2034
|
+
from synthbench.report import provider_slug as report_provider_slug
|
|
2035
|
+
|
|
2021
2036
|
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
|
|
2022
|
-
provider_slug = prov.name
|
|
2037
|
+
provider_slug = report_provider_slug(prov.name)
|
|
2023
2038
|
json_path = out_dir / f"contamination_{provider_slug}_{ts}.json"
|
|
2024
2039
|
json_path.write_text(json.dumps(result_data, indent=2))
|
|
2025
2040
|
click.echo(f"Results saved: {json_path}")
|
|
@@ -2152,8 +2167,10 @@ async def _contamination_deident_async(
|
|
|
2152
2167
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
2153
2168
|
from datetime import datetime
|
|
2154
2169
|
|
|
2170
|
+
from synthbench.report import provider_slug as report_provider_slug
|
|
2171
|
+
|
|
2155
2172
|
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
|
|
2156
|
-
provider_slug = prov.name
|
|
2173
|
+
provider_slug = report_provider_slug(prov.name)
|
|
2157
2174
|
json_path = out_dir / f"contamination_deident_{provider_slug}_{ts}.json"
|
|
2158
2175
|
json_path.write_text(json.dumps(result_data, indent=2))
|
|
2159
2176
|
click.echo(f"Results saved: {json_path}")
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import json
|
|
6
|
+
import re
|
|
6
7
|
from datetime import datetime, timezone
|
|
7
8
|
from pathlib import Path
|
|
8
9
|
|
|
@@ -469,6 +470,18 @@ def to_markdown(
|
|
|
469
470
|
return "\n".join(lines)
|
|
470
471
|
|
|
471
472
|
|
|
473
|
+
# Characters that are path separators or invalid in Windows filenames. A ':'
|
|
474
|
+
# is the dangerous one: NTFS treats ``name:rest`` as an alternate data
|
|
475
|
+
# stream, so a provider like ``althing/claude-code:haiku`` would silently
|
|
476
|
+
# write its score card into a hidden stream of a truncated file.
|
|
477
|
+
_UNSAFE_FILENAME_CHARS = re.compile(r'[\\/:*?"<>|]')
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
def provider_slug(provider_name: str) -> str:
|
|
481
|
+
"""Return *provider_name* made safe for use in a result filename."""
|
|
482
|
+
return _UNSAFE_FILENAME_CHARS.sub("_", provider_name)
|
|
483
|
+
|
|
484
|
+
|
|
472
485
|
def save(result: BenchmarkResult, output_dir: Path | str) -> tuple[Path, Path]:
|
|
473
486
|
"""Save JSON and markdown score cards to output_dir.
|
|
474
487
|
|
|
@@ -478,16 +491,17 @@ def save(result: BenchmarkResult, output_dir: Path | str) -> tuple[Path, Path]:
|
|
|
478
491
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
479
492
|
|
|
480
493
|
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
|
|
481
|
-
|
|
482
|
-
base = f"{result.dataset_name}_{provider_slug}_{ts}"
|
|
494
|
+
base = f"{result.dataset_name}_{provider_slug(result.provider_name)}_{ts}"
|
|
483
495
|
|
|
484
496
|
json_path = output_dir / f"{base}.json"
|
|
485
497
|
md_path = output_dir / f"{base}.md"
|
|
486
498
|
|
|
487
|
-
|
|
499
|
+
# Explicit UTF-8: the markdown card contains box-drawing bars (█░) that
|
|
500
|
+
# the Windows default codepage (cp1252) cannot encode.
|
|
501
|
+
with open(json_path, "w", encoding="utf-8") as f:
|
|
488
502
|
json.dump(to_json(result), f, indent=2)
|
|
489
503
|
|
|
490
|
-
with open(md_path, "w") as f:
|
|
504
|
+
with open(md_path, "w", encoding="utf-8") as f:
|
|
491
505
|
f.write(to_markdown(result))
|
|
492
506
|
|
|
493
507
|
return json_path, md_path
|
|
@@ -26,6 +26,33 @@ from synthbench.providers.base import Distribution, PersonaSpec, Provider, Respo
|
|
|
26
26
|
from synthbench.stats import bootstrap_ci, question_set_hash
|
|
27
27
|
|
|
28
28
|
|
|
29
|
+
class EmptyQuestionSetError(ValueError):
|
|
30
|
+
"""Raised when a run would evaluate zero questions.
|
|
31
|
+
|
|
32
|
+
Scoring an empty set yields empty-input defaults that look like a real
|
|
33
|
+
SPS (synthbench#353), so the runner refuses instead.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _empty_question_set_message(
|
|
38
|
+
dataset_name: str, *, loaded: int, filtered: bool
|
|
39
|
+
) -> str:
|
|
40
|
+
if not filtered:
|
|
41
|
+
return f"Dataset '{dataset_name}' loaded 0 questions; nothing to evaluate."
|
|
42
|
+
from synthbench.suites import SUITE_SOURCE_DATASET
|
|
43
|
+
|
|
44
|
+
msg = (
|
|
45
|
+
f"The --suite/--topic filter matched 0 of {loaded} questions in dataset "
|
|
46
|
+
f"'{dataset_name}'; nothing to evaluate."
|
|
47
|
+
)
|
|
48
|
+
if dataset_name != SUITE_SOURCE_DATASET:
|
|
49
|
+
msg += (
|
|
50
|
+
f" Pinned suites and topics are built from '{SUITE_SOURCE_DATASET}' "
|
|
51
|
+
"question keys — use --n to size a run on this dataset."
|
|
52
|
+
)
|
|
53
|
+
return msg
|
|
54
|
+
|
|
55
|
+
|
|
29
56
|
def _sha256_of(s: str) -> str:
|
|
30
57
|
"""Return ``sha256:<hex>`` digest of a UTF-8 string."""
|
|
31
58
|
return "sha256:" + hashlib.sha256(s.encode("utf-8")).hexdigest()
|
|
@@ -405,6 +432,7 @@ class BenchmarkRunner:
|
|
|
405
432
|
) -> BenchmarkResult:
|
|
406
433
|
t0 = time.monotonic()
|
|
407
434
|
questions = self.dataset.load(n=n)
|
|
435
|
+
loaded = len(questions)
|
|
408
436
|
|
|
409
437
|
# Filter by pinned question set if provided
|
|
410
438
|
if question_keys is not None:
|
|
@@ -412,6 +440,13 @@ class BenchmarkRunner:
|
|
|
412
440
|
|
|
413
441
|
questions = filter_questions_by_suite(questions, question_keys)
|
|
414
442
|
|
|
443
|
+
if not questions:
|
|
444
|
+
raise EmptyQuestionSetError(
|
|
445
|
+
_empty_question_set_message(
|
|
446
|
+
self.dataset.name, loaded=loaded, filtered=question_keys is not None
|
|
447
|
+
)
|
|
448
|
+
)
|
|
449
|
+
|
|
415
450
|
# Use batched evaluation when provider supports it
|
|
416
451
|
use_batch = self.provider.supports_distribution and hasattr(
|
|
417
452
|
self.provider, "batch_get_distribution"
|
|
@@ -25,6 +25,10 @@ AVAILABLE_SUITES = ("smoke", "core", "full")
|
|
|
25
25
|
|
|
26
26
|
AVAILABLE_TOPICS = ("political", "consumer", "neutral")
|
|
27
27
|
|
|
28
|
+
# Every pinned suite and topic set above is built from OpinionsQA question
|
|
29
|
+
# keys (``<VAR>_W<wave>``), so against any other dataset they match nothing.
|
|
30
|
+
SUITE_SOURCE_DATASET = "opinionsqa"
|
|
31
|
+
|
|
28
32
|
|
|
29
33
|
def load_suite(name: str) -> list[str] | None:
|
|
30
34
|
"""Load a pinned question set by name.
|
|
@@ -649,6 +649,17 @@ def _validate_counts(data: Mapping[str, Any]) -> list[Issue]:
|
|
|
649
649
|
per_question = data.get("per_question") or []
|
|
650
650
|
reported = aggregate.get("n_questions")
|
|
651
651
|
actual = len(per_question) if isinstance(per_question, list) else 0
|
|
652
|
+
if isinstance(per_question, list) and actual == 0:
|
|
653
|
+
# A zero-question run has no evidence behind it; its scores are
|
|
654
|
+
# empty-input defaults (synthbench#353), so never accept one.
|
|
655
|
+
issues.append(
|
|
656
|
+
Issue(
|
|
657
|
+
code="EMPTY_RUN",
|
|
658
|
+
severity=Severity.ERROR,
|
|
659
|
+
message="per_question is empty — the run evaluated zero questions",
|
|
660
|
+
path="per_question",
|
|
661
|
+
)
|
|
662
|
+
)
|
|
652
663
|
if isinstance(reported, int) and reported != actual:
|
|
653
664
|
issues.append(
|
|
654
665
|
Issue(
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: synthbench-eval
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.2
|
|
4
4
|
Summary: Open benchmark harness for synthetic survey respondent quality
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
Keywords: benchmark,survey,synthetic-data,llm,evaluation,opinionsqa,globalopinionqa,gss,convergence,althing
|
|
@@ -7,7 +7,7 @@ import json
|
|
|
7
7
|
import pytest
|
|
8
8
|
|
|
9
9
|
from synthbench.runner import BenchmarkResult, QuestionResult
|
|
10
|
-
from synthbench.report import to_json, to_markdown
|
|
10
|
+
from synthbench.report import provider_slug, save, to_json, to_markdown
|
|
11
11
|
|
|
12
12
|
|
|
13
13
|
@pytest.fixture
|
|
@@ -160,3 +160,25 @@ def test_latency_appears_in_per_question_records():
|
|
|
160
160
|
pq = data["per_question"]
|
|
161
161
|
assert pq[0]["latency_seconds"] == pytest.approx(0.42)
|
|
162
162
|
assert "latency_seconds" not in pq[1]
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
@pytest.mark.parametrize(
|
|
166
|
+
("name", "expected"),
|
|
167
|
+
[
|
|
168
|
+
("althing/claude-code:haiku", "althing_claude-code_haiku"),
|
|
169
|
+
("openrouter/openai/gpt-4o-mini t=0.7", "openrouter_openai_gpt-4o-mini t=0.7"),
|
|
170
|
+
('a\\b*c?d"e<f>g|h', "a_b_c_d_e_f_g_h"),
|
|
171
|
+
],
|
|
172
|
+
)
|
|
173
|
+
def test_provider_slug_is_filename_safe(name, expected):
|
|
174
|
+
assert provider_slug(name) == expected
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def test_save_uses_filename_safe_slug(sample_result, tmp_path):
|
|
178
|
+
sample_result.provider_name = "althing/claude-code:haiku"
|
|
179
|
+
json_path, md_path = save(sample_result, tmp_path)
|
|
180
|
+
assert ":" not in json_path.name
|
|
181
|
+
assert json_path.name.startswith("test_althing_claude-code_haiku_")
|
|
182
|
+
card = json.loads(json_path.read_text(encoding="utf-8"))
|
|
183
|
+
assert card["config"]["provider"] == "althing/claude-code:haiku"
|
|
184
|
+
assert "█" in md_path.read_text(encoding="utf-8")
|
|
@@ -14,6 +14,7 @@ from synthbench.providers.base import (
|
|
|
14
14
|
from synthbench.runner import (
|
|
15
15
|
BenchmarkRunner,
|
|
16
16
|
DemographicGroupResult,
|
|
17
|
+
EmptyQuestionSetError,
|
|
17
18
|
_aggregate_token_usage,
|
|
18
19
|
_normalize_model_dist,
|
|
19
20
|
)
|
|
@@ -660,3 +661,40 @@ async def test_runner_records_per_question_latency(mock_dataset, mock_provider):
|
|
|
660
661
|
assert qr.latency_seconds >= 0.0
|
|
661
662
|
# Per-question latency cannot exceed the total run elapsed time.
|
|
662
663
|
assert qr.latency_seconds <= result.elapsed_seconds + 1e-3
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
@pytest.mark.asyncio
|
|
667
|
+
async def test_runner_refuses_suite_that_matches_no_questions(
|
|
668
|
+
mock_dataset, mock_provider
|
|
669
|
+
):
|
|
670
|
+
"""synthbench#353: a filter that matches nothing must not yield a score."""
|
|
671
|
+
runner = BenchmarkRunner(
|
|
672
|
+
dataset=mock_dataset, provider=mock_provider, samples_per_question=2
|
|
673
|
+
)
|
|
674
|
+
with pytest.raises(EmptyQuestionSetError) as exc:
|
|
675
|
+
await runner.run(question_keys=["BIOTECHC_W34", "DIFF1B_W29"])
|
|
676
|
+
msg = str(exc.value)
|
|
677
|
+
assert "matched 0 of" in msg
|
|
678
|
+
assert "'mock'" in msg
|
|
679
|
+
assert "opinionsqa" in msg # points at the dataset the suites come from
|
|
680
|
+
assert "--n" in msg
|
|
681
|
+
|
|
682
|
+
|
|
683
|
+
@pytest.mark.asyncio
|
|
684
|
+
async def test_runner_refuses_empty_dataset(mock_provider):
|
|
685
|
+
class EmptyDataset(Dataset):
|
|
686
|
+
@property
|
|
687
|
+
def name(self) -> str:
|
|
688
|
+
return "empty"
|
|
689
|
+
|
|
690
|
+
def load(self, n=None):
|
|
691
|
+
return []
|
|
692
|
+
|
|
693
|
+
def info(self) -> dict:
|
|
694
|
+
return {}
|
|
695
|
+
|
|
696
|
+
runner = BenchmarkRunner(
|
|
697
|
+
dataset=EmptyDataset(), provider=mock_provider, samples_per_question=2
|
|
698
|
+
)
|
|
699
|
+
with pytest.raises(EmptyQuestionSetError, match="loaded 0 questions"):
|
|
700
|
+
await runner.run()
|
|
@@ -236,6 +236,16 @@ class TestCountMismatch:
|
|
|
236
236
|
assert any(i.code == "COUNT_MISMATCH" for i in report.errors)
|
|
237
237
|
|
|
238
238
|
|
|
239
|
+
class TestEmptyRun:
|
|
240
|
+
def test_zero_question_run_rejected(self, clean_submission):
|
|
241
|
+
bad = copy.deepcopy(clean_submission)
|
|
242
|
+
bad["per_question"] = []
|
|
243
|
+
bad["aggregate"]["n_questions"] = 0
|
|
244
|
+
report = validate_submission(bad, tier2=False)
|
|
245
|
+
assert not report.ok
|
|
246
|
+
assert any(i.code == "EMPTY_RUN" for i in report.errors)
|
|
247
|
+
|
|
248
|
+
|
|
239
249
|
class TestParseFailurePlausibility:
|
|
240
250
|
"""PARSE_SUSPICIOUS was retired in sb-a613.
|
|
241
251
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/majority_baseline.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/population_baseline.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/entry_points.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|