synthbench-eval 0.6.0__tar.gz → 0.6.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (149) hide show
  1. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/PKG-INFO +1 -1
  2. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/pyproject.toml +1 -1
  3. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/__init__.py +1 -1
  4. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/cli.py +22 -5
  5. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/report.py +18 -4
  6. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/runner.py +35 -0
  7. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/suites/__init__.py +4 -0
  8. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/validation.py +11 -0
  9. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/PKG-INFO +1 -1
  10. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_report.py +23 -1
  11. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_runner.py +38 -0
  12. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_validation.py +10 -0
  13. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/LICENSE +0 -0
  14. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/README.md +0 -0
  15. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/setup.cfg +0 -0
  16. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/__main__.py +0 -0
  17. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/adapter.py +0 -0
  18. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/anomaly.py +0 -0
  19. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/baseline_floors.py +0 -0
  20. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/baselines.py +0 -0
  21. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/config_id.py +0 -0
  22. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/contamination.py +0 -0
  23. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/__init__.py +0 -0
  24. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/baseline.py +0 -0
  25. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/bootstrap.py +0 -0
  26. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/cli_report.py +0 -0
  27. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/curves.py +0 -0
  28. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/real_sampling.py +0 -0
  29. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/convergence/thresholds.py +0 -0
  30. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/__init__.py +0 -0
  31. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/base.py +0 -0
  32. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/eurobarometer.py +0 -0
  33. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/globalopinionqa.py +0 -0
  34. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/gss.py +0 -0
  35. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/michigan.py +0 -0
  36. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/ntia.py +0 -0
  37. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/opinionsqa.py +0 -0
  38. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/pewtech.py +0 -0
  39. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/policy.py +0 -0
  40. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/subpop.py +0 -0
  41. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/datasets/wvs.py +0 -0
  42. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/findings.py +0 -0
  43. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/holdout.py +0 -0
  44. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/human_distributions.py +0 -0
  45. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/leaderboard.py +0 -0
  46. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/leaderboard_pr.py +0 -0
  47. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/__init__.py +0 -0
  48. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/composite.py +0 -0
  49. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/conditioning.py +0 -0
  50. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/distributional.py +0 -0
  51. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/ranking.py +0 -0
  52. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/refusal.py +0 -0
  53. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/metrics/subgroup.py +0 -0
  54. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/private_holdout.py +0 -0
  55. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/__init__.py +0 -0
  56. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/_parsing.py +0 -0
  57. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/_retry.py +0 -0
  58. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/althing.py +0 -0
  59. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/base.py +0 -0
  60. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/http.py +0 -0
  61. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/majority_baseline.py +0 -0
  62. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/ollama.py +0 -0
  63. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/openrouter.py +0 -0
  64. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/population_baseline.py +0 -0
  65. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/random_baseline.py +0 -0
  66. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/raw_anthropic.py +0 -0
  67. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/raw_gemini.py +0 -0
  68. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/providers/raw_openai.py +0 -0
  69. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/publish.py +0 -0
  70. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/r2_upload.py +0 -0
  71. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/recompute.py +0 -0
  72. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/run_hash.py +0 -0
  73. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/run_validity.py +0 -0
  74. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/stats.py +0 -0
  75. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/submission.py +0 -0
  76. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/submission_pr.py +0 -0
  77. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/submit_adapter.py +0 -0
  78. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/suite.py +0 -0
  79. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/topics.py +0 -0
  80. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/user_config.py +0 -0
  81. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench/visualize.py +0 -0
  82. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/SOURCES.txt +0 -0
  83. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/dependency_links.txt +0 -0
  84. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/entry_points.txt +0 -0
  85. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/requires.txt +0 -0
  86. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/src/synthbench_eval.egg-info/top_level.txt +0 -0
  87. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_anomaly.py +0 -0
  88. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_baseline_floors.py +0 -0
  89. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_baselines.py +0 -0
  90. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_cli_run_submit.py +0 -0
  91. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_cli_submit.py +0 -0
  92. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_cli_submit_adapter.py +0 -0
  93. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_config_id.py +0 -0
  94. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_contamination.py +0 -0
  95. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_convergence_baseline.py +0 -0
  96. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_convergence_bootstrap.py +0 -0
  97. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_eurobarometer.py +0 -0
  98. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_globalopinionqa.py +0 -0
  99. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_gss.py +0 -0
  100. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_michigan.py +0 -0
  101. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_ntia.py +0 -0
  102. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_policy.py +0 -0
  103. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_datasets_wvs.py +0 -0
  104. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_effort.py +0 -0
  105. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_findings_drift.py +0 -0
  106. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_findings_elicitation.py +0 -0
  107. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_findings_nonresponse.py +0 -0
  108. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_holdout.py +0 -0
  109. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_human_distributions.py +0 -0
  110. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_integration_tokens.py +0 -0
  111. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_metrics.py +0 -0
  112. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_microdata.py +0 -0
  113. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_private_holdout.py +0 -0
  114. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_provider_parsing.py +0 -0
  115. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_providers.py +0 -0
  116. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_config_id_consistency.py +0 -0
  117. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_cost.py +0 -0
  118. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_cross_provider_jsd.py +0 -0
  119. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_demographic_scorecard.py +0 -0
  120. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_gated_fail_closed.py +0 -0
  121. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_holdout.py +0 -0
  122. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_invalid_runs.py +0 -0
  123. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_normalized.py +0 -0
  124. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_policy.py +0 -0
  125. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_questions.py +0 -0
  126. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_r2_routing.py +0 -0
  127. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_rehydration.py +0 -0
  128. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_run_counts.py +0 -0
  129. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_runnable_ids.py +0 -0
  130. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_text_rehydration.py +0 -0
  131. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_publish_topic_metrics.py +0 -0
  132. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_r2_upload.py +0 -0
  133. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_recompute_integrity.py +0 -0
  134. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_run_hash.py +0 -0
  135. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_run_validity.py +0 -0
  136. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_site_dataset_cards.py +0 -0
  137. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_stats.py +0 -0
  138. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_stats_golden.py +0 -0
  139. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_strip_gated_guard.py +0 -0
  140. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_submission.py +0 -0
  141. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_submission_pr.py +0 -0
  142. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_suite.py +0 -0
  143. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_suites.py +0 -0
  144. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_suites_novel_products.py +0 -0
  145. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_topics.py +0 -0
  146. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_user_config.py +0 -0
  147. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_validation_holdout.py +0 -0
  148. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_validation_stripped.py +0 -0
  149. {synthbench_eval-0.6.0 → synthbench_eval-0.6.2}/tests/test_verify_publish_integrity.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: synthbench-eval
3
- Version: 0.6.0
3
+ Version: 0.6.2
4
4
  Summary: Open benchmark harness for synthetic survey respondent quality
5
5
  License-Expression: MIT
6
6
  Keywords: benchmark,survey,synthetic-data,llm,evaluation,opinionsqa,globalopinionqa,gss,convergence,althing
@@ -8,7 +8,7 @@ build-backend = "setuptools.build_meta"
8
8
  name = "synthbench-eval"
9
9
  # Placeholder — the publish job stamps the real version from the release tag
10
10
  # at build time (see publish-pypi in .github/workflows/auto-tag.yml).
11
- version = "0.6.0"
11
+ version = "0.6.2"
12
12
  description = "Open benchmark harness for synthetic survey respondent quality"
13
13
  readme = "README.md"
14
14
  license = "MIT"
@@ -6,7 +6,7 @@ from synthbench.convergence.baseline import (
6
6
  load_convergence_baseline,
7
7
  )
8
8
 
9
- __version__ = "0.6.0"
9
+ __version__ = "0.6.2"
10
10
 
11
11
  __all__ = [
12
12
  "BaselineGatedError",
@@ -390,7 +390,7 @@ def run(
390
390
  )
391
391
  sys.exit(2)
392
392
 
393
- asyncio.run(
393
+ _run_benchmark_or_exit(
394
394
  _run_async(
395
395
  provider,
396
396
  model,
@@ -423,6 +423,17 @@ def run(
423
423
  )
424
424
 
425
425
 
426
+ def _run_benchmark_or_exit(coro) -> None:
427
+ """Run a benchmark coroutine, turning an empty question set into a clean CLI error."""
428
+ from synthbench.runner import EmptyQuestionSetError
429
+
430
+ try:
431
+ asyncio.run(coro)
432
+ except EmptyQuestionSetError as exc:
433
+ click.echo(f"Error: {exc}", err=True)
434
+ sys.exit(2)
435
+
436
+
426
437
  async def _run_async(
427
438
  provider_name,
428
439
  model,
@@ -1147,7 +1158,7 @@ def replicate(
1147
1158
  Example:
1148
1159
  synthbench replicate --provider raw-anthropic --n-runs 5 --suite core
1149
1160
  """
1150
- asyncio.run(
1161
+ _run_benchmark_or_exit(
1151
1162
  _replicate_async(
1152
1163
  provider,
1153
1164
  model,
@@ -1284,8 +1295,10 @@ async def _replicate_async(
1284
1295
  out_dir.mkdir(parents=True, exist_ok=True)
1285
1296
  from datetime import datetime
1286
1297
 
1298
+ from synthbench.report import provider_slug as report_provider_slug
1299
+
1287
1300
  ts = datetime.now().strftime("%Y%m%d_%H%M%S")
1288
- provider_slug = prov.name.replace("/", "_")
1301
+ provider_slug = report_provider_slug(prov.name)
1289
1302
 
1290
1303
  json_data = {
1291
1304
  "benchmark": "synthbench",
@@ -2018,8 +2031,10 @@ async def _contamination_async(
2018
2031
  out_dir.mkdir(parents=True, exist_ok=True)
2019
2032
  from datetime import datetime
2020
2033
 
2034
+ from synthbench.report import provider_slug as report_provider_slug
2035
+
2021
2036
  ts = datetime.now().strftime("%Y%m%d_%H%M%S")
2022
- provider_slug = prov.name.replace("/", "_")
2037
+ provider_slug = report_provider_slug(prov.name)
2023
2038
  json_path = out_dir / f"contamination_{provider_slug}_{ts}.json"
2024
2039
  json_path.write_text(json.dumps(result_data, indent=2))
2025
2040
  click.echo(f"Results saved: {json_path}")
@@ -2152,8 +2167,10 @@ async def _contamination_deident_async(
2152
2167
  out_dir.mkdir(parents=True, exist_ok=True)
2153
2168
  from datetime import datetime
2154
2169
 
2170
+ from synthbench.report import provider_slug as report_provider_slug
2171
+
2155
2172
  ts = datetime.now().strftime("%Y%m%d_%H%M%S")
2156
- provider_slug = prov.name.replace("/", "_")
2173
+ provider_slug = report_provider_slug(prov.name)
2157
2174
  json_path = out_dir / f"contamination_deident_{provider_slug}_{ts}.json"
2158
2175
  json_path.write_text(json.dumps(result_data, indent=2))
2159
2176
  click.echo(f"Results saved: {json_path}")
@@ -3,6 +3,7 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import json
6
+ import re
6
7
  from datetime import datetime, timezone
7
8
  from pathlib import Path
8
9
 
@@ -469,6 +470,18 @@ def to_markdown(
469
470
  return "\n".join(lines)
470
471
 
471
472
 
473
+ # Characters that are path separators or invalid in Windows filenames. A ':'
474
+ # is the dangerous one: NTFS treats ``name:rest`` as an alternate data
475
+ # stream, so a provider like ``althing/claude-code:haiku`` would silently
476
+ # write its score card into a hidden stream of a truncated file.
477
+ _UNSAFE_FILENAME_CHARS = re.compile(r'[\\/:*?"<>|]')
478
+
479
+
480
+ def provider_slug(provider_name: str) -> str:
481
+ """Return *provider_name* made safe for use in a result filename."""
482
+ return _UNSAFE_FILENAME_CHARS.sub("_", provider_name)
483
+
484
+
472
485
  def save(result: BenchmarkResult, output_dir: Path | str) -> tuple[Path, Path]:
473
486
  """Save JSON and markdown score cards to output_dir.
474
487
 
@@ -478,16 +491,17 @@ def save(result: BenchmarkResult, output_dir: Path | str) -> tuple[Path, Path]:
478
491
  output_dir.mkdir(parents=True, exist_ok=True)
479
492
 
480
493
  ts = datetime.now().strftime("%Y%m%d_%H%M%S")
481
- provider_slug = result.provider_name.replace("/", "_")
482
- base = f"{result.dataset_name}_{provider_slug}_{ts}"
494
+ base = f"{result.dataset_name}_{provider_slug(result.provider_name)}_{ts}"
483
495
 
484
496
  json_path = output_dir / f"{base}.json"
485
497
  md_path = output_dir / f"{base}.md"
486
498
 
487
- with open(json_path, "w") as f:
499
+ # Explicit UTF-8: the markdown card contains box-drawing bars (█░) that
500
+ # the Windows default codepage (cp1252) cannot encode.
501
+ with open(json_path, "w", encoding="utf-8") as f:
488
502
  json.dump(to_json(result), f, indent=2)
489
503
 
490
- with open(md_path, "w") as f:
504
+ with open(md_path, "w", encoding="utf-8") as f:
491
505
  f.write(to_markdown(result))
492
506
 
493
507
  return json_path, md_path
@@ -26,6 +26,33 @@ from synthbench.providers.base import Distribution, PersonaSpec, Provider, Respo
26
26
  from synthbench.stats import bootstrap_ci, question_set_hash
27
27
 
28
28
 
29
+ class EmptyQuestionSetError(ValueError):
30
+ """Raised when a run would evaluate zero questions.
31
+
32
+ Scoring an empty set yields empty-input defaults that look like a real
33
+ SPS (synthbench#353), so the runner refuses instead.
34
+ """
35
+
36
+
37
+ def _empty_question_set_message(
38
+ dataset_name: str, *, loaded: int, filtered: bool
39
+ ) -> str:
40
+ if not filtered:
41
+ return f"Dataset '{dataset_name}' loaded 0 questions; nothing to evaluate."
42
+ from synthbench.suites import SUITE_SOURCE_DATASET
43
+
44
+ msg = (
45
+ f"The --suite/--topic filter matched 0 of {loaded} questions in dataset "
46
+ f"'{dataset_name}'; nothing to evaluate."
47
+ )
48
+ if dataset_name != SUITE_SOURCE_DATASET:
49
+ msg += (
50
+ f" Pinned suites and topics are built from '{SUITE_SOURCE_DATASET}' "
51
+ "question keys — use --n to size a run on this dataset."
52
+ )
53
+ return msg
54
+
55
+
29
56
  def _sha256_of(s: str) -> str:
30
57
  """Return ``sha256:<hex>`` digest of a UTF-8 string."""
31
58
  return "sha256:" + hashlib.sha256(s.encode("utf-8")).hexdigest()
@@ -405,6 +432,7 @@ class BenchmarkRunner:
405
432
  ) -> BenchmarkResult:
406
433
  t0 = time.monotonic()
407
434
  questions = self.dataset.load(n=n)
435
+ loaded = len(questions)
408
436
 
409
437
  # Filter by pinned question set if provided
410
438
  if question_keys is not None:
@@ -412,6 +440,13 @@ class BenchmarkRunner:
412
440
 
413
441
  questions = filter_questions_by_suite(questions, question_keys)
414
442
 
443
+ if not questions:
444
+ raise EmptyQuestionSetError(
445
+ _empty_question_set_message(
446
+ self.dataset.name, loaded=loaded, filtered=question_keys is not None
447
+ )
448
+ )
449
+
415
450
  # Use batched evaluation when provider supports it
416
451
  use_batch = self.provider.supports_distribution and hasattr(
417
452
  self.provider, "batch_get_distribution"
@@ -25,6 +25,10 @@ AVAILABLE_SUITES = ("smoke", "core", "full")
25
25
 
26
26
  AVAILABLE_TOPICS = ("political", "consumer", "neutral")
27
27
 
28
+ # Every pinned suite and topic set above is built from OpinionsQA question
29
+ # keys (``<VAR>_W<wave>``), so against any other dataset they match nothing.
30
+ SUITE_SOURCE_DATASET = "opinionsqa"
31
+
28
32
 
29
33
  def load_suite(name: str) -> list[str] | None:
30
34
  """Load a pinned question set by name.
@@ -649,6 +649,17 @@ def _validate_counts(data: Mapping[str, Any]) -> list[Issue]:
649
649
  per_question = data.get("per_question") or []
650
650
  reported = aggregate.get("n_questions")
651
651
  actual = len(per_question) if isinstance(per_question, list) else 0
652
+ if isinstance(per_question, list) and actual == 0:
653
+ # A zero-question run has no evidence behind it; its scores are
654
+ # empty-input defaults (synthbench#353), so never accept one.
655
+ issues.append(
656
+ Issue(
657
+ code="EMPTY_RUN",
658
+ severity=Severity.ERROR,
659
+ message="per_question is empty — the run evaluated zero questions",
660
+ path="per_question",
661
+ )
662
+ )
652
663
  if isinstance(reported, int) and reported != actual:
653
664
  issues.append(
654
665
  Issue(
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: synthbench-eval
3
- Version: 0.6.0
3
+ Version: 0.6.2
4
4
  Summary: Open benchmark harness for synthetic survey respondent quality
5
5
  License-Expression: MIT
6
6
  Keywords: benchmark,survey,synthetic-data,llm,evaluation,opinionsqa,globalopinionqa,gss,convergence,althing
@@ -7,7 +7,7 @@ import json
7
7
  import pytest
8
8
 
9
9
  from synthbench.runner import BenchmarkResult, QuestionResult
10
- from synthbench.report import to_json, to_markdown
10
+ from synthbench.report import provider_slug, save, to_json, to_markdown
11
11
 
12
12
 
13
13
  @pytest.fixture
@@ -160,3 +160,25 @@ def test_latency_appears_in_per_question_records():
160
160
  pq = data["per_question"]
161
161
  assert pq[0]["latency_seconds"] == pytest.approx(0.42)
162
162
  assert "latency_seconds" not in pq[1]
163
+
164
+
165
+ @pytest.mark.parametrize(
166
+ ("name", "expected"),
167
+ [
168
+ ("althing/claude-code:haiku", "althing_claude-code_haiku"),
169
+ ("openrouter/openai/gpt-4o-mini t=0.7", "openrouter_openai_gpt-4o-mini t=0.7"),
170
+ ('a\\b*c?d"e<f>g|h', "a_b_c_d_e_f_g_h"),
171
+ ],
172
+ )
173
+ def test_provider_slug_is_filename_safe(name, expected):
174
+ assert provider_slug(name) == expected
175
+
176
+
177
+ def test_save_uses_filename_safe_slug(sample_result, tmp_path):
178
+ sample_result.provider_name = "althing/claude-code:haiku"
179
+ json_path, md_path = save(sample_result, tmp_path)
180
+ assert ":" not in json_path.name
181
+ assert json_path.name.startswith("test_althing_claude-code_haiku_")
182
+ card = json.loads(json_path.read_text(encoding="utf-8"))
183
+ assert card["config"]["provider"] == "althing/claude-code:haiku"
184
+ assert "█" in md_path.read_text(encoding="utf-8")
@@ -14,6 +14,7 @@ from synthbench.providers.base import (
14
14
  from synthbench.runner import (
15
15
  BenchmarkRunner,
16
16
  DemographicGroupResult,
17
+ EmptyQuestionSetError,
17
18
  _aggregate_token_usage,
18
19
  _normalize_model_dist,
19
20
  )
@@ -660,3 +661,40 @@ async def test_runner_records_per_question_latency(mock_dataset, mock_provider):
660
661
  assert qr.latency_seconds >= 0.0
661
662
  # Per-question latency cannot exceed the total run elapsed time.
662
663
  assert qr.latency_seconds <= result.elapsed_seconds + 1e-3
664
+
665
+
666
+ @pytest.mark.asyncio
667
+ async def test_runner_refuses_suite_that_matches_no_questions(
668
+ mock_dataset, mock_provider
669
+ ):
670
+ """synthbench#353: a filter that matches nothing must not yield a score."""
671
+ runner = BenchmarkRunner(
672
+ dataset=mock_dataset, provider=mock_provider, samples_per_question=2
673
+ )
674
+ with pytest.raises(EmptyQuestionSetError) as exc:
675
+ await runner.run(question_keys=["BIOTECHC_W34", "DIFF1B_W29"])
676
+ msg = str(exc.value)
677
+ assert "matched 0 of" in msg
678
+ assert "'mock'" in msg
679
+ assert "opinionsqa" in msg # points at the dataset the suites come from
680
+ assert "--n" in msg
681
+
682
+
683
+ @pytest.mark.asyncio
684
+ async def test_runner_refuses_empty_dataset(mock_provider):
685
+ class EmptyDataset(Dataset):
686
+ @property
687
+ def name(self) -> str:
688
+ return "empty"
689
+
690
+ def load(self, n=None):
691
+ return []
692
+
693
+ def info(self) -> dict:
694
+ return {}
695
+
696
+ runner = BenchmarkRunner(
697
+ dataset=EmptyDataset(), provider=mock_provider, samples_per_question=2
698
+ )
699
+ with pytest.raises(EmptyQuestionSetError, match="loaded 0 questions"):
700
+ await runner.run()
@@ -236,6 +236,16 @@ class TestCountMismatch:
236
236
  assert any(i.code == "COUNT_MISMATCH" for i in report.errors)
237
237
 
238
238
 
239
+ class TestEmptyRun:
240
+ def test_zero_question_run_rejected(self, clean_submission):
241
+ bad = copy.deepcopy(clean_submission)
242
+ bad["per_question"] = []
243
+ bad["aggregate"]["n_questions"] = 0
244
+ report = validate_submission(bad, tier2=False)
245
+ assert not report.ok
246
+ assert any(i.code == "EMPTY_RUN" for i in report.errors)
247
+
248
+
239
249
  class TestParseFailurePlausibility:
240
250
  """PARSE_SUSPICIOUS was retired in sb-a613.
241
251
 
File without changes