synthbench-eval 0.6.2__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (149) hide show
  1. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/PKG-INFO +1 -1
  2. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/pyproject.toml +1 -1
  3. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/__init__.py +1 -1
  4. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/_parsing.py +126 -0
  5. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/runner.py +4 -0
  6. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/PKG-INFO +1 -1
  7. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_provider_parsing.py +108 -0
  8. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_config_id_consistency.py +12 -0
  9. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_runner.py +10 -0
  10. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/LICENSE +0 -0
  11. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/README.md +0 -0
  12. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/setup.cfg +0 -0
  13. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/__main__.py +0 -0
  14. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/adapter.py +0 -0
  15. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/anomaly.py +0 -0
  16. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/baseline_floors.py +0 -0
  17. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/baselines.py +0 -0
  18. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/cli.py +0 -0
  19. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/config_id.py +0 -0
  20. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/contamination.py +0 -0
  21. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/__init__.py +0 -0
  22. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/baseline.py +0 -0
  23. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/bootstrap.py +0 -0
  24. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/cli_report.py +0 -0
  25. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/curves.py +0 -0
  26. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/real_sampling.py +0 -0
  27. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/thresholds.py +0 -0
  28. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/__init__.py +0 -0
  29. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/base.py +0 -0
  30. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/eurobarometer.py +0 -0
  31. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/globalopinionqa.py +0 -0
  32. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/gss.py +0 -0
  33. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/michigan.py +0 -0
  34. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/ntia.py +0 -0
  35. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/opinionsqa.py +0 -0
  36. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/pewtech.py +0 -0
  37. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/policy.py +0 -0
  38. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/subpop.py +0 -0
  39. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/wvs.py +0 -0
  40. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/findings.py +0 -0
  41. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/holdout.py +0 -0
  42. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/human_distributions.py +0 -0
  43. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/leaderboard.py +0 -0
  44. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/leaderboard_pr.py +0 -0
  45. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/__init__.py +0 -0
  46. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/composite.py +0 -0
  47. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/conditioning.py +0 -0
  48. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/distributional.py +0 -0
  49. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/ranking.py +0 -0
  50. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/refusal.py +0 -0
  51. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/subgroup.py +0 -0
  52. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/private_holdout.py +0 -0
  53. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/__init__.py +0 -0
  54. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/_retry.py +0 -0
  55. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/althing.py +0 -0
  56. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/base.py +0 -0
  57. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/http.py +0 -0
  58. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/majority_baseline.py +0 -0
  59. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/ollama.py +0 -0
  60. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/openrouter.py +0 -0
  61. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/population_baseline.py +0 -0
  62. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/random_baseline.py +0 -0
  63. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/raw_anthropic.py +0 -0
  64. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/raw_gemini.py +0 -0
  65. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/raw_openai.py +0 -0
  66. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/publish.py +0 -0
  67. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/r2_upload.py +0 -0
  68. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/recompute.py +0 -0
  69. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/report.py +0 -0
  70. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/run_hash.py +0 -0
  71. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/run_validity.py +0 -0
  72. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/stats.py +0 -0
  73. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/submission.py +0 -0
  74. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/submission_pr.py +0 -0
  75. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/submit_adapter.py +0 -0
  76. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/suite.py +0 -0
  77. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/suites/__init__.py +0 -0
  78. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/topics.py +0 -0
  79. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/user_config.py +0 -0
  80. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/validation.py +0 -0
  81. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/visualize.py +0 -0
  82. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/SOURCES.txt +0 -0
  83. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/dependency_links.txt +0 -0
  84. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/entry_points.txt +0 -0
  85. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/requires.txt +0 -0
  86. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/top_level.txt +0 -0
  87. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_anomaly.py +0 -0
  88. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_baseline_floors.py +0 -0
  89. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_baselines.py +0 -0
  90. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_cli_run_submit.py +0 -0
  91. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_cli_submit.py +0 -0
  92. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_cli_submit_adapter.py +0 -0
  93. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_config_id.py +0 -0
  94. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_contamination.py +0 -0
  95. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_convergence_baseline.py +0 -0
  96. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_convergence_bootstrap.py +0 -0
  97. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_eurobarometer.py +0 -0
  98. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_globalopinionqa.py +0 -0
  99. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_gss.py +0 -0
  100. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_michigan.py +0 -0
  101. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_ntia.py +0 -0
  102. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_policy.py +0 -0
  103. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_wvs.py +0 -0
  104. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_effort.py +0 -0
  105. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_findings_drift.py +0 -0
  106. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_findings_elicitation.py +0 -0
  107. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_findings_nonresponse.py +0 -0
  108. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_holdout.py +0 -0
  109. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_human_distributions.py +0 -0
  110. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_integration_tokens.py +0 -0
  111. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_metrics.py +0 -0
  112. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_microdata.py +0 -0
  113. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_private_holdout.py +0 -0
  114. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_providers.py +0 -0
  115. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_cost.py +0 -0
  116. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_cross_provider_jsd.py +0 -0
  117. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_demographic_scorecard.py +0 -0
  118. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_gated_fail_closed.py +0 -0
  119. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_holdout.py +0 -0
  120. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_invalid_runs.py +0 -0
  121. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_normalized.py +0 -0
  122. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_policy.py +0 -0
  123. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_questions.py +0 -0
  124. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_r2_routing.py +0 -0
  125. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_rehydration.py +0 -0
  126. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_run_counts.py +0 -0
  127. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_runnable_ids.py +0 -0
  128. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_text_rehydration.py +0 -0
  129. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_topic_metrics.py +0 -0
  130. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_r2_upload.py +0 -0
  131. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_recompute_integrity.py +0 -0
  132. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_report.py +0 -0
  133. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_run_hash.py +0 -0
  134. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_run_validity.py +0 -0
  135. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_site_dataset_cards.py +0 -0
  136. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_stats.py +0 -0
  137. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_stats_golden.py +0 -0
  138. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_strip_gated_guard.py +0 -0
  139. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_submission.py +0 -0
  140. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_submission_pr.py +0 -0
  141. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_suite.py +0 -0
  142. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_suites.py +0 -0
  143. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_suites_novel_products.py +0 -0
  144. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_topics.py +0 -0
  145. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_user_config.py +0 -0
  146. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_validation.py +0 -0
  147. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_validation_holdout.py +0 -0
  148. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_validation_stripped.py +0 -0
  149. {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_verify_publish_integrity.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: synthbench-eval
3
- Version: 0.6.2
3
+ Version: 0.7.0
4
4
  Summary: Open benchmark harness for synthetic survey respondent quality
5
5
  License-Expression: MIT
6
6
  Keywords: benchmark,survey,synthetic-data,llm,evaluation,opinionsqa,globalopinionqa,gss,convergence,althing
@@ -8,7 +8,7 @@ build-backend = "setuptools.build_meta"
8
8
  name = "synthbench-eval"
9
9
  # Placeholder — the publish job stamps the real version from the release tag
10
10
  # at build time (see publish-pypi in .github/workflows/auto-tag.yml).
11
- version = "0.6.2"
11
+ version = "0.7.0"
12
12
  description = "Open benchmark harness for synthetic survey respondent quality"
13
13
  readme = "README.md"
14
14
  license = "MIT"
@@ -6,7 +6,7 @@ from synthbench.convergence.baseline import (
6
6
  load_convergence_baseline,
7
7
  )
8
8
 
9
- __version__ = "0.6.2"
9
+ __version__ = "0.7.0"
10
10
 
11
11
  __all__ = [
12
12
  "BaselineGatedError",
@@ -28,6 +28,24 @@ This module is the single replacement. Matching order:
28
28
 
29
29
  Anything else is a parse failure: ``option is None`` and ``refusal`` is
30
30
  False. Callers must NOT substitute a default option.
31
+
32
+ Parser versions (``config.option_parser_version``; absent key = v1):
33
+
34
+ * **v1** — the matching order above, verbatim. Kept callable so historical
35
+ runs re-parse bit-for-bit.
36
+ * **v2** (current, synthbench#352) — newer models answer in markdown prose
37
+ (``**(D) Depends on the situation.**`` followed by an explanation). v1
38
+ counted those as parse failures, and its whole-text containment could
39
+ mis-score them by matching an option mentioned in the explanation. v2:
40
+
41
+ - strips markdown emphasis/heading/quote markup before matching;
42
+ - accepts a *leading* option label (``(D) ...``, ``D. ...``, ``D) ...``)
43
+ as the answer, unless the text right after it exactly names a
44
+ different option (contradictory → parse failure);
45
+ - restricts containment to the first line, falling back to the whole
46
+ response only when exactly one option is contained in it.
47
+
48
+ Refusal detection is unchanged (it is versioned separately).
31
49
  """
32
50
 
33
51
  from __future__ import annotations
@@ -47,6 +65,28 @@ _BARE_LETTER_RE = re.compile(r"^\s*\(?([A-Za-z])\)?[.):]?\s*$")
47
65
 
48
66
  _WS_RE = re.compile(r"\s+")
49
67
 
68
+ #: Parser version stamped into run metadata (``config.option_parser_version``).
69
+ #: Absent key on a committed run file means v1.
70
+ OPTION_PARSER_VERSION = 2
71
+
72
+ # v2: markdown markup to drop before matching — emphasis runs (``**``,
73
+ # ``__``, ``*``), and line-leading heading / blockquote / bullet markers.
74
+ _MD_EMPHASIS_RE = re.compile(r"(\*{1,3}|_{2,3})")
75
+ _MD_LINE_PREFIX_RE = re.compile(
76
+ r"^[ \t]*(?:#{1,6}[ \t]+|>[ \t]*|[-+][ \t]+)", re.MULTILINE
77
+ )
78
+
79
+ # v2: leading option label, optionally after an "Answer:" style prefix.
80
+ # Parenthesized letters may be either case; bare letters must be uppercase
81
+ # and delimited (checked in _leading_label_match) so "I think..." or
82
+ # "A lot of people..." never parse as a label.
83
+ _LEADING_LABEL_RE = re.compile(
84
+ r"^\s*(?:(?:answer|my answer|choice)\s*[:\-]\s*)?"
85
+ r"(?:\(([A-Za-z])\)|([A-Za-z])[.):](?=\s|$))\s*(.*)",
86
+ re.IGNORECASE | re.DOTALL,
87
+ )
88
+ _PAREN_LABEL_RE = re.compile(r"\(([A-Za-z])\)")
89
+
50
90
 
51
91
  @dataclass(frozen=True)
52
92
  class ParsedResponse:
@@ -96,11 +136,91 @@ def _containment_match(text: str, options: list[str]) -> str | None:
96
136
  return None
97
137
 
98
138
 
139
+ def _strip_markdown(text: str) -> str:
140
+ """Drop markdown emphasis and line-leading markup (v2)."""
141
+ return _MD_EMPHASIS_RE.sub("", _MD_LINE_PREFIX_RE.sub("", text))
142
+
143
+
144
+ def _leading_label_match(text: str, options: list[str]) -> ParsedResponse | None:
145
+ """Resolve a leading ``(D) ...`` / ``D. ...`` label (v2).
146
+
147
+ Returns ``None`` when there is no label; a parse failure when the label
148
+ is out of range or contradicted by an exact option name right after it.
149
+ The lowercase bare form (``a. ...``) is not a label — only ``(a)``.
150
+ """
151
+ match = _LEADING_LABEL_RE.match(text)
152
+ if not match:
153
+ return None
154
+ paren, bare, rest = match.groups()
155
+ if bare is not None and not bare.isupper():
156
+ return None
157
+ letter = (paren or bare).upper()
158
+ idx = ord(letter) - ord("A")
159
+ if not 0 <= idx < len(options):
160
+ return ParsedResponse()
161
+ # "(B) Favor" where B is "Oppose": the label and the echoed text
162
+ # disagree, so neither can be trusted.
163
+ first_line = rest.strip().split("\n", 1)[0]
164
+ by_norm = {_normalize(opt): opt for opt in options}
165
+ echoed = by_norm.get(_normalize(first_line))
166
+ if echoed is not None and echoed != options[idx]:
167
+ return ParsedResponse()
168
+ return ParsedResponse(option=options[idx])
169
+
170
+
171
+ def _parse_v2_options(stripped: str, options: list[str]) -> ParsedResponse:
172
+ """v2 option matching on markdown-stripped text (after refusal checks)."""
173
+ text = _strip_markdown(stripped).strip()
174
+ if not text:
175
+ return ParsedResponse()
176
+
177
+ by_norm = {_normalize(opt): opt for opt in options}
178
+ text_norm = _normalize(text)
179
+ if text_norm in by_norm:
180
+ return ParsedResponse(option=by_norm[text_norm])
181
+
182
+ match = _BARE_LETTER_RE.match(text)
183
+ if match:
184
+ idx = ord(match.group(1).upper()) - ord("A")
185
+ if 0 <= idx < len(options):
186
+ return ParsedResponse(option=options[idx])
187
+ return ParsedResponse()
188
+
189
+ labelled = _leading_label_match(text, options)
190
+ if labelled is not None:
191
+ return labelled
192
+
193
+ lines = [ln for ln in text.splitlines() if ln.strip()]
194
+ # A parenthesized label after a short lead-in ("I'd say (B) agree",
195
+ # possibly below a roleplay line like "*thinks*"). Only when the opening
196
+ # lines carry exactly one distinct label; several means a list or a
197
+ # comparison, not an answer.
198
+ lead = "\n".join(lines[:2])
199
+ labels = list(_PAREN_LABEL_RE.finditer(lead))
200
+ if labels and len({m.group(1).upper() for m in labels}) == 1:
201
+ inline = _leading_label_match(lead[labels[0].start() :], options)
202
+ if inline is not None:
203
+ return inline
204
+
205
+ first_line = lines[0] if lines else ""
206
+ contained = _containment_match(first_line, options)
207
+ if contained is not None:
208
+ return ParsedResponse(option=contained)
209
+
210
+ # Whole-response containment only when unambiguous: an explanation that
211
+ # mentions several options must not be scored as whichever matched first.
212
+ hits = [opt for opt in options if _containment_match(text, [opt]) is not None]
213
+ if len(hits) == 1:
214
+ return ParsedResponse(option=hits[0])
215
+ return ParsedResponse()
216
+
217
+
99
218
  def parse_option_response(
100
219
  text: str,
101
220
  options: list[str],
102
221
  *,
103
222
  refusal_detector_version: int = REFUSAL_DETECTOR_VERSION,
223
+ option_parser_version: int = OPTION_PARSER_VERSION,
104
224
  ) -> ParsedResponse:
105
225
  """Parse a raw model response into an option selection, refusal, or failure.
106
226
 
@@ -111,6 +231,9 @@ def parse_option_response(
111
231
  answer-initial anchoring + option-echo exemption) or 1 (the legacy
112
232
  un-anchored patterns, kept callable so historical runs can be
113
233
  reproduced bit-for-bit).
234
+
235
+ ``option_parser_version`` selects option matching the same way: 2
236
+ (default, markdown-aware) or 1 (legacy).
114
237
  """
115
238
  raw = str(text)
116
239
  stripped = raw.strip()
@@ -134,6 +257,9 @@ def parse_option_response(
134
257
  if is_refusal:
135
258
  return ParsedResponse(refusal=True)
136
259
 
260
+ if option_parser_version >= 2:
261
+ return _parse_v2_options(stripped, options)
262
+
137
263
  # 3. Bare letter, anchored full-match only.
138
264
  match = _BARE_LETTER_RE.match(stripped)
139
265
  if match:
@@ -22,6 +22,7 @@ from synthbench.metrics import (
22
22
  extract_human_refusal_rate,
23
23
  conditioning_fidelity,
24
24
  )
25
+ from synthbench.providers._parsing import OPTION_PARSER_VERSION
25
26
  from synthbench.providers.base import Distribution, PersonaSpec, Provider, Response
26
27
  from synthbench.stats import bootstrap_ci, question_set_hash
27
28
 
@@ -476,6 +477,9 @@ class BenchmarkRunner:
476
477
  # this run's raw responses. Absent on pre-v2 files (= v1).
477
478
  # Never feeds build_config_id, so config_ids are stable.
478
479
  "refusal_detector_version": REFUSAL_DETECTOR_VERSION,
480
+ # Same contract for option matching (synthbench#352): absent
481
+ # on pre-v2 files (= v1); never feeds build_config_id.
482
+ "option_parser_version": OPTION_PARSER_VERSION,
479
483
  **_provider_reproducibility_hashes(self.provider),
480
484
  },
481
485
  elapsed_seconds=elapsed,
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: synthbench-eval
3
- Version: 0.6.2
3
+ Version: 0.7.0
4
4
  Summary: Open benchmark harness for synthetic survey respondent quality
5
5
  License-Expression: MIT
6
6
  Keywords: benchmark,survey,synthetic-data,llm,evaluation,opinionsqa,globalopinionqa,gss,convergence,althing
@@ -846,6 +846,114 @@ def test_refusal_detector_version_constant():
846
846
  assert REFUSAL_DETECTOR_VERSION == 2
847
847
 
848
848
 
849
+ # ---------------------------------------------------------------------------
850
+ # Option parser v2 (synthbench#352) — markdown-aware matching
851
+ # ---------------------------------------------------------------------------
852
+
853
+ _OPTS_ALCOHOL = [
854
+ "Morally acceptable",
855
+ "Morally unacceptable",
856
+ "Not a moral issue",
857
+ "Depends on the situation (VOL)",
858
+ ]
859
+ _OPTS_CHINA = [
860
+ "Next 10 years",
861
+ "Next 20 years",
862
+ "Next 50 years",
863
+ "China will not replace U.S.",
864
+ ]
865
+ _OPTS_CONFIDENCE = ["a great deal", "only some", "hardly any", "don't know"]
866
+
867
+
868
+ @pytest.mark.parametrize(
869
+ ("text", "options", "expected_v2", "expected_v1"),
870
+ [
871
+ # Markdown-bolded leading label + explanation that mentions another
872
+ # option: v1 mis-scored via whole-text containment.
873
+ (
874
+ "**(D) Depends on the situation.**\n\nA drink with friends is fine. "
875
+ "For most people it is not a moral issue.",
876
+ _OPTS_ALCOHOL,
877
+ "Depends on the situation (VOL)",
878
+ "Not a moral issue",
879
+ ),
880
+ (
881
+ "**(D) China will not replace the U.S.**\n\nMaybe the gap closes in the next 20 years.",
882
+ _OPTS_CHINA,
883
+ "China will not replace U.S.",
884
+ "Next 20 years",
885
+ ),
886
+ # Label after a lead-in, below a roleplay line (from committed runs).
887
+ (
888
+ "I'd say **(B) only some confidence**.\n\nThey could do a great deal better.",
889
+ _OPTS_CONFIDENCE,
890
+ "only some",
891
+ "a great deal",
892
+ ),
893
+ ("D. Depends", _OPTS_ALCOHOL, "Depends on the situation (VOL)", None),
894
+ ("Answer: (C)", _OPTS_ALCOHOL, "Not a moral issue", None),
895
+ (
896
+ "**Not a moral issue**",
897
+ _OPTS_ALCOHOL,
898
+ "Not a moral issue",
899
+ "Not a moral issue",
900
+ ),
901
+ (
902
+ "## (A) Morally acceptable\nBecause.",
903
+ _OPTS_ALCOHOL,
904
+ "Morally acceptable",
905
+ "Morally acceptable",
906
+ ),
907
+ # Unchanged plain forms.
908
+ ("B", _OPTS_ALCOHOL, "Morally unacceptable", "Morally unacceptable"),
909
+ (
910
+ "(B) Morally unacceptable",
911
+ _OPTS_ALCOHOL,
912
+ "Morally unacceptable",
913
+ "Morally unacceptable",
914
+ ),
915
+ ],
916
+ )
917
+ def test_option_parser_v2_markdown_answers(text, options, expected_v2, expected_v1):
918
+ assert parse_option_response(text, options).option == expected_v2
919
+ assert (
920
+ parse_option_response(text, options, option_parser_version=1).option
921
+ == expected_v1
922
+ )
923
+
924
+
925
+ @pytest.mark.parametrize(
926
+ "text",
927
+ [
928
+ "(B) Morally acceptable", # label and echoed option disagree
929
+ "(Z) whatever", # label out of range
930
+ "I think A.", # not a label
931
+ "a. something", # lowercase bare letter is not a label
932
+ "Hard to say. (A) and (C) both have merit.", # several labels
933
+ # Explanation names two options and no answer line resolves it.
934
+ "Hard to say.\nSome think it is not a moral issue; others find it morally unacceptable.",
935
+ ],
936
+ )
937
+ def test_option_parser_v2_refuses_ambiguous_answers(text):
938
+ assert parse_option_response(text, _OPTS_ALCOHOL) == ParsedResponse()
939
+
940
+
941
+ def test_option_parser_v2_unambiguous_whole_text_containment():
942
+ text = "Hard to say.\nIn the end it is not a moral issue for me."
943
+ assert parse_option_response(text, _OPTS_ALCOHOL).option == "Not a moral issue"
944
+
945
+
946
+ def test_option_parser_v2_keeps_refusal_detection():
947
+ text = "**I'd rather not answer that.**"
948
+ assert parse_option_response(text, _OPTS_ALCOHOL) == ParsedResponse(refusal=True)
949
+
950
+
951
+ def test_option_parser_version_constant():
952
+ from synthbench.providers._parsing import OPTION_PARSER_VERSION
953
+
954
+ assert OPTION_PARSER_VERSION == 2
955
+
956
+
849
957
  # ---------------------------------------------------------------------------
850
958
  # Structured elicitation (tpl=structured) — schema-forced extraction
851
959
  # ---------------------------------------------------------------------------
@@ -214,6 +214,18 @@ def test_refusal_detector_version_stamp_never_changes_config_id():
214
214
  )
215
215
 
216
216
 
217
+ def test_option_parser_version_stamp_never_changes_config_id():
218
+ """``config.option_parser_version`` is metadata-only, like the refusal stamp."""
219
+ for provider in PROVIDER_FIXTURES:
220
+ base = _result(provider)
221
+ stamped = _result(provider)
222
+ stamped["config"]["option_parser_version"] = 2
223
+ assert (
224
+ _build_entry(base, rank=1)["config_id"]
225
+ == _build_entry(stamped, rank=1)["config_id"]
226
+ ), provider
227
+
228
+
217
229
  def test_all_committed_config_ids_unchanged_by_detector_stamp():
218
230
  """Recompute every committed run's config_id with and without the stamp.
219
231
 
@@ -698,3 +698,13 @@ async def test_runner_refuses_empty_dataset(mock_provider):
698
698
  )
699
699
  with pytest.raises(EmptyQuestionSetError, match="loaded 0 questions"):
700
700
  await runner.run()
701
+
702
+
703
+ async def test_runner_stamps_option_parser_version(mock_dataset, mock_provider):
704
+ from synthbench.providers._parsing import OPTION_PARSER_VERSION
705
+
706
+ runner = BenchmarkRunner(
707
+ dataset=mock_dataset, provider=mock_provider, samples_per_question=2
708
+ )
709
+ result = await runner.run(n=1)
710
+ assert result.config["option_parser_version"] == OPTION_PARSER_VERSION
File without changes