synthbench-eval 0.6.2__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/PKG-INFO +1 -1
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/pyproject.toml +1 -1
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/__init__.py +1 -1
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/_parsing.py +126 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/runner.py +4 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/PKG-INFO +1 -1
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_provider_parsing.py +108 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_config_id_consistency.py +12 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_runner.py +10 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/LICENSE +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/README.md +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/setup.cfg +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/__main__.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/adapter.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/anomaly.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/baseline_floors.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/baselines.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/cli.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/config_id.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/contamination.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/__init__.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/baseline.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/bootstrap.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/cli_report.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/curves.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/real_sampling.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/convergence/thresholds.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/__init__.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/base.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/eurobarometer.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/globalopinionqa.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/gss.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/michigan.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/ntia.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/opinionsqa.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/pewtech.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/policy.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/subpop.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/datasets/wvs.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/findings.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/holdout.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/human_distributions.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/leaderboard.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/leaderboard_pr.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/__init__.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/composite.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/conditioning.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/distributional.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/ranking.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/refusal.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/metrics/subgroup.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/private_holdout.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/__init__.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/_retry.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/althing.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/base.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/http.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/majority_baseline.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/ollama.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/openrouter.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/population_baseline.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/random_baseline.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/raw_anthropic.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/raw_gemini.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/raw_openai.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/publish.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/r2_upload.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/recompute.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/report.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/run_hash.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/run_validity.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/stats.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/submission.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/submission_pr.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/submit_adapter.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/suite.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/suites/__init__.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/topics.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/user_config.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/validation.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/visualize.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/SOURCES.txt +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/dependency_links.txt +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/entry_points.txt +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/requires.txt +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/top_level.txt +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_anomaly.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_baseline_floors.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_baselines.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_cli_run_submit.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_cli_submit.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_cli_submit_adapter.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_config_id.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_contamination.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_convergence_baseline.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_convergence_bootstrap.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_eurobarometer.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_globalopinionqa.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_gss.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_michigan.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_ntia.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_policy.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_datasets_wvs.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_effort.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_findings_drift.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_findings_elicitation.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_findings_nonresponse.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_holdout.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_human_distributions.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_integration_tokens.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_metrics.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_microdata.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_private_holdout.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_providers.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_cost.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_cross_provider_jsd.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_demographic_scorecard.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_gated_fail_closed.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_holdout.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_invalid_runs.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_normalized.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_policy.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_questions.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_r2_routing.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_rehydration.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_run_counts.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_runnable_ids.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_text_rehydration.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_publish_topic_metrics.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_r2_upload.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_recompute_integrity.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_report.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_run_hash.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_run_validity.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_site_dataset_cards.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_stats.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_stats_golden.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_strip_gated_guard.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_submission.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_submission_pr.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_suite.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_suites.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_suites_novel_products.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_topics.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_user_config.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_validation.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_validation_holdout.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_validation_stripped.py +0 -0
- {synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/tests/test_verify_publish_integrity.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: synthbench-eval
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: Open benchmark harness for synthetic survey respondent quality
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
Keywords: benchmark,survey,synthetic-data,llm,evaluation,opinionsqa,globalopinionqa,gss,convergence,althing
|
|
@@ -8,7 +8,7 @@ build-backend = "setuptools.build_meta"
|
|
|
8
8
|
name = "synthbench-eval"
|
|
9
9
|
# Placeholder — the publish job stamps the real version from the release tag
|
|
10
10
|
# at build time (see publish-pypi in .github/workflows/auto-tag.yml).
|
|
11
|
-
version = "0.
|
|
11
|
+
version = "0.7.0"
|
|
12
12
|
description = "Open benchmark harness for synthetic survey respondent quality"
|
|
13
13
|
readme = "README.md"
|
|
14
14
|
license = "MIT"
|
|
@@ -28,6 +28,24 @@ This module is the single replacement. Matching order:
|
|
|
28
28
|
|
|
29
29
|
Anything else is a parse failure: ``option is None`` and ``refusal`` is
|
|
30
30
|
False. Callers must NOT substitute a default option.
|
|
31
|
+
|
|
32
|
+
Parser versions (``config.option_parser_version``; absent key = v1):
|
|
33
|
+
|
|
34
|
+
* **v1** — the matching order above, verbatim. Kept callable so historical
|
|
35
|
+
runs re-parse bit-for-bit.
|
|
36
|
+
* **v2** (current, synthbench#352) — newer models answer in markdown prose
|
|
37
|
+
(``**(D) Depends on the situation.**`` followed by an explanation). v1
|
|
38
|
+
counted those as parse failures, and its whole-text containment could
|
|
39
|
+
mis-score them by matching an option mentioned in the explanation. v2:
|
|
40
|
+
|
|
41
|
+
- strips markdown emphasis/heading/quote markup before matching;
|
|
42
|
+
- accepts a *leading* option label (``(D) ...``, ``D. ...``, ``D) ...``)
|
|
43
|
+
as the answer, unless the text right after it exactly names a
|
|
44
|
+
different option (contradictory → parse failure);
|
|
45
|
+
- restricts containment to the first line, falling back to the whole
|
|
46
|
+
response only when exactly one option is contained in it.
|
|
47
|
+
|
|
48
|
+
Refusal detection is unchanged (it is versioned separately).
|
|
31
49
|
"""
|
|
32
50
|
|
|
33
51
|
from __future__ import annotations
|
|
@@ -47,6 +65,28 @@ _BARE_LETTER_RE = re.compile(r"^\s*\(?([A-Za-z])\)?[.):]?\s*$")
|
|
|
47
65
|
|
|
48
66
|
_WS_RE = re.compile(r"\s+")
|
|
49
67
|
|
|
68
|
+
#: Parser version stamped into run metadata (``config.option_parser_version``).
|
|
69
|
+
#: Absent key on a committed run file means v1.
|
|
70
|
+
OPTION_PARSER_VERSION = 2
|
|
71
|
+
|
|
72
|
+
# v2: markdown markup to drop before matching — emphasis runs (``**``,
|
|
73
|
+
# ``__``, ``*``), and line-leading heading / blockquote / bullet markers.
|
|
74
|
+
_MD_EMPHASIS_RE = re.compile(r"(\*{1,3}|_{2,3})")
|
|
75
|
+
_MD_LINE_PREFIX_RE = re.compile(
|
|
76
|
+
r"^[ \t]*(?:#{1,6}[ \t]+|>[ \t]*|[-+][ \t]+)", re.MULTILINE
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
# v2: leading option label, optionally after an "Answer:" style prefix.
|
|
80
|
+
# Parenthesized letters may be either case; bare letters must be uppercase
|
|
81
|
+
# and delimited (checked in _leading_label_match) so "I think..." or
|
|
82
|
+
# "A lot of people..." never parse as a label.
|
|
83
|
+
_LEADING_LABEL_RE = re.compile(
|
|
84
|
+
r"^\s*(?:(?:answer|my answer|choice)\s*[:\-]\s*)?"
|
|
85
|
+
r"(?:\(([A-Za-z])\)|([A-Za-z])[.):](?=\s|$))\s*(.*)",
|
|
86
|
+
re.IGNORECASE | re.DOTALL,
|
|
87
|
+
)
|
|
88
|
+
_PAREN_LABEL_RE = re.compile(r"\(([A-Za-z])\)")
|
|
89
|
+
|
|
50
90
|
|
|
51
91
|
@dataclass(frozen=True)
|
|
52
92
|
class ParsedResponse:
|
|
@@ -96,11 +136,91 @@ def _containment_match(text: str, options: list[str]) -> str | None:
|
|
|
96
136
|
return None
|
|
97
137
|
|
|
98
138
|
|
|
139
|
+
def _strip_markdown(text: str) -> str:
|
|
140
|
+
"""Drop markdown emphasis and line-leading markup (v2)."""
|
|
141
|
+
return _MD_EMPHASIS_RE.sub("", _MD_LINE_PREFIX_RE.sub("", text))
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _leading_label_match(text: str, options: list[str]) -> ParsedResponse | None:
|
|
145
|
+
"""Resolve a leading ``(D) ...`` / ``D. ...`` label (v2).
|
|
146
|
+
|
|
147
|
+
Returns ``None`` when there is no label; a parse failure when the label
|
|
148
|
+
is out of range or contradicted by an exact option name right after it.
|
|
149
|
+
The lowercase bare form (``a. ...``) is not a label — only ``(a)``.
|
|
150
|
+
"""
|
|
151
|
+
match = _LEADING_LABEL_RE.match(text)
|
|
152
|
+
if not match:
|
|
153
|
+
return None
|
|
154
|
+
paren, bare, rest = match.groups()
|
|
155
|
+
if bare is not None and not bare.isupper():
|
|
156
|
+
return None
|
|
157
|
+
letter = (paren or bare).upper()
|
|
158
|
+
idx = ord(letter) - ord("A")
|
|
159
|
+
if not 0 <= idx < len(options):
|
|
160
|
+
return ParsedResponse()
|
|
161
|
+
# "(B) Favor" where B is "Oppose": the label and the echoed text
|
|
162
|
+
# disagree, so neither can be trusted.
|
|
163
|
+
first_line = rest.strip().split("\n", 1)[0]
|
|
164
|
+
by_norm = {_normalize(opt): opt for opt in options}
|
|
165
|
+
echoed = by_norm.get(_normalize(first_line))
|
|
166
|
+
if echoed is not None and echoed != options[idx]:
|
|
167
|
+
return ParsedResponse()
|
|
168
|
+
return ParsedResponse(option=options[idx])
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _parse_v2_options(stripped: str, options: list[str]) -> ParsedResponse:
|
|
172
|
+
"""v2 option matching on markdown-stripped text (after refusal checks)."""
|
|
173
|
+
text = _strip_markdown(stripped).strip()
|
|
174
|
+
if not text:
|
|
175
|
+
return ParsedResponse()
|
|
176
|
+
|
|
177
|
+
by_norm = {_normalize(opt): opt for opt in options}
|
|
178
|
+
text_norm = _normalize(text)
|
|
179
|
+
if text_norm in by_norm:
|
|
180
|
+
return ParsedResponse(option=by_norm[text_norm])
|
|
181
|
+
|
|
182
|
+
match = _BARE_LETTER_RE.match(text)
|
|
183
|
+
if match:
|
|
184
|
+
idx = ord(match.group(1).upper()) - ord("A")
|
|
185
|
+
if 0 <= idx < len(options):
|
|
186
|
+
return ParsedResponse(option=options[idx])
|
|
187
|
+
return ParsedResponse()
|
|
188
|
+
|
|
189
|
+
labelled = _leading_label_match(text, options)
|
|
190
|
+
if labelled is not None:
|
|
191
|
+
return labelled
|
|
192
|
+
|
|
193
|
+
lines = [ln for ln in text.splitlines() if ln.strip()]
|
|
194
|
+
# A parenthesized label after a short lead-in ("I'd say (B) agree",
|
|
195
|
+
# possibly below a roleplay line like "*thinks*"). Only when the opening
|
|
196
|
+
# lines carry exactly one distinct label; several means a list or a
|
|
197
|
+
# comparison, not an answer.
|
|
198
|
+
lead = "\n".join(lines[:2])
|
|
199
|
+
labels = list(_PAREN_LABEL_RE.finditer(lead))
|
|
200
|
+
if labels and len({m.group(1).upper() for m in labels}) == 1:
|
|
201
|
+
inline = _leading_label_match(lead[labels[0].start() :], options)
|
|
202
|
+
if inline is not None:
|
|
203
|
+
return inline
|
|
204
|
+
|
|
205
|
+
first_line = lines[0] if lines else ""
|
|
206
|
+
contained = _containment_match(first_line, options)
|
|
207
|
+
if contained is not None:
|
|
208
|
+
return ParsedResponse(option=contained)
|
|
209
|
+
|
|
210
|
+
# Whole-response containment only when unambiguous: an explanation that
|
|
211
|
+
# mentions several options must not be scored as whichever matched first.
|
|
212
|
+
hits = [opt for opt in options if _containment_match(text, [opt]) is not None]
|
|
213
|
+
if len(hits) == 1:
|
|
214
|
+
return ParsedResponse(option=hits[0])
|
|
215
|
+
return ParsedResponse()
|
|
216
|
+
|
|
217
|
+
|
|
99
218
|
def parse_option_response(
|
|
100
219
|
text: str,
|
|
101
220
|
options: list[str],
|
|
102
221
|
*,
|
|
103
222
|
refusal_detector_version: int = REFUSAL_DETECTOR_VERSION,
|
|
223
|
+
option_parser_version: int = OPTION_PARSER_VERSION,
|
|
104
224
|
) -> ParsedResponse:
|
|
105
225
|
"""Parse a raw model response into an option selection, refusal, or failure.
|
|
106
226
|
|
|
@@ -111,6 +231,9 @@ def parse_option_response(
|
|
|
111
231
|
answer-initial anchoring + option-echo exemption) or 1 (the legacy
|
|
112
232
|
un-anchored patterns, kept callable so historical runs can be
|
|
113
233
|
reproduced bit-for-bit).
|
|
234
|
+
|
|
235
|
+
``option_parser_version`` selects option matching the same way: 2
|
|
236
|
+
(default, markdown-aware) or 1 (legacy).
|
|
114
237
|
"""
|
|
115
238
|
raw = str(text)
|
|
116
239
|
stripped = raw.strip()
|
|
@@ -134,6 +257,9 @@ def parse_option_response(
|
|
|
134
257
|
if is_refusal:
|
|
135
258
|
return ParsedResponse(refusal=True)
|
|
136
259
|
|
|
260
|
+
if option_parser_version >= 2:
|
|
261
|
+
return _parse_v2_options(stripped, options)
|
|
262
|
+
|
|
137
263
|
# 3. Bare letter, anchored full-match only.
|
|
138
264
|
match = _BARE_LETTER_RE.match(stripped)
|
|
139
265
|
if match:
|
|
@@ -22,6 +22,7 @@ from synthbench.metrics import (
|
|
|
22
22
|
extract_human_refusal_rate,
|
|
23
23
|
conditioning_fidelity,
|
|
24
24
|
)
|
|
25
|
+
from synthbench.providers._parsing import OPTION_PARSER_VERSION
|
|
25
26
|
from synthbench.providers.base import Distribution, PersonaSpec, Provider, Response
|
|
26
27
|
from synthbench.stats import bootstrap_ci, question_set_hash
|
|
27
28
|
|
|
@@ -476,6 +477,9 @@ class BenchmarkRunner:
|
|
|
476
477
|
# this run's raw responses. Absent on pre-v2 files (= v1).
|
|
477
478
|
# Never feeds build_config_id, so config_ids are stable.
|
|
478
479
|
"refusal_detector_version": REFUSAL_DETECTOR_VERSION,
|
|
480
|
+
# Same contract for option matching (synthbench#352): absent
|
|
481
|
+
# on pre-v2 files (= v1); never feeds build_config_id.
|
|
482
|
+
"option_parser_version": OPTION_PARSER_VERSION,
|
|
479
483
|
**_provider_reproducibility_hashes(self.provider),
|
|
480
484
|
},
|
|
481
485
|
elapsed_seconds=elapsed,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: synthbench-eval
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: Open benchmark harness for synthetic survey respondent quality
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
Keywords: benchmark,survey,synthetic-data,llm,evaluation,opinionsqa,globalopinionqa,gss,convergence,althing
|
|
@@ -846,6 +846,114 @@ def test_refusal_detector_version_constant():
|
|
|
846
846
|
assert REFUSAL_DETECTOR_VERSION == 2
|
|
847
847
|
|
|
848
848
|
|
|
849
|
+
# ---------------------------------------------------------------------------
|
|
850
|
+
# Option parser v2 (synthbench#352) — markdown-aware matching
|
|
851
|
+
# ---------------------------------------------------------------------------
|
|
852
|
+
|
|
853
|
+
_OPTS_ALCOHOL = [
|
|
854
|
+
"Morally acceptable",
|
|
855
|
+
"Morally unacceptable",
|
|
856
|
+
"Not a moral issue",
|
|
857
|
+
"Depends on the situation (VOL)",
|
|
858
|
+
]
|
|
859
|
+
_OPTS_CHINA = [
|
|
860
|
+
"Next 10 years",
|
|
861
|
+
"Next 20 years",
|
|
862
|
+
"Next 50 years",
|
|
863
|
+
"China will not replace U.S.",
|
|
864
|
+
]
|
|
865
|
+
_OPTS_CONFIDENCE = ["a great deal", "only some", "hardly any", "don't know"]
|
|
866
|
+
|
|
867
|
+
|
|
868
|
+
@pytest.mark.parametrize(
|
|
869
|
+
("text", "options", "expected_v2", "expected_v1"),
|
|
870
|
+
[
|
|
871
|
+
# Markdown-bolded leading label + explanation that mentions another
|
|
872
|
+
# option: v1 mis-scored via whole-text containment.
|
|
873
|
+
(
|
|
874
|
+
"**(D) Depends on the situation.**\n\nA drink with friends is fine. "
|
|
875
|
+
"For most people it is not a moral issue.",
|
|
876
|
+
_OPTS_ALCOHOL,
|
|
877
|
+
"Depends on the situation (VOL)",
|
|
878
|
+
"Not a moral issue",
|
|
879
|
+
),
|
|
880
|
+
(
|
|
881
|
+
"**(D) China will not replace the U.S.**\n\nMaybe the gap closes in the next 20 years.",
|
|
882
|
+
_OPTS_CHINA,
|
|
883
|
+
"China will not replace U.S.",
|
|
884
|
+
"Next 20 years",
|
|
885
|
+
),
|
|
886
|
+
# Label after a lead-in, below a roleplay line (from committed runs).
|
|
887
|
+
(
|
|
888
|
+
"I'd say **(B) only some confidence**.\n\nThey could do a great deal better.",
|
|
889
|
+
_OPTS_CONFIDENCE,
|
|
890
|
+
"only some",
|
|
891
|
+
"a great deal",
|
|
892
|
+
),
|
|
893
|
+
("D. Depends", _OPTS_ALCOHOL, "Depends on the situation (VOL)", None),
|
|
894
|
+
("Answer: (C)", _OPTS_ALCOHOL, "Not a moral issue", None),
|
|
895
|
+
(
|
|
896
|
+
"**Not a moral issue**",
|
|
897
|
+
_OPTS_ALCOHOL,
|
|
898
|
+
"Not a moral issue",
|
|
899
|
+
"Not a moral issue",
|
|
900
|
+
),
|
|
901
|
+
(
|
|
902
|
+
"## (A) Morally acceptable\nBecause.",
|
|
903
|
+
_OPTS_ALCOHOL,
|
|
904
|
+
"Morally acceptable",
|
|
905
|
+
"Morally acceptable",
|
|
906
|
+
),
|
|
907
|
+
# Unchanged plain forms.
|
|
908
|
+
("B", _OPTS_ALCOHOL, "Morally unacceptable", "Morally unacceptable"),
|
|
909
|
+
(
|
|
910
|
+
"(B) Morally unacceptable",
|
|
911
|
+
_OPTS_ALCOHOL,
|
|
912
|
+
"Morally unacceptable",
|
|
913
|
+
"Morally unacceptable",
|
|
914
|
+
),
|
|
915
|
+
],
|
|
916
|
+
)
|
|
917
|
+
def test_option_parser_v2_markdown_answers(text, options, expected_v2, expected_v1):
|
|
918
|
+
assert parse_option_response(text, options).option == expected_v2
|
|
919
|
+
assert (
|
|
920
|
+
parse_option_response(text, options, option_parser_version=1).option
|
|
921
|
+
== expected_v1
|
|
922
|
+
)
|
|
923
|
+
|
|
924
|
+
|
|
925
|
+
@pytest.mark.parametrize(
|
|
926
|
+
"text",
|
|
927
|
+
[
|
|
928
|
+
"(B) Morally acceptable", # label and echoed option disagree
|
|
929
|
+
"(Z) whatever", # label out of range
|
|
930
|
+
"I think A.", # not a label
|
|
931
|
+
"a. something", # lowercase bare letter is not a label
|
|
932
|
+
"Hard to say. (A) and (C) both have merit.", # several labels
|
|
933
|
+
# Explanation names two options and no answer line resolves it.
|
|
934
|
+
"Hard to say.\nSome think it is not a moral issue; others find it morally unacceptable.",
|
|
935
|
+
],
|
|
936
|
+
)
|
|
937
|
+
def test_option_parser_v2_refuses_ambiguous_answers(text):
|
|
938
|
+
assert parse_option_response(text, _OPTS_ALCOHOL) == ParsedResponse()
|
|
939
|
+
|
|
940
|
+
|
|
941
|
+
def test_option_parser_v2_unambiguous_whole_text_containment():
|
|
942
|
+
text = "Hard to say.\nIn the end it is not a moral issue for me."
|
|
943
|
+
assert parse_option_response(text, _OPTS_ALCOHOL).option == "Not a moral issue"
|
|
944
|
+
|
|
945
|
+
|
|
946
|
+
def test_option_parser_v2_keeps_refusal_detection():
|
|
947
|
+
text = "**I'd rather not answer that.**"
|
|
948
|
+
assert parse_option_response(text, _OPTS_ALCOHOL) == ParsedResponse(refusal=True)
|
|
949
|
+
|
|
950
|
+
|
|
951
|
+
def test_option_parser_version_constant():
|
|
952
|
+
from synthbench.providers._parsing import OPTION_PARSER_VERSION
|
|
953
|
+
|
|
954
|
+
assert OPTION_PARSER_VERSION == 2
|
|
955
|
+
|
|
956
|
+
|
|
849
957
|
# ---------------------------------------------------------------------------
|
|
850
958
|
# Structured elicitation (tpl=structured) — schema-forced extraction
|
|
851
959
|
# ---------------------------------------------------------------------------
|
|
@@ -214,6 +214,18 @@ def test_refusal_detector_version_stamp_never_changes_config_id():
|
|
|
214
214
|
)
|
|
215
215
|
|
|
216
216
|
|
|
217
|
+
def test_option_parser_version_stamp_never_changes_config_id():
|
|
218
|
+
"""``config.option_parser_version`` is metadata-only, like the refusal stamp."""
|
|
219
|
+
for provider in PROVIDER_FIXTURES:
|
|
220
|
+
base = _result(provider)
|
|
221
|
+
stamped = _result(provider)
|
|
222
|
+
stamped["config"]["option_parser_version"] = 2
|
|
223
|
+
assert (
|
|
224
|
+
_build_entry(base, rank=1)["config_id"]
|
|
225
|
+
== _build_entry(stamped, rank=1)["config_id"]
|
|
226
|
+
), provider
|
|
227
|
+
|
|
228
|
+
|
|
217
229
|
def test_all_committed_config_ids_unchanged_by_detector_stamp():
|
|
218
230
|
"""Recompute every committed run's config_id with and without the stamp.
|
|
219
231
|
|
|
@@ -698,3 +698,13 @@ async def test_runner_refuses_empty_dataset(mock_provider):
|
|
|
698
698
|
)
|
|
699
699
|
with pytest.raises(EmptyQuestionSetError, match="loaded 0 questions"):
|
|
700
700
|
await runner.run()
|
|
701
|
+
|
|
702
|
+
|
|
703
|
+
async def test_runner_stamps_option_parser_version(mock_dataset, mock_provider):
|
|
704
|
+
from synthbench.providers._parsing import OPTION_PARSER_VERSION
|
|
705
|
+
|
|
706
|
+
runner = BenchmarkRunner(
|
|
707
|
+
dataset=mock_dataset, provider=mock_provider, samples_per_question=2
|
|
708
|
+
)
|
|
709
|
+
result = await runner.run(n=1)
|
|
710
|
+
assert result.config["option_parser_version"] == OPTION_PARSER_VERSION
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/majority_baseline.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench/providers/population_baseline.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{synthbench_eval-0.6.2 → synthbench_eval-0.7.0}/src/synthbench_eval.egg-info/entry_points.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|