ehq 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ehq/__init__.py +7 -0
- ehq/__main__.py +7 -0
- ehq/analysis/__init__.py +26 -0
- ehq/analysis/construct.py +196 -0
- ehq/analysis/report.py +468 -0
- ehq/analysis/statistics.py +319 -0
- ehq/artifacts.py +101 -0
- ehq/cache.py +101 -0
- ehq/checkpoint.py +88 -0
- ehq/cli.py +1303 -0
- ehq/clients/__init__.py +13 -0
- ehq/clients/base.py +184 -0
- ehq/clients/http.py +172 -0
- ehq/clients/mock.py +37 -0
- ehq/clients/openai.py +97 -0
- ehq/clients/openai_compatible.py +98 -0
- ehq/config.py +501 -0
- ehq/constants.py +39 -0
- ehq/datasets/__init__.py +15 -0
- ehq/datasets/answer_quality.py +38 -0
- ehq/datasets/ccq_audit.py +218 -0
- ehq/datasets/io.py +39 -0
- ehq/datasets/ledger.py +330 -0
- ehq/datasets/ledger_audit.py +246 -0
- ehq/datasets/legacy.py +129 -0
- ehq/datasets/pcq_sources.py +568 -0
- ehq/datasets/question_quality.py +55 -0
- ehq/datasets/temporal.py +152 -0
- ehq/datasets/validate.py +420 -0
- ehq/env.py +33 -0
- ehq/errors.py +29 -0
- ehq/evaluation/__init__.py +7 -0
- ehq/evaluation/classifier.py +130 -0
- ehq/evaluation/confidence.py +59 -0
- ehq/evaluation/correctness.py +109 -0
- ehq/evaluation/runner.py +433 -0
- ehq/evaluation/scoring.py +209 -0
- ehq/hashing.py +48 -0
- ehq/prompts.py +50 -0
- ehq/provenance.py +159 -0
- ehq/publication.py +2330 -0
- ehq/py.typed +1 -0
- ehq/reporting.py +234 -0
- ehq/resources/project/.env.example +9 -0
- ehq/resources/project/config/adjudication.example.json +11 -0
- ehq/resources/project/config/capability_scores.example.csv +3 -0
- ehq/resources/project/config/experiment.json +39 -0
- ehq/resources/project/config/models.json +50 -0
- ehq/resources/project/config/pilot_primary_review.json +39 -0
- ehq/resources/project/config/smoke.json +39 -0
- ehq/resources/project/data/examples/EHQ-20-smoke.json +2356 -0
- ehq/resources/project/data/releases/EHQ-3000.json +434067 -0
- ehq/selection.py +91 -0
- ehq/tls.py +24 -0
- ehq/types.py +135 -0
- ehq/validation.py +90 -0
- ehq-0.4.0.dist-info/METADATA +463 -0
- ehq-0.4.0.dist-info/RECORD +62 -0
- ehq-0.4.0.dist-info/WHEEL +5 -0
- ehq-0.4.0.dist-info/entry_points.txt +2 -0
- ehq-0.4.0.dist-info/licenses/LICENSE +21 -0
- ehq-0.4.0.dist-info/top_level.txt +1 -0
ehq/__init__.py
ADDED
ehq/__main__.py
ADDED
ehq/analysis/__init__.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Statistical analyses for the EHQ experiment."""
|
|
2
|
+
|
|
3
|
+
from .statistics import (
|
|
4
|
+
bootstrap_correlation_ci,
|
|
5
|
+
paired_generation_analysis,
|
|
6
|
+
pearson_correlation,
|
|
7
|
+
spearman_correlation,
|
|
8
|
+
)
|
|
9
|
+
|
|
10
|
+
from .report import (
|
|
11
|
+
build_analysis_report,
|
|
12
|
+
load_capability_counts,
|
|
13
|
+
load_capability_scores,
|
|
14
|
+
)
|
|
15
|
+
from .construct import build_construct_scope_sensitivity
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"bootstrap_correlation_ci",
|
|
19
|
+
"build_analysis_report",
|
|
20
|
+
"build_construct_scope_sensitivity",
|
|
21
|
+
"load_capability_counts",
|
|
22
|
+
"load_capability_scores",
|
|
23
|
+
"paired_generation_analysis",
|
|
24
|
+
"pearson_correlation",
|
|
25
|
+
"spearman_correlation",
|
|
26
|
+
]
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
"""Sensitivity of EHQ to the benchmark scope used for scoring.
|
|
2
|
+
|
|
3
|
+
The analysis is derived only from retained records. It never contacts a
|
|
4
|
+
provider and never rewrites a response or label.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from collections import Counter, defaultdict
|
|
10
|
+
from typing import Any, Dict, Iterable, Mapping, Optional, Sequence
|
|
11
|
+
|
|
12
|
+
from ..evaluation.scoring import compute_ehq_scores
|
|
13
|
+
from .statistics import pearson_correlation, spearman_correlation
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
SCENARIOS = (
|
|
17
|
+
("official", "Official EHQ", ("FEQ", "PCQ", "HNQ", "CCQ")),
|
|
18
|
+
(
|
|
19
|
+
"strict_unavailability",
|
|
20
|
+
"Strict unavailability (without HNQ)",
|
|
21
|
+
("FEQ", "PCQ", "CCQ"),
|
|
22
|
+
),
|
|
23
|
+
(
|
|
24
|
+
"knowledge_boundary",
|
|
25
|
+
"Knowledge boundary (EHQ-K)",
|
|
26
|
+
("FEQ", "PCQ", "HNQ"),
|
|
27
|
+
),
|
|
28
|
+
(
|
|
29
|
+
"strict_knowledge_boundary",
|
|
30
|
+
"Strict knowledge boundary",
|
|
31
|
+
("FEQ", "PCQ"),
|
|
32
|
+
),
|
|
33
|
+
("context_boundary", "Context boundary (EHQ-C)", ("CCQ",)),
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _rank_desc(values: Mapping[str, float]) -> Dict[str, float]:
|
|
38
|
+
ordered = sorted(values.items(), key=lambda pair: (-pair[1], pair[0]))
|
|
39
|
+
output: Dict[str, float] = {}
|
|
40
|
+
start = 0
|
|
41
|
+
while start < len(ordered):
|
|
42
|
+
end = start + 1
|
|
43
|
+
while end < len(ordered) and ordered[end][1] == ordered[start][1]:
|
|
44
|
+
end += 1
|
|
45
|
+
rank = (start + 1 + end) / 2
|
|
46
|
+
for position in range(start, end):
|
|
47
|
+
output[ordered[position][0]] = rank
|
|
48
|
+
start = end
|
|
49
|
+
return output
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def build_construct_scope_sensitivity(
|
|
53
|
+
records: Iterable[Mapping[str, Any]],
|
|
54
|
+
*,
|
|
55
|
+
models: Sequence[str],
|
|
56
|
+
weights: Mapping[str, float],
|
|
57
|
+
) -> Dict[str, Any]:
|
|
58
|
+
"""Re-score the same retained records under transparent category scopes."""
|
|
59
|
+
|
|
60
|
+
model_set = set(models)
|
|
61
|
+
by_model: Dict[str, list[Mapping[str, Any]]] = defaultdict(list)
|
|
62
|
+
confident_correct = Counter()
|
|
63
|
+
for row in records:
|
|
64
|
+
model = str(row.get("model") or "")
|
|
65
|
+
if model not in model_set:
|
|
66
|
+
continue
|
|
67
|
+
by_model[model].append(row)
|
|
68
|
+
classification = row.get("classification")
|
|
69
|
+
if (
|
|
70
|
+
isinstance(classification, Mapping)
|
|
71
|
+
and classification.get("label") == "CONFIDENT_CORRECT"
|
|
72
|
+
):
|
|
73
|
+
confident_correct[str(row.get("category"))] += 1
|
|
74
|
+
missing = sorted(model_set - set(by_model))
|
|
75
|
+
if missing:
|
|
76
|
+
raise ValueError(f"No retained records for models: {', '.join(missing)}")
|
|
77
|
+
|
|
78
|
+
scenario_scores: Dict[str, Dict[str, Optional[float]]] = {}
|
|
79
|
+
scenario_details: Dict[str, Dict[str, Dict[str, Any]]] = {}
|
|
80
|
+
for scenario_id, _, categories in SCENARIOS:
|
|
81
|
+
category_set = set(categories)
|
|
82
|
+
details = {}
|
|
83
|
+
scores = {}
|
|
84
|
+
for model in models:
|
|
85
|
+
selected = [
|
|
86
|
+
row
|
|
87
|
+
for row in by_model[model]
|
|
88
|
+
if str(row.get("category")) in category_set
|
|
89
|
+
]
|
|
90
|
+
result = compute_ehq_scores(selected, weights=weights)
|
|
91
|
+
if scenario_id == "official" and result.get("EHQ") is None:
|
|
92
|
+
raise ValueError(
|
|
93
|
+
f"{scenario_id} produced undefined EHQ for {model}"
|
|
94
|
+
)
|
|
95
|
+
details[model] = result
|
|
96
|
+
scores[model] = (
|
|
97
|
+
float(result["EHQ"]) if result.get("EHQ") is not None else None
|
|
98
|
+
)
|
|
99
|
+
scenario_details[scenario_id] = details
|
|
100
|
+
scenario_scores[scenario_id] = scores
|
|
101
|
+
|
|
102
|
+
official = scenario_scores["official"]
|
|
103
|
+
official_values_defined = {
|
|
104
|
+
model: value for model, value in official.items() if value is not None
|
|
105
|
+
}
|
|
106
|
+
official_ranks = _rank_desc(official_values_defined)
|
|
107
|
+
rows = []
|
|
108
|
+
for scenario_id, label, categories in SCENARIOS:
|
|
109
|
+
values = scenario_scores[scenario_id]
|
|
110
|
+
usable_models = [
|
|
111
|
+
model
|
|
112
|
+
for model in models
|
|
113
|
+
if official.get(model) is not None and values.get(model) is not None
|
|
114
|
+
]
|
|
115
|
+
usable_values = {
|
|
116
|
+
model: float(values[model])
|
|
117
|
+
for model in usable_models
|
|
118
|
+
if values[model] is not None
|
|
119
|
+
}
|
|
120
|
+
ranks = _rank_desc(usable_values)
|
|
121
|
+
complete_ranking = len(usable_models) == len(models)
|
|
122
|
+
shifts = (
|
|
123
|
+
{
|
|
124
|
+
model: ranks[model] - official_ranks[model]
|
|
125
|
+
for model in usable_models
|
|
126
|
+
}
|
|
127
|
+
if complete_ranking
|
|
128
|
+
else {}
|
|
129
|
+
)
|
|
130
|
+
official_vector = [float(official[model]) for model in usable_models]
|
|
131
|
+
scenario_vector = [float(values[model]) for model in usable_models]
|
|
132
|
+
has_correlation = (
|
|
133
|
+
len(usable_models) >= 3
|
|
134
|
+
and len(set(official_vector)) > 1
|
|
135
|
+
and len(set(scenario_vector)) > 1
|
|
136
|
+
)
|
|
137
|
+
rows.append(
|
|
138
|
+
{
|
|
139
|
+
"scenario": scenario_id,
|
|
140
|
+
"label": label,
|
|
141
|
+
"categories": list(categories),
|
|
142
|
+
"n_models_scored": len(usable_models),
|
|
143
|
+
"pearson_with_official": (
|
|
144
|
+
1.0
|
|
145
|
+
if scenario_id == "official"
|
|
146
|
+
else (
|
|
147
|
+
pearson_correlation(official_vector, scenario_vector)
|
|
148
|
+
if has_correlation
|
|
149
|
+
else None
|
|
150
|
+
)
|
|
151
|
+
),
|
|
152
|
+
"spearman_with_official": (
|
|
153
|
+
1.0
|
|
154
|
+
if scenario_id == "official"
|
|
155
|
+
else (
|
|
156
|
+
spearman_correlation(official_vector, scenario_vector)
|
|
157
|
+
if has_correlation
|
|
158
|
+
else None
|
|
159
|
+
)
|
|
160
|
+
),
|
|
161
|
+
"n_models_with_rank_change": (
|
|
162
|
+
sum(shift != 0 for shift in shifts.values())
|
|
163
|
+
if complete_ranking
|
|
164
|
+
else None
|
|
165
|
+
),
|
|
166
|
+
"maximum_absolute_rank_shift": (
|
|
167
|
+
max((abs(shift) for shift in shifts.values()), default=0.0)
|
|
168
|
+
if complete_ranking
|
|
169
|
+
else None
|
|
170
|
+
),
|
|
171
|
+
"model_scores_and_ranks": [
|
|
172
|
+
{
|
|
173
|
+
"model": model,
|
|
174
|
+
"EHQ": values[model],
|
|
175
|
+
"rank": ranks.get(model),
|
|
176
|
+
"official_EHQ": official[model],
|
|
177
|
+
"official_rank": official_ranks[model],
|
|
178
|
+
"rank_shift": shifts.get(model),
|
|
179
|
+
}
|
|
180
|
+
for model in models
|
|
181
|
+
],
|
|
182
|
+
}
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
total_correct = sum(confident_correct.values())
|
|
186
|
+
return {
|
|
187
|
+
"scenarios": rows,
|
|
188
|
+
"confident_correct_by_category": dict(sorted(confident_correct.items())),
|
|
189
|
+
"n_confident_correct": total_correct,
|
|
190
|
+
"hnq_share_of_confident_correct": (
|
|
191
|
+
confident_correct.get("HNQ", 0) / total_correct
|
|
192
|
+
if total_correct
|
|
193
|
+
else None
|
|
194
|
+
),
|
|
195
|
+
"scenario_details": scenario_details,
|
|
196
|
+
}
|