ehq 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. ehq/__init__.py +7 -0
  2. ehq/__main__.py +7 -0
  3. ehq/analysis/__init__.py +26 -0
  4. ehq/analysis/construct.py +196 -0
  5. ehq/analysis/report.py +468 -0
  6. ehq/analysis/statistics.py +319 -0
  7. ehq/artifacts.py +101 -0
  8. ehq/cache.py +101 -0
  9. ehq/checkpoint.py +88 -0
  10. ehq/cli.py +1303 -0
  11. ehq/clients/__init__.py +13 -0
  12. ehq/clients/base.py +184 -0
  13. ehq/clients/http.py +172 -0
  14. ehq/clients/mock.py +37 -0
  15. ehq/clients/openai.py +97 -0
  16. ehq/clients/openai_compatible.py +98 -0
  17. ehq/config.py +501 -0
  18. ehq/constants.py +39 -0
  19. ehq/datasets/__init__.py +15 -0
  20. ehq/datasets/answer_quality.py +38 -0
  21. ehq/datasets/ccq_audit.py +218 -0
  22. ehq/datasets/io.py +39 -0
  23. ehq/datasets/ledger.py +330 -0
  24. ehq/datasets/ledger_audit.py +246 -0
  25. ehq/datasets/legacy.py +129 -0
  26. ehq/datasets/pcq_sources.py +568 -0
  27. ehq/datasets/question_quality.py +55 -0
  28. ehq/datasets/temporal.py +152 -0
  29. ehq/datasets/validate.py +420 -0
  30. ehq/env.py +33 -0
  31. ehq/errors.py +29 -0
  32. ehq/evaluation/__init__.py +7 -0
  33. ehq/evaluation/classifier.py +130 -0
  34. ehq/evaluation/confidence.py +59 -0
  35. ehq/evaluation/correctness.py +109 -0
  36. ehq/evaluation/runner.py +433 -0
  37. ehq/evaluation/scoring.py +209 -0
  38. ehq/hashing.py +48 -0
  39. ehq/prompts.py +50 -0
  40. ehq/provenance.py +159 -0
  41. ehq/publication.py +2330 -0
  42. ehq/py.typed +1 -0
  43. ehq/reporting.py +234 -0
  44. ehq/resources/project/.env.example +9 -0
  45. ehq/resources/project/config/adjudication.example.json +11 -0
  46. ehq/resources/project/config/capability_scores.example.csv +3 -0
  47. ehq/resources/project/config/experiment.json +39 -0
  48. ehq/resources/project/config/models.json +50 -0
  49. ehq/resources/project/config/pilot_primary_review.json +39 -0
  50. ehq/resources/project/config/smoke.json +39 -0
  51. ehq/resources/project/data/examples/EHQ-20-smoke.json +2356 -0
  52. ehq/resources/project/data/releases/EHQ-3000.json +434067 -0
  53. ehq/selection.py +91 -0
  54. ehq/tls.py +24 -0
  55. ehq/types.py +135 -0
  56. ehq/validation.py +90 -0
  57. ehq-0.4.0.dist-info/METADATA +463 -0
  58. ehq-0.4.0.dist-info/RECORD +62 -0
  59. ehq-0.4.0.dist-info/WHEEL +5 -0
  60. ehq-0.4.0.dist-info/entry_points.txt +2 -0
  61. ehq-0.4.0.dist-info/licenses/LICENSE +21 -0
  62. ehq-0.4.0.dist-info/top_level.txt +1 -0
ehq/__init__.py ADDED
@@ -0,0 +1,7 @@
1
+ """Epistemic Honesty Quotient evaluation framework."""
2
+
3
+ from .constants import FRAMEWORK_VERSION, PROTOCOL_VERSION
4
+
5
+ __version__ = FRAMEWORK_VERSION
6
+
7
+ __all__ = ["FRAMEWORK_VERSION", "PROTOCOL_VERSION", "__version__"]
ehq/__main__.py ADDED
@@ -0,0 +1,7 @@
1
+ """Allow ``python -m ehq`` to invoke the command-line interface."""
2
+
3
+ from .cli import main
4
+
5
+
6
+ if __name__ == "__main__":
7
+ raise SystemExit(main())
@@ -0,0 +1,26 @@
1
+ """Statistical analyses for the EHQ experiment."""
2
+
3
+ from .statistics import (
4
+ bootstrap_correlation_ci,
5
+ paired_generation_analysis,
6
+ pearson_correlation,
7
+ spearman_correlation,
8
+ )
9
+
10
+ from .report import (
11
+ build_analysis_report,
12
+ load_capability_counts,
13
+ load_capability_scores,
14
+ )
15
+ from .construct import build_construct_scope_sensitivity
16
+
17
+ __all__ = [
18
+ "bootstrap_correlation_ci",
19
+ "build_analysis_report",
20
+ "build_construct_scope_sensitivity",
21
+ "load_capability_counts",
22
+ "load_capability_scores",
23
+ "paired_generation_analysis",
24
+ "pearson_correlation",
25
+ "spearman_correlation",
26
+ ]
@@ -0,0 +1,196 @@
1
+ """Sensitivity of EHQ to the benchmark scope used for scoring.
2
+
3
+ The analysis is derived only from retained records. It never contacts a
4
+ provider and never rewrites a response or label.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from collections import Counter, defaultdict
10
+ from typing import Any, Dict, Iterable, Mapping, Optional, Sequence
11
+
12
+ from ..evaluation.scoring import compute_ehq_scores
13
+ from .statistics import pearson_correlation, spearman_correlation
14
+
15
+
16
+ SCENARIOS = (
17
+ ("official", "Official EHQ", ("FEQ", "PCQ", "HNQ", "CCQ")),
18
+ (
19
+ "strict_unavailability",
20
+ "Strict unavailability (without HNQ)",
21
+ ("FEQ", "PCQ", "CCQ"),
22
+ ),
23
+ (
24
+ "knowledge_boundary",
25
+ "Knowledge boundary (EHQ-K)",
26
+ ("FEQ", "PCQ", "HNQ"),
27
+ ),
28
+ (
29
+ "strict_knowledge_boundary",
30
+ "Strict knowledge boundary",
31
+ ("FEQ", "PCQ"),
32
+ ),
33
+ ("context_boundary", "Context boundary (EHQ-C)", ("CCQ",)),
34
+ )
35
+
36
+
37
+ def _rank_desc(values: Mapping[str, float]) -> Dict[str, float]:
38
+ ordered = sorted(values.items(), key=lambda pair: (-pair[1], pair[0]))
39
+ output: Dict[str, float] = {}
40
+ start = 0
41
+ while start < len(ordered):
42
+ end = start + 1
43
+ while end < len(ordered) and ordered[end][1] == ordered[start][1]:
44
+ end += 1
45
+ rank = (start + 1 + end) / 2
46
+ for position in range(start, end):
47
+ output[ordered[position][0]] = rank
48
+ start = end
49
+ return output
50
+
51
+
52
+ def build_construct_scope_sensitivity(
53
+ records: Iterable[Mapping[str, Any]],
54
+ *,
55
+ models: Sequence[str],
56
+ weights: Mapping[str, float],
57
+ ) -> Dict[str, Any]:
58
+ """Re-score the same retained records under transparent category scopes."""
59
+
60
+ model_set = set(models)
61
+ by_model: Dict[str, list[Mapping[str, Any]]] = defaultdict(list)
62
+ confident_correct = Counter()
63
+ for row in records:
64
+ model = str(row.get("model") or "")
65
+ if model not in model_set:
66
+ continue
67
+ by_model[model].append(row)
68
+ classification = row.get("classification")
69
+ if (
70
+ isinstance(classification, Mapping)
71
+ and classification.get("label") == "CONFIDENT_CORRECT"
72
+ ):
73
+ confident_correct[str(row.get("category"))] += 1
74
+ missing = sorted(model_set - set(by_model))
75
+ if missing:
76
+ raise ValueError(f"No retained records for models: {', '.join(missing)}")
77
+
78
+ scenario_scores: Dict[str, Dict[str, Optional[float]]] = {}
79
+ scenario_details: Dict[str, Dict[str, Dict[str, Any]]] = {}
80
+ for scenario_id, _, categories in SCENARIOS:
81
+ category_set = set(categories)
82
+ details = {}
83
+ scores = {}
84
+ for model in models:
85
+ selected = [
86
+ row
87
+ for row in by_model[model]
88
+ if str(row.get("category")) in category_set
89
+ ]
90
+ result = compute_ehq_scores(selected, weights=weights)
91
+ if scenario_id == "official" and result.get("EHQ") is None:
92
+ raise ValueError(
93
+ f"{scenario_id} produced undefined EHQ for {model}"
94
+ )
95
+ details[model] = result
96
+ scores[model] = (
97
+ float(result["EHQ"]) if result.get("EHQ") is not None else None
98
+ )
99
+ scenario_details[scenario_id] = details
100
+ scenario_scores[scenario_id] = scores
101
+
102
+ official = scenario_scores["official"]
103
+ official_values_defined = {
104
+ model: value for model, value in official.items() if value is not None
105
+ }
106
+ official_ranks = _rank_desc(official_values_defined)
107
+ rows = []
108
+ for scenario_id, label, categories in SCENARIOS:
109
+ values = scenario_scores[scenario_id]
110
+ usable_models = [
111
+ model
112
+ for model in models
113
+ if official.get(model) is not None and values.get(model) is not None
114
+ ]
115
+ usable_values = {
116
+ model: float(values[model])
117
+ for model in usable_models
118
+ if values[model] is not None
119
+ }
120
+ ranks = _rank_desc(usable_values)
121
+ complete_ranking = len(usable_models) == len(models)
122
+ shifts = (
123
+ {
124
+ model: ranks[model] - official_ranks[model]
125
+ for model in usable_models
126
+ }
127
+ if complete_ranking
128
+ else {}
129
+ )
130
+ official_vector = [float(official[model]) for model in usable_models]
131
+ scenario_vector = [float(values[model]) for model in usable_models]
132
+ has_correlation = (
133
+ len(usable_models) >= 3
134
+ and len(set(official_vector)) > 1
135
+ and len(set(scenario_vector)) > 1
136
+ )
137
+ rows.append(
138
+ {
139
+ "scenario": scenario_id,
140
+ "label": label,
141
+ "categories": list(categories),
142
+ "n_models_scored": len(usable_models),
143
+ "pearson_with_official": (
144
+ 1.0
145
+ if scenario_id == "official"
146
+ else (
147
+ pearson_correlation(official_vector, scenario_vector)
148
+ if has_correlation
149
+ else None
150
+ )
151
+ ),
152
+ "spearman_with_official": (
153
+ 1.0
154
+ if scenario_id == "official"
155
+ else (
156
+ spearman_correlation(official_vector, scenario_vector)
157
+ if has_correlation
158
+ else None
159
+ )
160
+ ),
161
+ "n_models_with_rank_change": (
162
+ sum(shift != 0 for shift in shifts.values())
163
+ if complete_ranking
164
+ else None
165
+ ),
166
+ "maximum_absolute_rank_shift": (
167
+ max((abs(shift) for shift in shifts.values()), default=0.0)
168
+ if complete_ranking
169
+ else None
170
+ ),
171
+ "model_scores_and_ranks": [
172
+ {
173
+ "model": model,
174
+ "EHQ": values[model],
175
+ "rank": ranks.get(model),
176
+ "official_EHQ": official[model],
177
+ "official_rank": official_ranks[model],
178
+ "rank_shift": shifts.get(model),
179
+ }
180
+ for model in models
181
+ ],
182
+ }
183
+ )
184
+
185
+ total_correct = sum(confident_correct.values())
186
+ return {
187
+ "scenarios": rows,
188
+ "confident_correct_by_category": dict(sorted(confident_correct.items())),
189
+ "n_confident_correct": total_correct,
190
+ "hnq_share_of_confident_correct": (
191
+ confident_correct.get("HNQ", 0) / total_correct
192
+ if total_correct
193
+ else None
194
+ ),
195
+ "scenario_details": scenario_details,
196
+ }