evalstats 0.2.4__tar.gz → 0.2.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {evalstats-0.2.4 → evalstats-0.2.5}/PKG-INFO +1 -1
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/__init__.py +1 -1
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/alignment.py +328 -19
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/api.py +38 -10
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/config.py +18 -19
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/resampling.py +24 -11
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/ppi.py +609 -52
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/tests/__init__.py +1118 -107
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/PKG-INFO +1 -1
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/SOURCES.txt +1 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/pyproject.toml +1 -1
- evalstats-0.2.5/tests/test_ppi_ci_methods.py +306 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_ppi_corrections.py +25 -1
- {evalstats-0.2.4 → evalstats-0.2.5}/LICENSE +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/README.md +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/cli.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/__init__.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/bayes_evals.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/bundles.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/mixed_effects.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/paired.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/ranking.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/router.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/stats_utils.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/summary.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/types.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/variance.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/io.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/loader.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/__init__.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/critical_difference.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/forest.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/heatmap.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/point_estimates.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/scoreboard.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/dependency_links.txt +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/entry_points.txt +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/requires.txt +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/top_level.txt +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/setup.cfg +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_alignment.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_analyze.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_analyze_factorial.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_auto_ci_routing.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_bayes_binary_routing.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_bootstrap_t_pairwise_ranking.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_cli.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_compare.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_critical_difference_plot.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_io.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_lmm.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_lmm_backend_parity.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_lmm_statsmodels.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_nig_ci_methods.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_p_values.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_permutation.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_ppi_core.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_resampling.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_scoreboard_plot.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_set_alpha_ci.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_simultaneous_ci.py +0 -0
- {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_wilson_newcombe.py +0 -0
|
@@ -43,6 +43,12 @@ class AlignmentResult:
|
|
|
43
43
|
Point estimates and bootstrap CIs for each alignment metric.
|
|
44
44
|
representativeness : dict
|
|
45
45
|
Representativeness check results (distribution and slice columns).
|
|
46
|
+
bias_check : dict or None
|
|
47
|
+
For likert/continuous/grade score types, compares the correlation-type
|
|
48
|
+
metric (weighted κ or Pearson r) against ICC(2,1) to flag whether the
|
|
49
|
+
judge is systematically biased in absolute scale despite tracking
|
|
50
|
+
human relative ordering. ``None`` for binary score types, where ICC
|
|
51
|
+
isn't computed.
|
|
46
52
|
"""
|
|
47
53
|
|
|
48
54
|
def __init__(
|
|
@@ -56,6 +62,7 @@ class AlignmentResult:
|
|
|
56
62
|
calibration: dict,
|
|
57
63
|
alignment_metrics: dict,
|
|
58
64
|
representativeness: dict,
|
|
65
|
+
bias_check: Optional[dict] = None,
|
|
59
66
|
) -> None:
|
|
60
67
|
self.llm_metric = llm_metric
|
|
61
68
|
self.human_col = human_col
|
|
@@ -65,6 +72,7 @@ class AlignmentResult:
|
|
|
65
72
|
self._calibration = calibration
|
|
66
73
|
self.alignment_metrics = alignment_metrics
|
|
67
74
|
self.representativeness = representativeness
|
|
75
|
+
self.bias_check = bias_check
|
|
68
76
|
|
|
69
77
|
# ── sampling ─────────────────────────────────────────────────────────────
|
|
70
78
|
|
|
@@ -130,18 +138,97 @@ class AlignmentResult:
|
|
|
130
138
|
|
|
131
139
|
# ── display ──────────────────────────────────────────────────────────────
|
|
132
140
|
|
|
133
|
-
def summary(self) -> None:
|
|
134
|
-
"""Print
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
141
|
+
def summary(self, verbose: bool = False) -> None:
|
|
142
|
+
"""Print an alignment and representativeness report.
|
|
143
|
+
|
|
144
|
+
Parameters
|
|
145
|
+
----------
|
|
146
|
+
verbose : bool
|
|
147
|
+
If ``False`` (default), print a short, plain-language report:
|
|
148
|
+
one line per check/metric, with an explanation only where
|
|
149
|
+
something looks off. Aimed at readers who don't need the
|
|
150
|
+
statistical background spelled out every time.
|
|
151
|
+
If ``True``, print the full report — every metric's definition,
|
|
152
|
+
why it was chosen, how to interpret it, and citation-ready
|
|
153
|
+
wording for a paper.
|
|
154
|
+
"""
|
|
155
|
+
if verbose:
|
|
156
|
+
self._summary_verbose()
|
|
157
|
+
else:
|
|
158
|
+
self._summary_simple()
|
|
159
|
+
|
|
160
|
+
def _header(self) -> None:
|
|
138
161
|
pct = 100.0 * self.n_labeled / self.n_total if self.n_total > 0 else 0.0
|
|
162
|
+
print("Judge alignment report")
|
|
163
|
+
print("─" * 58)
|
|
139
164
|
print(
|
|
140
165
|
f"Alignment set : {self.n_labeled} of {self.n_total} items "
|
|
141
166
|
f"have human labels ({pct:.1f}%)"
|
|
142
167
|
)
|
|
143
168
|
print()
|
|
144
169
|
|
|
170
|
+
def _summary_simple(self) -> None:
|
|
171
|
+
self._header()
|
|
172
|
+
|
|
173
|
+
if self.bias_check is not None:
|
|
174
|
+
bc = self.bias_check
|
|
175
|
+
if not bc["passed"]:
|
|
176
|
+
print(
|
|
177
|
+
f"⚠ Possible judge bias: {bc['corr_label']} = "
|
|
178
|
+
f"{bc['corr_estimate']:.2f} but ICC(2,1) = {bc['icc_estimate']:.2f} "
|
|
179
|
+
"— the judge ranks items like humans do, but its raw scores "
|
|
180
|
+
"look shifted or compressed relative to human scores."
|
|
181
|
+
)
|
|
182
|
+
print(
|
|
183
|
+
" Treat raw judge scores with caution; consider "
|
|
184
|
+
"recalibrating (compare(alignment=...)) before using them "
|
|
185
|
+
"directly. Run .summary(verbose=True) for the full check."
|
|
186
|
+
)
|
|
187
|
+
else:
|
|
188
|
+
print(
|
|
189
|
+
f"✓ No sign of judge bias: {bc['corr_label']} and ICC(2,1) "
|
|
190
|
+
"roughly agree."
|
|
191
|
+
)
|
|
192
|
+
print()
|
|
193
|
+
|
|
194
|
+
rep = self.representativeness
|
|
195
|
+
rep_failed = [
|
|
196
|
+
(k, v) for k, v in rep.items() if not v["passed"]
|
|
197
|
+
]
|
|
198
|
+
if rep_failed:
|
|
199
|
+
print("⚠ Representativeness: the labeled sample may not be representative")
|
|
200
|
+
for key, val in rep_failed:
|
|
201
|
+
name = "score distribution" if key == "score_distribution" else key[len("slice_"):]
|
|
202
|
+
print(f" - {name}: {val['message']}")
|
|
203
|
+
else:
|
|
204
|
+
print("✓ Representativeness: labeled items look like the full item pool")
|
|
205
|
+
print()
|
|
206
|
+
|
|
207
|
+
score_type_note = _SCORE_TYPE_NOTES.get(
|
|
208
|
+
self.score_type, f"score type detected as {self.score_type!r}"
|
|
209
|
+
)
|
|
210
|
+
print(f"Alignment metrics ({self.score_type} scores — {score_type_note}):")
|
|
211
|
+
for entry in self.alignment_metrics.values():
|
|
212
|
+
label = entry.get("label", "")
|
|
213
|
+
est = entry["estimate"]
|
|
214
|
+
lo = entry["ci_low"]
|
|
215
|
+
hi = entry["ci_high"]
|
|
216
|
+
band = entry.get("band")
|
|
217
|
+
tail = f" {band}" if band else ""
|
|
218
|
+
print(f" {label:<20} {est:6.2f} [{lo:5.2f}, {hi:5.2f}]{tail}")
|
|
219
|
+
print()
|
|
220
|
+
print("Run .summary(verbose=True) for definitions, rationale, and")
|
|
221
|
+
print("citation-ready wording for each check above.")
|
|
222
|
+
print("─" * 58)
|
|
223
|
+
|
|
224
|
+
def _summary_verbose(self) -> None:
|
|
225
|
+
self._header()
|
|
226
|
+
|
|
227
|
+
if self.bias_check is not None and not self.bias_check["passed"]:
|
|
228
|
+
print(f"⚠ Judge bias flag: {self.bias_check['message']}")
|
|
229
|
+
print(" (see 'Bias diagnostics' below for details)")
|
|
230
|
+
print()
|
|
231
|
+
|
|
145
232
|
# Representativeness
|
|
146
233
|
rep = self.representativeness
|
|
147
234
|
|
|
@@ -196,7 +283,12 @@ class AlignmentResult:
|
|
|
196
283
|
if example:
|
|
197
284
|
print(f" -> Example paper reporting: {example}")
|
|
198
285
|
print()
|
|
199
|
-
|
|
286
|
+
|
|
287
|
+
if self.bias_check is not None:
|
|
288
|
+
print("Bias diagnostics:")
|
|
289
|
+
_print_check("Judge scale bias (correlation vs. ICC)", self.bias_check)
|
|
290
|
+
|
|
291
|
+
print("─" * 58)
|
|
200
292
|
|
|
201
293
|
def __repr__(self) -> str:
|
|
202
294
|
return (
|
|
@@ -299,7 +391,7 @@ _SCORE_TYPE_NOTES = {
|
|
|
299
391
|
}
|
|
300
392
|
|
|
301
393
|
|
|
302
|
-
def _interpret_kappa(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str]:
|
|
394
|
+
def _interpret_kappa(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str, str]:
|
|
303
395
|
"""Landis & Koch (1977) benchmarks for kappa-type statistics."""
|
|
304
396
|
if est < 0:
|
|
305
397
|
band = "poor"
|
|
@@ -313,16 +405,17 @@ def _interpret_kappa(est: float, lo: float, hi: float, n: int, label: str) -> tu
|
|
|
313
405
|
band = "substantial"
|
|
314
406
|
else:
|
|
315
407
|
band = "almost perfect"
|
|
408
|
+
band_phrase = f"{band} agreement"
|
|
316
409
|
interpretation = f"{band} agreement (Landis & Koch, 1977 benchmarks)"
|
|
317
410
|
example = (
|
|
318
411
|
f'"{label} = {est:.2f}, 95% CI [{lo:.2f}, {hi:.2f}] (n={n}), indicating '
|
|
319
412
|
f'{band} agreement between the LLM judge and human raters, per the Landis '
|
|
320
413
|
f'& Koch (1977) benchmarks."'
|
|
321
414
|
)
|
|
322
|
-
return interpretation, example
|
|
415
|
+
return band_phrase, interpretation, example
|
|
323
416
|
|
|
324
417
|
|
|
325
|
-
def _interpret_corr(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str]:
|
|
418
|
+
def _interpret_corr(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str, str]:
|
|
326
419
|
"""Cohen (1988) conventions for correlation-coefficient magnitude."""
|
|
327
420
|
a = abs(est)
|
|
328
421
|
if a < 0.10:
|
|
@@ -334,16 +427,37 @@ def _interpret_corr(est: float, lo: float, hi: float, n: int, label: str) -> tup
|
|
|
334
427
|
else:
|
|
335
428
|
band = "large"
|
|
336
429
|
direction = "positive" if est >= 0 else "negative"
|
|
337
|
-
|
|
430
|
+
band_phrase = f"{band} {direction} correlation"
|
|
431
|
+
interpretation = f"{band_phrase} (Cohen, 1988 conventions)"
|
|
338
432
|
example = (
|
|
339
433
|
f'"{label} = {est:.2f}, 95% CI [{lo:.2f}, {hi:.2f}] (n={n}), a {band} '
|
|
340
434
|
f'{direction} correlation between the LLM judge and human scores (Cohen, '
|
|
341
435
|
f'1988 conventions)."'
|
|
342
436
|
)
|
|
343
|
-
return interpretation, example
|
|
437
|
+
return band_phrase, interpretation, example
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def _interpret_icc(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str, str]:
|
|
441
|
+
"""Koo & Li (2016) benchmarks for ICC magnitude."""
|
|
442
|
+
if est < 0.50:
|
|
443
|
+
band = "poor"
|
|
444
|
+
elif est < 0.75:
|
|
445
|
+
band = "moderate"
|
|
446
|
+
elif est < 0.90:
|
|
447
|
+
band = "good"
|
|
448
|
+
else:
|
|
449
|
+
band = "excellent"
|
|
450
|
+
band_phrase = f"{band} absolute agreement"
|
|
451
|
+
interpretation = f"{band_phrase} (Koo & Li, 2016 benchmarks)"
|
|
452
|
+
example = (
|
|
453
|
+
f'"{label} = {est:.2f}, 95% CI [{lo:.2f}, {hi:.2f}] (n={n}), indicating '
|
|
454
|
+
f'{band} absolute agreement between the LLM judge and human raters, per '
|
|
455
|
+
f'Koo & Li (2016) benchmarks."'
|
|
456
|
+
)
|
|
457
|
+
return band_phrase, interpretation, example
|
|
344
458
|
|
|
345
459
|
|
|
346
|
-
def _interpret_pct_agreement(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str]:
|
|
460
|
+
def _interpret_pct_agreement(est: float, lo: float, hi: float, n: int, label: str) -> tuple[Optional[str], str, str]:
|
|
347
461
|
interpretation = (
|
|
348
462
|
"no universally-agreed threshold exists for raw percent agreement — read it "
|
|
349
463
|
"alongside Cohen's κ, since it does not correct for chance and can look high "
|
|
@@ -353,7 +467,7 @@ def _interpret_pct_agreement(est: float, lo: float, hi: float, n: int, label: st
|
|
|
353
467
|
f'"the LLM judge matched human labels on {est * 100:.1f}% of items, 95% CI '
|
|
354
468
|
f'[{lo * 100:.1f}%, {hi * 100:.1f}%] (n={n})."'
|
|
355
469
|
)
|
|
356
|
-
return interpretation, example
|
|
470
|
+
return None, interpretation, example
|
|
357
471
|
|
|
358
472
|
|
|
359
473
|
def _bootstrap_ci_2(
|
|
@@ -376,6 +490,134 @@ def _bootstrap_ci_2(
|
|
|
376
490
|
return obs, lo, hi
|
|
377
491
|
|
|
378
492
|
|
|
493
|
+
def _icc_21(a: np.ndarray, b: np.ndarray) -> float:
|
|
494
|
+
"""Shrout & Fleiss (1979) ICC(2,1): two-way random effects, single rater,
|
|
495
|
+
absolute agreement, for exactly two raters (``a``, ``b``).
|
|
496
|
+
|
|
497
|
+
Unlike Pearson/Spearman r or weighted kappa's category-index distance,
|
|
498
|
+
this is sensitive to a systematic offset or scale mismatch between the
|
|
499
|
+
two raters — it measures whether they land on the same absolute values,
|
|
500
|
+
not just whether they move together.
|
|
501
|
+
"""
|
|
502
|
+
n = len(a)
|
|
503
|
+
data = np.column_stack([a, b]).astype(float)
|
|
504
|
+
k = 2
|
|
505
|
+
grand_mean = data.mean()
|
|
506
|
+
row_means = data.mean(axis=1)
|
|
507
|
+
col_means = data.mean(axis=0)
|
|
508
|
+
|
|
509
|
+
df_row = max(n - 1, 1)
|
|
510
|
+
SSR = k * np.sum((row_means - grand_mean) ** 2)
|
|
511
|
+
SSC = n * np.sum((col_means - grand_mean) ** 2) # (k-1) == 1
|
|
512
|
+
SST = np.sum((data - grand_mean) ** 2)
|
|
513
|
+
SSE = SST - SSR - SSC
|
|
514
|
+
|
|
515
|
+
MSR = SSR / df_row
|
|
516
|
+
MSC = SSC
|
|
517
|
+
MSE = SSE / df_row # (n-1)(k-1) == n-1
|
|
518
|
+
|
|
519
|
+
denom = MSR + MSE + 2.0 * (MSC - MSE) / n
|
|
520
|
+
if denom <= 1e-12:
|
|
521
|
+
return 1.0
|
|
522
|
+
return float((MSR - MSE) / denom)
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def _bootstrap_ci_gap(
|
|
526
|
+
fn_corr,
|
|
527
|
+
fn_icc,
|
|
528
|
+
a: np.ndarray,
|
|
529
|
+
b: np.ndarray,
|
|
530
|
+
*,
|
|
531
|
+
n_boot: int = 2000,
|
|
532
|
+
alpha: float = 0.05,
|
|
533
|
+
rng: np.random.Generator,
|
|
534
|
+
) -> tuple[float, float, float]:
|
|
535
|
+
"""Paired bootstrap CI for (correlation-type metric − ICC(2,1)).
|
|
536
|
+
|
|
537
|
+
Resamples items once per draw and evaluates both statistics on the same
|
|
538
|
+
resample, so the CI reflects the sampling distribution of the *gap*
|
|
539
|
+
itself, not the (looser, more conservative) union of two independently
|
|
540
|
+
bootstrapped CIs.
|
|
541
|
+
"""
|
|
542
|
+
n = len(a)
|
|
543
|
+
obs = float(fn_corr(a, b) - fn_icc(a, b))
|
|
544
|
+
boot = np.empty(n_boot)
|
|
545
|
+
for i in range(n_boot):
|
|
546
|
+
idx = rng.integers(0, n, size=n)
|
|
547
|
+
boot[i] = fn_corr(a[idx], b[idx]) - fn_icc(a[idx], b[idx])
|
|
548
|
+
lo = float(np.percentile(boot, 100.0 * alpha / 2))
|
|
549
|
+
hi = float(np.percentile(boot, 100.0 * (1.0 - alpha / 2)))
|
|
550
|
+
return obs, lo, hi
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def _build_bias_check(
|
|
554
|
+
corr_label: str,
|
|
555
|
+
corr_est: float,
|
|
556
|
+
icc_est: float,
|
|
557
|
+
gap_est: float,
|
|
558
|
+
gap_lo: float,
|
|
559
|
+
gap_hi: float,
|
|
560
|
+
) -> dict:
|
|
561
|
+
"""Package the correlation-vs-ICC(2,1) discrepancy check into a dict with
|
|
562
|
+
the same shape as the representativeness checks, so it can be rendered
|
|
563
|
+
with the same ``_print_check`` helper in ``AlignmentResult.summary()``.
|
|
564
|
+
"""
|
|
565
|
+
flagged = gap_lo > 0.0
|
|
566
|
+
if flagged:
|
|
567
|
+
message = (
|
|
568
|
+
f"possible judge bias: {corr_label} = {corr_est:.3f} but "
|
|
569
|
+
f"ICC(2,1) = {icc_est:.3f} (gap = {gap_est:.3f}, 95% CI "
|
|
570
|
+
f"[{gap_lo:.3f}, {gap_hi:.3f}], excludes 0)"
|
|
571
|
+
)
|
|
572
|
+
else:
|
|
573
|
+
message = (
|
|
574
|
+
f"no evidence of scale bias: {corr_label} = {corr_est:.3f} and "
|
|
575
|
+
f"ICC(2,1) = {icc_est:.3f} are consistent (gap 95% CI "
|
|
576
|
+
f"[{gap_lo:.3f}, {gap_hi:.3f}] includes 0)"
|
|
577
|
+
)
|
|
578
|
+
what = (
|
|
579
|
+
f"Compares {corr_label}, which is insensitive to a systematic offset "
|
|
580
|
+
"or scale difference between judge and human scores, against "
|
|
581
|
+
"ICC(2,1), which penalizes exactly that. A paired bootstrap CI is "
|
|
582
|
+
"used so the comparison reflects the sampling distribution of the "
|
|
583
|
+
"gap itself rather than the union of two separate CIs."
|
|
584
|
+
)
|
|
585
|
+
why = (
|
|
586
|
+
"A correlation-type metric can look strong even when a judge is "
|
|
587
|
+
"systematically shifted or compressed relative to human scores, "
|
|
588
|
+
"since it only requires the two to move together. This check exists "
|
|
589
|
+
"to catch that failure mode before it's mistaken for genuine "
|
|
590
|
+
"agreement."
|
|
591
|
+
)
|
|
592
|
+
if flagged:
|
|
593
|
+
interpretation = (
|
|
594
|
+
"the judge tracks human relative ordering but disagrees on "
|
|
595
|
+
"absolute scale — treat raw judge scores as biased; consider "
|
|
596
|
+
"using the Bayesian calibration model fit by validate_alignment "
|
|
597
|
+
"(e.g. via compare(alignment=...)) to correct for it before "
|
|
598
|
+
"drawing conclusions from raw judge scores"
|
|
599
|
+
)
|
|
600
|
+
else:
|
|
601
|
+
interpretation = (
|
|
602
|
+
"the correlation and absolute-agreement metrics tell a "
|
|
603
|
+
"consistent story — no sign that the judge's ranking ability is "
|
|
604
|
+
"masking a scale or offset problem"
|
|
605
|
+
)
|
|
606
|
+
return {
|
|
607
|
+
"passed": not flagged,
|
|
608
|
+
"message": message,
|
|
609
|
+
"corr_label": corr_label,
|
|
610
|
+
"corr_estimate": corr_est,
|
|
611
|
+
"icc_estimate": icc_est,
|
|
612
|
+
"gap": gap_est,
|
|
613
|
+
"gap_ci_low": gap_lo,
|
|
614
|
+
"gap_ci_high": gap_hi,
|
|
615
|
+
"what": what,
|
|
616
|
+
"why": why,
|
|
617
|
+
"interpretation": interpretation,
|
|
618
|
+
}
|
|
619
|
+
|
|
620
|
+
|
|
379
621
|
def _compute_alignment_metrics(
|
|
380
622
|
llm: np.ndarray,
|
|
381
623
|
human: np.ndarray,
|
|
@@ -399,10 +641,11 @@ def _compute_alignment_metrics(
|
|
|
399
641
|
return (p_o - p_e) / (1.0 - p_e) if p_e < 1.0 else 1.0
|
|
400
642
|
|
|
401
643
|
est, lo, hi = _bootstrap_ci_2(agree, llm, human, alpha=alpha, rng=rng)
|
|
402
|
-
interp, example = _interpret_pct_agreement(est, lo, hi, len(llm), "Percent agreement")
|
|
644
|
+
band, interp, example = _interpret_pct_agreement(est, lo, hi, len(llm), "Percent agreement")
|
|
403
645
|
metrics["percent_agreement"] = {
|
|
404
646
|
"estimate": est, "ci_low": lo, "ci_high": hi,
|
|
405
647
|
"label": "Percent agreement",
|
|
648
|
+
"band": band,
|
|
406
649
|
"what": (
|
|
407
650
|
"The fraction of items where the LLM judge's label exactly matches "
|
|
408
651
|
"the human label."
|
|
@@ -415,10 +658,11 @@ def _compute_alignment_metrics(
|
|
|
415
658
|
"example": example,
|
|
416
659
|
}
|
|
417
660
|
est, lo, hi = _bootstrap_ci_2(kappa, llm, human, alpha=alpha, rng=rng)
|
|
418
|
-
interp, example = _interpret_kappa(est, lo, hi, len(llm), "Cohen's κ")
|
|
661
|
+
band, interp, example = _interpret_kappa(est, lo, hi, len(llm), "Cohen's κ")
|
|
419
662
|
metrics["cohens_kappa"] = {
|
|
420
663
|
"estimate": est, "ci_low": lo, "ci_high": hi,
|
|
421
664
|
"label": "Cohen's κ",
|
|
665
|
+
"band": band,
|
|
422
666
|
"what": (
|
|
423
667
|
"Percent agreement adjusted for the rate of agreement expected from "
|
|
424
668
|
"two raters guessing at random, given the observed marginal label "
|
|
@@ -457,10 +701,11 @@ def _compute_alignment_metrics(
|
|
|
457
701
|
|
|
458
702
|
if k >= 2:
|
|
459
703
|
est, lo, hi = _bootstrap_ci_2(wk, llm, human, alpha=alpha, rng=rng)
|
|
460
|
-
interp, example = _interpret_kappa(est, lo, hi, len(llm), "Weighted Cohen's κ")
|
|
704
|
+
band, interp, example = _interpret_kappa(est, lo, hi, len(llm), "Weighted Cohen's κ")
|
|
461
705
|
metrics["weighted_kappa"] = {
|
|
462
706
|
"estimate": est, "ci_low": lo, "ci_high": hi,
|
|
463
707
|
"label": "Weighted Cohen's κ",
|
|
708
|
+
"band": band,
|
|
464
709
|
"what": (
|
|
465
710
|
"Cohen's κ extended so that disagreements receive larger penalties "
|
|
466
711
|
"as ratings become farther apart on the ordinal scale (Cohen, 1968)."
|
|
@@ -475,10 +720,11 @@ def _compute_alignment_metrics(
|
|
|
475
720
|
"example": example,
|
|
476
721
|
}
|
|
477
722
|
est, lo, hi = _bootstrap_ci_2(sp, llm, human, alpha=alpha, rng=rng)
|
|
478
|
-
interp, example = _interpret_corr(est, lo, hi, len(llm), "Spearman r")
|
|
723
|
+
band, interp, example = _interpret_corr(est, lo, hi, len(llm), "Spearman r")
|
|
479
724
|
metrics["spearman_r"] = {
|
|
480
725
|
"estimate": est, "ci_low": lo, "ci_high": hi,
|
|
481
726
|
"label": "Spearman r",
|
|
727
|
+
"band": band,
|
|
482
728
|
"what": (
|
|
483
729
|
"Rank correlation between judge and human scores — checks whether "
|
|
484
730
|
"higher judge scores correspond to higher human scores, without "
|
|
@@ -493,6 +739,36 @@ def _compute_alignment_metrics(
|
|
|
493
739
|
"example": example,
|
|
494
740
|
}
|
|
495
741
|
|
|
742
|
+
if k >= 2:
|
|
743
|
+
icc_est, icc_lo, icc_hi = _bootstrap_ci_2(_icc_21, llm, human, alpha=alpha, rng=rng)
|
|
744
|
+
band, interp, example = _interpret_icc(icc_est, icc_lo, icc_hi, len(llm), "ICC(2,1)")
|
|
745
|
+
metrics["icc_21"] = {
|
|
746
|
+
"estimate": icc_est, "ci_low": icc_lo, "ci_high": icc_hi,
|
|
747
|
+
"label": "ICC(2,1)",
|
|
748
|
+
"band": band,
|
|
749
|
+
"what": (
|
|
750
|
+
"Two-way random-effects intraclass correlation for absolute "
|
|
751
|
+
"agreement (Shrout & Fleiss, 1979): unlike weighted κ's "
|
|
752
|
+
"category-index distance or Spearman r's rank comparison, it "
|
|
753
|
+
"is sensitive to a systematic offset between the judge and "
|
|
754
|
+
"human scale, not just whether they move together."
|
|
755
|
+
),
|
|
756
|
+
"why": (
|
|
757
|
+
"Computed alongside weighted κ to check for absolute-scale "
|
|
758
|
+
"bias: a judge that ranks items correctly but is shifted or "
|
|
759
|
+
"compressed relative to human scores can still get a high "
|
|
760
|
+
"weighted κ / Spearman r while scoring poorly here."
|
|
761
|
+
),
|
|
762
|
+
"interpretation": interp,
|
|
763
|
+
"example": example,
|
|
764
|
+
}
|
|
765
|
+
|
|
766
|
+
gap_est, gap_lo, gap_hi = _bootstrap_ci_gap(wk, _icc_21, llm, human, alpha=alpha, rng=rng)
|
|
767
|
+
metrics["_bias_check"] = _build_bias_check(
|
|
768
|
+
"Weighted Cohen's κ", metrics["weighted_kappa"]["estimate"],
|
|
769
|
+
icc_est, gap_est, gap_lo, gap_hi,
|
|
770
|
+
)
|
|
771
|
+
|
|
496
772
|
else: # continuous / grade
|
|
497
773
|
def pe(a, b):
|
|
498
774
|
r, _ = pearsonr(a, b)
|
|
@@ -503,10 +779,11 @@ def _compute_alignment_metrics(
|
|
|
503
779
|
return float(r)
|
|
504
780
|
|
|
505
781
|
est, lo, hi = _bootstrap_ci_2(pe, llm, human, alpha=alpha, rng=rng)
|
|
506
|
-
interp, example = _interpret_corr(est, lo, hi, len(llm), "Pearson r")
|
|
782
|
+
band, interp, example = _interpret_corr(est, lo, hi, len(llm), "Pearson r")
|
|
507
783
|
metrics["pearson_r"] = {
|
|
508
784
|
"estimate": est, "ci_low": lo, "ci_high": hi,
|
|
509
785
|
"label": "Pearson r",
|
|
786
|
+
"band": band,
|
|
510
787
|
"what": "Linear correlation coefficient between judge and human scores.",
|
|
511
788
|
"why": (
|
|
512
789
|
"Your judge produces continuous/numeric scores, so a correlation "
|
|
@@ -516,10 +793,11 @@ def _compute_alignment_metrics(
|
|
|
516
793
|
"example": example,
|
|
517
794
|
}
|
|
518
795
|
est, lo, hi = _bootstrap_ci_2(sp, llm, human, alpha=alpha, rng=rng)
|
|
519
|
-
interp, example = _interpret_corr(est, lo, hi, len(llm), "Spearman r")
|
|
796
|
+
band, interp, example = _interpret_corr(est, lo, hi, len(llm), "Spearman r")
|
|
520
797
|
metrics["spearman_r"] = {
|
|
521
798
|
"estimate": est, "ci_low": lo, "ci_high": hi,
|
|
522
799
|
"label": "Spearman r",
|
|
800
|
+
"band": band,
|
|
523
801
|
"what": "Rank correlation between judge and human scores.",
|
|
524
802
|
"why": (
|
|
525
803
|
"Reported alongside Pearson r to check whether agreement holds even "
|
|
@@ -530,6 +808,35 @@ def _compute_alignment_metrics(
|
|
|
530
808
|
"example": example,
|
|
531
809
|
}
|
|
532
810
|
|
|
811
|
+
icc_est, icc_lo, icc_hi = _bootstrap_ci_2(_icc_21, llm, human, alpha=alpha, rng=rng)
|
|
812
|
+
band, interp, example = _interpret_icc(icc_est, icc_lo, icc_hi, len(llm), "ICC(2,1)")
|
|
813
|
+
metrics["icc_21"] = {
|
|
814
|
+
"estimate": icc_est, "ci_low": icc_lo, "ci_high": icc_hi,
|
|
815
|
+
"label": "ICC(2,1)",
|
|
816
|
+
"band": band,
|
|
817
|
+
"what": (
|
|
818
|
+
"Two-way random-effects intraclass correlation for absolute "
|
|
819
|
+
"agreement (Shrout & Fleiss, 1979): unlike Pearson/Spearman r, "
|
|
820
|
+
"which are invariant to any linear rescaling of one variable, "
|
|
821
|
+
"this is sensitive to a systematic offset or scale mismatch "
|
|
822
|
+
"between the judge and human scale."
|
|
823
|
+
),
|
|
824
|
+
"why": (
|
|
825
|
+
"Computed alongside Pearson r to check for absolute-scale bias: "
|
|
826
|
+
"a judge that is consistently shifted or compressed relative to "
|
|
827
|
+
"human scores can still score a perfect Pearson r while "
|
|
828
|
+
"disagreeing badly here."
|
|
829
|
+
),
|
|
830
|
+
"interpretation": interp,
|
|
831
|
+
"example": example,
|
|
832
|
+
}
|
|
833
|
+
|
|
834
|
+
gap_est, gap_lo, gap_hi = _bootstrap_ci_gap(pe, _icc_21, llm, human, alpha=alpha, rng=rng)
|
|
835
|
+
metrics["_bias_check"] = _build_bias_check(
|
|
836
|
+
"Pearson r", metrics["pearson_r"]["estimate"],
|
|
837
|
+
icc_est, gap_est, gap_lo, gap_hi,
|
|
838
|
+
)
|
|
839
|
+
|
|
533
840
|
return metrics
|
|
534
841
|
|
|
535
842
|
|
|
@@ -756,6 +1063,7 @@ def validate_alignment(
|
|
|
756
1063
|
alignment_metrics = _compute_alignment_metrics(
|
|
757
1064
|
llm_aligned, human_aligned, score_type, alpha=alpha, rng=rng
|
|
758
1065
|
)
|
|
1066
|
+
bias_check = alignment_metrics.pop("_bias_check", None)
|
|
759
1067
|
|
|
760
1068
|
# Representativeness: score distribution
|
|
761
1069
|
rep: dict = {}
|
|
@@ -801,4 +1109,5 @@ def validate_alignment(
|
|
|
801
1109
|
calibration=calibration,
|
|
802
1110
|
alignment_metrics=alignment_metrics,
|
|
803
1111
|
representativeness=rep,
|
|
1112
|
+
bias_check=bias_check,
|
|
804
1113
|
)
|
|
@@ -719,8 +719,8 @@ def _bridge_to_io(
|
|
|
719
719
|
# PPI alignment correction
|
|
720
720
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
721
721
|
|
|
722
|
-
_PPI_PAIRWISE_SUPPORTED = ("tango", "t_interval", "bootstrap", "wilcoxon", "mannwhitney", "bootstrap_t", "bayes_bootstrap")
|
|
723
|
-
_PPI_ROBUSTNESS_SUPPORTED = ("wilson", "bootstrap", "bootstrap_t")
|
|
722
|
+
_PPI_PAIRWISE_SUPPORTED = ("tango", "t_interval", "bootstrap", "wilcoxon", "mannwhitney", "bootstrap_t", "bayes_bootstrap", "ppi_t_interval", "ppi_logit_t")
|
|
723
|
+
_PPI_ROBUSTNESS_SUPPORTED = ("wilson", "bootstrap", "bootstrap_t", "ppi_t_interval", "ppi_logit_t")
|
|
724
724
|
|
|
725
725
|
|
|
726
726
|
def _ppi_pairwise_dispatch(method: str, a, b, a_lab, b_lab, alpha: float, n_boot: int, rng):
|
|
@@ -729,10 +729,18 @@ def _ppi_pairwise_dispatch(method: str, a, b, a_lab, b_lab, alpha: float, n_boot
|
|
|
729
729
|
Only methods with a validated PPI-corrected counterpart (see
|
|
730
730
|
``evalstats.tests``'s ``_ppi_paired_*``/``_ppi_two_sample`` functions,
|
|
731
731
|
calibrated via ``simulations/harness --mode ppi``) are supported here.
|
|
732
|
+
|
|
733
|
+
"ppi_t_interval"/"ppi_logit_t" are DISTINCT method strings from the
|
|
734
|
+
existing bare "t_interval" (below) -- that one already maps to
|
|
735
|
+
``_ppi_paired_arrays(..., np.mean, rectifier_func=np.mean)``, the
|
|
736
|
+
generic PPI-mean-diff bootstrap routine, not the closed-form analytic
|
|
737
|
+
construction these two use. Reusing "t_interval"/"logit_t" here would
|
|
738
|
+
silently collide with that existing mapping.
|
|
732
739
|
"""
|
|
733
740
|
from evalstats.tests import (
|
|
734
741
|
_ppi_paired_tango, _ppi_paired_bootstrap_t, _ppi_paired_bayes_bootstrap,
|
|
735
|
-
_ppi_paired_arrays,
|
|
742
|
+
_ppi_paired_arrays, _ppi_two_sample, _p_x_gt_y_midrank,
|
|
743
|
+
_ppi_paired_t_interval, _ppi_paired_logit_t,
|
|
736
744
|
)
|
|
737
745
|
if method == "tango":
|
|
738
746
|
return _ppi_paired_tango(a, b, a_lab, b_lab, alpha)
|
|
@@ -740,17 +748,30 @@ def _ppi_pairwise_dispatch(method: str, a, b, a_lab, b_lab, alpha: float, n_boot
|
|
|
740
748
|
return _ppi_paired_bootstrap_t(a, b, a_lab, b_lab, alpha, n_boot, rng)
|
|
741
749
|
if method == "bayes_bootstrap":
|
|
742
750
|
return _ppi_paired_bayes_bootstrap(a, b, a_lab, b_lab, alpha, n_boot, rng)
|
|
751
|
+
if method == "ppi_t_interval":
|
|
752
|
+
return _ppi_paired_t_interval(a, b, a_lab, b_lab, alpha)
|
|
753
|
+
if method == "ppi_logit_t":
|
|
754
|
+
# lo/hi default (0.0, 1.0): this dispatch path has no score_range
|
|
755
|
+
# concept (see _run_alignment_ppi's is_bounded_01_scores check --
|
|
756
|
+
# "bounded_01" always means raw scores are literally in [0, 1] here).
|
|
757
|
+
return _ppi_paired_logit_t(a, b, a_lab, b_lab, alpha)
|
|
743
758
|
if method in ("t_interval", "bootstrap"):
|
|
744
759
|
return _ppi_paired_arrays(a, b, a_lab, b_lab, np.mean, alpha, n_boot, rng, rectifier_func=np.mean)
|
|
745
760
|
if method == "wilcoxon":
|
|
746
761
|
return _ppi_paired_arrays(a, b, a_lab, b_lab, np.median, alpha, n_boot, rng, rectifier_func=np.mean)
|
|
747
762
|
if method == "mannwhitney":
|
|
748
|
-
#
|
|
749
|
-
#
|
|
750
|
-
#
|
|
751
|
-
#
|
|
752
|
-
#
|
|
753
|
-
|
|
763
|
+
# Matches evalstats.tests.mannwhitney's method="global" default
|
|
764
|
+
# (reinstated 2026-08-02, a few hours after "ridge" -- see
|
|
765
|
+
# mannwhitney's `method` docstring's "REVERTED TO 'global'" note
|
|
766
|
+
# for the full six-default history). NOT a finding against
|
|
767
|
+
# "ridge" -- it remains the best-validated option by every number
|
|
768
|
+
# gathered -- reverted because the harness's --official-tests
|
|
769
|
+
# suite has always tested plain "global" under the name "mwu"
|
|
770
|
+
# (hardcoded, doesn't track mannwhitney()'s actual default), so
|
|
771
|
+
# production stays aligned with what's actually been exercised by
|
|
772
|
+
# that sanctioned pipeline until "ridge" gets its own official
|
|
773
|
+
# pass.
|
|
774
|
+
return _ppi_two_sample(a, b, a_lab, b_lab, lambda xa, ya: _p_x_gt_y_midrank(xa, ya) - 0.5, alpha, n_boot, rng)
|
|
754
775
|
raise ValueError(
|
|
755
776
|
f"PPI alignment correction has no validated implementation for pairwise "
|
|
756
777
|
f"method {method!r}. Supported pairwise methods: "
|
|
@@ -917,11 +938,18 @@ def _max_t_from_joint_stats(
|
|
|
917
938
|
|
|
918
939
|
def _ppi_robustness_dispatch(method: str, a, a_lab, alpha: float, n_boot: int, rng):
|
|
919
940
|
"""Dispatch to the PPI-corrected single-sample implementation of *method*."""
|
|
920
|
-
from evalstats.tests import
|
|
941
|
+
from evalstats.tests import (
|
|
942
|
+
_ppi_single_wilson, _ppi_single_bootstrap_t, _ppi_single_t_interval, _ppi_single_logit_t,
|
|
943
|
+
)
|
|
921
944
|
if method == "wilson":
|
|
922
945
|
return _ppi_single_wilson(a, a_lab, alpha)
|
|
923
946
|
if method == "bootstrap_t":
|
|
924
947
|
return _ppi_single_bootstrap_t(a, a_lab, alpha, n_boot, rng)
|
|
948
|
+
if method == "ppi_t_interval":
|
|
949
|
+
return _ppi_single_t_interval(a, a_lab, alpha)
|
|
950
|
+
if method == "ppi_logit_t":
|
|
951
|
+
# lo/hi default (0.0, 1.0) -- see _ppi_pairwise_dispatch's matching note.
|
|
952
|
+
return _ppi_single_logit_t(a, a_lab, alpha)
|
|
925
953
|
if method == "bootstrap":
|
|
926
954
|
from evalstats.ppi import correct as _ppi_correct
|
|
927
955
|
mask = ~np.isnan(a_lab)
|