evalstats 0.2.4__tar.gz → 0.2.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. {evalstats-0.2.4 → evalstats-0.2.5}/PKG-INFO +1 -1
  2. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/__init__.py +1 -1
  3. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/alignment.py +328 -19
  4. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/api.py +38 -10
  5. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/config.py +18 -19
  6. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/resampling.py +24 -11
  7. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/ppi.py +609 -52
  8. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/tests/__init__.py +1118 -107
  9. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/PKG-INFO +1 -1
  10. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/SOURCES.txt +1 -0
  11. {evalstats-0.2.4 → evalstats-0.2.5}/pyproject.toml +1 -1
  12. evalstats-0.2.5/tests/test_ppi_ci_methods.py +306 -0
  13. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_ppi_corrections.py +25 -1
  14. {evalstats-0.2.4 → evalstats-0.2.5}/LICENSE +0 -0
  15. {evalstats-0.2.4 → evalstats-0.2.5}/README.md +0 -0
  16. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/cli.py +0 -0
  17. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/__init__.py +0 -0
  18. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/bayes_evals.py +0 -0
  19. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/bundles.py +0 -0
  20. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/mixed_effects.py +0 -0
  21. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/paired.py +0 -0
  22. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/ranking.py +0 -0
  23. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/router.py +0 -0
  24. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/stats_utils.py +0 -0
  25. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/summary.py +0 -0
  26. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/types.py +0 -0
  27. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/core/variance.py +0 -0
  28. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/io.py +0 -0
  29. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/loader.py +0 -0
  30. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/__init__.py +0 -0
  31. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/critical_difference.py +0 -0
  32. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/forest.py +0 -0
  33. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/heatmap.py +0 -0
  34. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/point_estimates.py +0 -0
  35. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats/vis/scoreboard.py +0 -0
  36. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/dependency_links.txt +0 -0
  37. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/entry_points.txt +0 -0
  38. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/requires.txt +0 -0
  39. {evalstats-0.2.4 → evalstats-0.2.5}/evalstats.egg-info/top_level.txt +0 -0
  40. {evalstats-0.2.4 → evalstats-0.2.5}/setup.cfg +0 -0
  41. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_alignment.py +0 -0
  42. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_analyze.py +0 -0
  43. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_analyze_factorial.py +0 -0
  44. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_auto_ci_routing.py +0 -0
  45. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_bayes_binary_routing.py +0 -0
  46. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_bootstrap_t_pairwise_ranking.py +0 -0
  47. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_cli.py +0 -0
  48. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_compare.py +0 -0
  49. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_critical_difference_plot.py +0 -0
  50. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_io.py +0 -0
  51. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_lmm.py +0 -0
  52. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_lmm_backend_parity.py +0 -0
  53. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_lmm_statsmodels.py +0 -0
  54. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_nig_ci_methods.py +0 -0
  55. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_p_values.py +0 -0
  56. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_permutation.py +0 -0
  57. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_ppi_core.py +0 -0
  58. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_resampling.py +0 -0
  59. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_scoreboard_plot.py +0 -0
  60. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_set_alpha_ci.py +0 -0
  61. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_simultaneous_ci.py +0 -0
  62. {evalstats-0.2.4 → evalstats-0.2.5}/tests/test_wilson_newcombe.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: evalstats
3
- Version: 0.2.4
3
+ Version: 0.2.5
4
4
  Summary: Statistically sane analysis methods for comparing AI model and prompt performance.
5
5
  Author: Ian Arawjo
6
6
  License-Expression: MIT
@@ -37,7 +37,7 @@ from evalstats.alignment import validate_alignment, AlignmentResult
37
37
  from evalstats import ppi
38
38
  from evalstats import tests
39
39
 
40
- __version__ = "0.2.4"
40
+ __version__ = "0.2.5"
41
41
 
42
42
  __all__ = [
43
43
  # High-level spec API
@@ -43,6 +43,12 @@ class AlignmentResult:
43
43
  Point estimates and bootstrap CIs for each alignment metric.
44
44
  representativeness : dict
45
45
  Representativeness check results (distribution and slice columns).
46
+ bias_check : dict or None
47
+ For likert/continuous/grade score types, compares the correlation-type
48
+ metric (weighted κ or Pearson r) against ICC(2,1) to flag whether the
49
+ judge is systematically biased in absolute scale despite tracking
50
+ human relative ordering. ``None`` for binary score types, where ICC
51
+ isn't computed.
46
52
  """
47
53
 
48
54
  def __init__(
@@ -56,6 +62,7 @@ class AlignmentResult:
56
62
  calibration: dict,
57
63
  alignment_metrics: dict,
58
64
  representativeness: dict,
65
+ bias_check: Optional[dict] = None,
59
66
  ) -> None:
60
67
  self.llm_metric = llm_metric
61
68
  self.human_col = human_col
@@ -65,6 +72,7 @@ class AlignmentResult:
65
72
  self._calibration = calibration
66
73
  self.alignment_metrics = alignment_metrics
67
74
  self.representativeness = representativeness
75
+ self.bias_check = bias_check
68
76
 
69
77
  # ── sampling ─────────────────────────────────────────────────────────────
70
78
 
@@ -130,18 +138,97 @@ class AlignmentResult:
130
138
 
131
139
  # ── display ──────────────────────────────────────────────────────────────
132
140
 
133
- def summary(self) -> None:
134
- """Print a plain-language alignment and representativeness report."""
135
- width = 58
136
- print("Judge alignment report")
137
- print("─" * width)
141
+ def summary(self, verbose: bool = False) -> None:
142
+ """Print an alignment and representativeness report.
143
+
144
+ Parameters
145
+ ----------
146
+ verbose : bool
147
+ If ``False`` (default), print a short, plain-language report:
148
+ one line per check/metric, with an explanation only where
149
+ something looks off. Aimed at readers who don't need the
150
+ statistical background spelled out every time.
151
+ If ``True``, print the full report — every metric's definition,
152
+ why it was chosen, how to interpret it, and citation-ready
153
+ wording for a paper.
154
+ """
155
+ if verbose:
156
+ self._summary_verbose()
157
+ else:
158
+ self._summary_simple()
159
+
160
+ def _header(self) -> None:
138
161
  pct = 100.0 * self.n_labeled / self.n_total if self.n_total > 0 else 0.0
162
+ print("Judge alignment report")
163
+ print("─" * 58)
139
164
  print(
140
165
  f"Alignment set : {self.n_labeled} of {self.n_total} items "
141
166
  f"have human labels ({pct:.1f}%)"
142
167
  )
143
168
  print()
144
169
 
170
+ def _summary_simple(self) -> None:
171
+ self._header()
172
+
173
+ if self.bias_check is not None:
174
+ bc = self.bias_check
175
+ if not bc["passed"]:
176
+ print(
177
+ f"⚠ Possible judge bias: {bc['corr_label']} = "
178
+ f"{bc['corr_estimate']:.2f} but ICC(2,1) = {bc['icc_estimate']:.2f} "
179
+ "— the judge ranks items like humans do, but its raw scores "
180
+ "look shifted or compressed relative to human scores."
181
+ )
182
+ print(
183
+ " Treat raw judge scores with caution; consider "
184
+ "recalibrating (compare(alignment=...)) before using them "
185
+ "directly. Run .summary(verbose=True) for the full check."
186
+ )
187
+ else:
188
+ print(
189
+ f"✓ No sign of judge bias: {bc['corr_label']} and ICC(2,1) "
190
+ "roughly agree."
191
+ )
192
+ print()
193
+
194
+ rep = self.representativeness
195
+ rep_failed = [
196
+ (k, v) for k, v in rep.items() if not v["passed"]
197
+ ]
198
+ if rep_failed:
199
+ print("⚠ Representativeness: the labeled sample may not be representative")
200
+ for key, val in rep_failed:
201
+ name = "score distribution" if key == "score_distribution" else key[len("slice_"):]
202
+ print(f" - {name}: {val['message']}")
203
+ else:
204
+ print("✓ Representativeness: labeled items look like the full item pool")
205
+ print()
206
+
207
+ score_type_note = _SCORE_TYPE_NOTES.get(
208
+ self.score_type, f"score type detected as {self.score_type!r}"
209
+ )
210
+ print(f"Alignment metrics ({self.score_type} scores — {score_type_note}):")
211
+ for entry in self.alignment_metrics.values():
212
+ label = entry.get("label", "")
213
+ est = entry["estimate"]
214
+ lo = entry["ci_low"]
215
+ hi = entry["ci_high"]
216
+ band = entry.get("band")
217
+ tail = f" {band}" if band else ""
218
+ print(f" {label:<20} {est:6.2f} [{lo:5.2f}, {hi:5.2f}]{tail}")
219
+ print()
220
+ print("Run .summary(verbose=True) for definitions, rationale, and")
221
+ print("citation-ready wording for each check above.")
222
+ print("─" * 58)
223
+
224
+ def _summary_verbose(self) -> None:
225
+ self._header()
226
+
227
+ if self.bias_check is not None and not self.bias_check["passed"]:
228
+ print(f"⚠ Judge bias flag: {self.bias_check['message']}")
229
+ print(" (see 'Bias diagnostics' below for details)")
230
+ print()
231
+
145
232
  # Representativeness
146
233
  rep = self.representativeness
147
234
 
@@ -196,7 +283,12 @@ class AlignmentResult:
196
283
  if example:
197
284
  print(f" -> Example paper reporting: {example}")
198
285
  print()
199
- print("─" * width)
286
+
287
+ if self.bias_check is not None:
288
+ print("Bias diagnostics:")
289
+ _print_check("Judge scale bias (correlation vs. ICC)", self.bias_check)
290
+
291
+ print("─" * 58)
200
292
 
201
293
  def __repr__(self) -> str:
202
294
  return (
@@ -299,7 +391,7 @@ _SCORE_TYPE_NOTES = {
299
391
  }
300
392
 
301
393
 
302
- def _interpret_kappa(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str]:
394
+ def _interpret_kappa(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str, str]:
303
395
  """Landis & Koch (1977) benchmarks for kappa-type statistics."""
304
396
  if est < 0:
305
397
  band = "poor"
@@ -313,16 +405,17 @@ def _interpret_kappa(est: float, lo: float, hi: float, n: int, label: str) -> tu
313
405
  band = "substantial"
314
406
  else:
315
407
  band = "almost perfect"
408
+ band_phrase = f"{band} agreement"
316
409
  interpretation = f"{band} agreement (Landis & Koch, 1977 benchmarks)"
317
410
  example = (
318
411
  f'"{label} = {est:.2f}, 95% CI [{lo:.2f}, {hi:.2f}] (n={n}), indicating '
319
412
  f'{band} agreement between the LLM judge and human raters, per the Landis '
320
413
  f'& Koch (1977) benchmarks."'
321
414
  )
322
- return interpretation, example
415
+ return band_phrase, interpretation, example
323
416
 
324
417
 
325
- def _interpret_corr(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str]:
418
+ def _interpret_corr(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str, str]:
326
419
  """Cohen (1988) conventions for correlation-coefficient magnitude."""
327
420
  a = abs(est)
328
421
  if a < 0.10:
@@ -334,16 +427,37 @@ def _interpret_corr(est: float, lo: float, hi: float, n: int, label: str) -> tup
334
427
  else:
335
428
  band = "large"
336
429
  direction = "positive" if est >= 0 else "negative"
337
- interpretation = f"{band} {direction} correlation (Cohen, 1988 conventions)"
430
+ band_phrase = f"{band} {direction} correlation"
431
+ interpretation = f"{band_phrase} (Cohen, 1988 conventions)"
338
432
  example = (
339
433
  f'"{label} = {est:.2f}, 95% CI [{lo:.2f}, {hi:.2f}] (n={n}), a {band} '
340
434
  f'{direction} correlation between the LLM judge and human scores (Cohen, '
341
435
  f'1988 conventions)."'
342
436
  )
343
- return interpretation, example
437
+ return band_phrase, interpretation, example
438
+
439
+
440
+ def _interpret_icc(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str, str]:
441
+ """Koo & Li (2016) benchmarks for ICC magnitude."""
442
+ if est < 0.50:
443
+ band = "poor"
444
+ elif est < 0.75:
445
+ band = "moderate"
446
+ elif est < 0.90:
447
+ band = "good"
448
+ else:
449
+ band = "excellent"
450
+ band_phrase = f"{band} absolute agreement"
451
+ interpretation = f"{band_phrase} (Koo & Li, 2016 benchmarks)"
452
+ example = (
453
+ f'"{label} = {est:.2f}, 95% CI [{lo:.2f}, {hi:.2f}] (n={n}), indicating '
454
+ f'{band} absolute agreement between the LLM judge and human raters, per '
455
+ f'Koo & Li (2016) benchmarks."'
456
+ )
457
+ return band_phrase, interpretation, example
344
458
 
345
459
 
346
- def _interpret_pct_agreement(est: float, lo: float, hi: float, n: int, label: str) -> tuple[str, str]:
460
+ def _interpret_pct_agreement(est: float, lo: float, hi: float, n: int, label: str) -> tuple[Optional[str], str, str]:
347
461
  interpretation = (
348
462
  "no universally-agreed threshold exists for raw percent agreement — read it "
349
463
  "alongside Cohen's κ, since it does not correct for chance and can look high "
@@ -353,7 +467,7 @@ def _interpret_pct_agreement(est: float, lo: float, hi: float, n: int, label: st
353
467
  f'"the LLM judge matched human labels on {est * 100:.1f}% of items, 95% CI '
354
468
  f'[{lo * 100:.1f}%, {hi * 100:.1f}%] (n={n})."'
355
469
  )
356
- return interpretation, example
470
+ return None, interpretation, example
357
471
 
358
472
 
359
473
  def _bootstrap_ci_2(
@@ -376,6 +490,134 @@ def _bootstrap_ci_2(
376
490
  return obs, lo, hi
377
491
 
378
492
 
493
+ def _icc_21(a: np.ndarray, b: np.ndarray) -> float:
494
+ """Shrout & Fleiss (1979) ICC(2,1): two-way random effects, single rater,
495
+ absolute agreement, for exactly two raters (``a``, ``b``).
496
+
497
+ Unlike Pearson/Spearman r or weighted kappa's category-index distance,
498
+ this is sensitive to a systematic offset or scale mismatch between the
499
+ two raters — it measures whether they land on the same absolute values,
500
+ not just whether they move together.
501
+ """
502
+ n = len(a)
503
+ data = np.column_stack([a, b]).astype(float)
504
+ k = 2
505
+ grand_mean = data.mean()
506
+ row_means = data.mean(axis=1)
507
+ col_means = data.mean(axis=0)
508
+
509
+ df_row = max(n - 1, 1)
510
+ SSR = k * np.sum((row_means - grand_mean) ** 2)
511
+ SSC = n * np.sum((col_means - grand_mean) ** 2) # (k-1) == 1
512
+ SST = np.sum((data - grand_mean) ** 2)
513
+ SSE = SST - SSR - SSC
514
+
515
+ MSR = SSR / df_row
516
+ MSC = SSC
517
+ MSE = SSE / df_row # (n-1)(k-1) == n-1
518
+
519
+ denom = MSR + MSE + 2.0 * (MSC - MSE) / n
520
+ if denom <= 1e-12:
521
+ return 1.0
522
+ return float((MSR - MSE) / denom)
523
+
524
+
525
+ def _bootstrap_ci_gap(
526
+ fn_corr,
527
+ fn_icc,
528
+ a: np.ndarray,
529
+ b: np.ndarray,
530
+ *,
531
+ n_boot: int = 2000,
532
+ alpha: float = 0.05,
533
+ rng: np.random.Generator,
534
+ ) -> tuple[float, float, float]:
535
+ """Paired bootstrap CI for (correlation-type metric − ICC(2,1)).
536
+
537
+ Resamples items once per draw and evaluates both statistics on the same
538
+ resample, so the CI reflects the sampling distribution of the *gap*
539
+ itself, not the (looser, more conservative) union of two independently
540
+ bootstrapped CIs.
541
+ """
542
+ n = len(a)
543
+ obs = float(fn_corr(a, b) - fn_icc(a, b))
544
+ boot = np.empty(n_boot)
545
+ for i in range(n_boot):
546
+ idx = rng.integers(0, n, size=n)
547
+ boot[i] = fn_corr(a[idx], b[idx]) - fn_icc(a[idx], b[idx])
548
+ lo = float(np.percentile(boot, 100.0 * alpha / 2))
549
+ hi = float(np.percentile(boot, 100.0 * (1.0 - alpha / 2)))
550
+ return obs, lo, hi
551
+
552
+
553
+ def _build_bias_check(
554
+ corr_label: str,
555
+ corr_est: float,
556
+ icc_est: float,
557
+ gap_est: float,
558
+ gap_lo: float,
559
+ gap_hi: float,
560
+ ) -> dict:
561
+ """Package the correlation-vs-ICC(2,1) discrepancy check into a dict with
562
+ the same shape as the representativeness checks, so it can be rendered
563
+ with the same ``_print_check`` helper in ``AlignmentResult.summary()``.
564
+ """
565
+ flagged = gap_lo > 0.0
566
+ if flagged:
567
+ message = (
568
+ f"possible judge bias: {corr_label} = {corr_est:.3f} but "
569
+ f"ICC(2,1) = {icc_est:.3f} (gap = {gap_est:.3f}, 95% CI "
570
+ f"[{gap_lo:.3f}, {gap_hi:.3f}], excludes 0)"
571
+ )
572
+ else:
573
+ message = (
574
+ f"no evidence of scale bias: {corr_label} = {corr_est:.3f} and "
575
+ f"ICC(2,1) = {icc_est:.3f} are consistent (gap 95% CI "
576
+ f"[{gap_lo:.3f}, {gap_hi:.3f}] includes 0)"
577
+ )
578
+ what = (
579
+ f"Compares {corr_label}, which is insensitive to a systematic offset "
580
+ "or scale difference between judge and human scores, against "
581
+ "ICC(2,1), which penalizes exactly that. A paired bootstrap CI is "
582
+ "used so the comparison reflects the sampling distribution of the "
583
+ "gap itself rather than the union of two separate CIs."
584
+ )
585
+ why = (
586
+ "A correlation-type metric can look strong even when a judge is "
587
+ "systematically shifted or compressed relative to human scores, "
588
+ "since it only requires the two to move together. This check exists "
589
+ "to catch that failure mode before it's mistaken for genuine "
590
+ "agreement."
591
+ )
592
+ if flagged:
593
+ interpretation = (
594
+ "the judge tracks human relative ordering but disagrees on "
595
+ "absolute scale — treat raw judge scores as biased; consider "
596
+ "using the Bayesian calibration model fit by validate_alignment "
597
+ "(e.g. via compare(alignment=...)) to correct for it before "
598
+ "drawing conclusions from raw judge scores"
599
+ )
600
+ else:
601
+ interpretation = (
602
+ "the correlation and absolute-agreement metrics tell a "
603
+ "consistent story — no sign that the judge's ranking ability is "
604
+ "masking a scale or offset problem"
605
+ )
606
+ return {
607
+ "passed": not flagged,
608
+ "message": message,
609
+ "corr_label": corr_label,
610
+ "corr_estimate": corr_est,
611
+ "icc_estimate": icc_est,
612
+ "gap": gap_est,
613
+ "gap_ci_low": gap_lo,
614
+ "gap_ci_high": gap_hi,
615
+ "what": what,
616
+ "why": why,
617
+ "interpretation": interpretation,
618
+ }
619
+
620
+
379
621
  def _compute_alignment_metrics(
380
622
  llm: np.ndarray,
381
623
  human: np.ndarray,
@@ -399,10 +641,11 @@ def _compute_alignment_metrics(
399
641
  return (p_o - p_e) / (1.0 - p_e) if p_e < 1.0 else 1.0
400
642
 
401
643
  est, lo, hi = _bootstrap_ci_2(agree, llm, human, alpha=alpha, rng=rng)
402
- interp, example = _interpret_pct_agreement(est, lo, hi, len(llm), "Percent agreement")
644
+ band, interp, example = _interpret_pct_agreement(est, lo, hi, len(llm), "Percent agreement")
403
645
  metrics["percent_agreement"] = {
404
646
  "estimate": est, "ci_low": lo, "ci_high": hi,
405
647
  "label": "Percent agreement",
648
+ "band": band,
406
649
  "what": (
407
650
  "The fraction of items where the LLM judge's label exactly matches "
408
651
  "the human label."
@@ -415,10 +658,11 @@ def _compute_alignment_metrics(
415
658
  "example": example,
416
659
  }
417
660
  est, lo, hi = _bootstrap_ci_2(kappa, llm, human, alpha=alpha, rng=rng)
418
- interp, example = _interpret_kappa(est, lo, hi, len(llm), "Cohen's κ")
661
+ band, interp, example = _interpret_kappa(est, lo, hi, len(llm), "Cohen's κ")
419
662
  metrics["cohens_kappa"] = {
420
663
  "estimate": est, "ci_low": lo, "ci_high": hi,
421
664
  "label": "Cohen's κ",
665
+ "band": band,
422
666
  "what": (
423
667
  "Percent agreement adjusted for the rate of agreement expected from "
424
668
  "two raters guessing at random, given the observed marginal label "
@@ -457,10 +701,11 @@ def _compute_alignment_metrics(
457
701
 
458
702
  if k >= 2:
459
703
  est, lo, hi = _bootstrap_ci_2(wk, llm, human, alpha=alpha, rng=rng)
460
- interp, example = _interpret_kappa(est, lo, hi, len(llm), "Weighted Cohen's κ")
704
+ band, interp, example = _interpret_kappa(est, lo, hi, len(llm), "Weighted Cohen's κ")
461
705
  metrics["weighted_kappa"] = {
462
706
  "estimate": est, "ci_low": lo, "ci_high": hi,
463
707
  "label": "Weighted Cohen's κ",
708
+ "band": band,
464
709
  "what": (
465
710
  "Cohen's κ extended so that disagreements receive larger penalties "
466
711
  "as ratings become farther apart on the ordinal scale (Cohen, 1968)."
@@ -475,10 +720,11 @@ def _compute_alignment_metrics(
475
720
  "example": example,
476
721
  }
477
722
  est, lo, hi = _bootstrap_ci_2(sp, llm, human, alpha=alpha, rng=rng)
478
- interp, example = _interpret_corr(est, lo, hi, len(llm), "Spearman r")
723
+ band, interp, example = _interpret_corr(est, lo, hi, len(llm), "Spearman r")
479
724
  metrics["spearman_r"] = {
480
725
  "estimate": est, "ci_low": lo, "ci_high": hi,
481
726
  "label": "Spearman r",
727
+ "band": band,
482
728
  "what": (
483
729
  "Rank correlation between judge and human scores — checks whether "
484
730
  "higher judge scores correspond to higher human scores, without "
@@ -493,6 +739,36 @@ def _compute_alignment_metrics(
493
739
  "example": example,
494
740
  }
495
741
 
742
+ if k >= 2:
743
+ icc_est, icc_lo, icc_hi = _bootstrap_ci_2(_icc_21, llm, human, alpha=alpha, rng=rng)
744
+ band, interp, example = _interpret_icc(icc_est, icc_lo, icc_hi, len(llm), "ICC(2,1)")
745
+ metrics["icc_21"] = {
746
+ "estimate": icc_est, "ci_low": icc_lo, "ci_high": icc_hi,
747
+ "label": "ICC(2,1)",
748
+ "band": band,
749
+ "what": (
750
+ "Two-way random-effects intraclass correlation for absolute "
751
+ "agreement (Shrout & Fleiss, 1979): unlike weighted κ's "
752
+ "category-index distance or Spearman r's rank comparison, it "
753
+ "is sensitive to a systematic offset between the judge and "
754
+ "human scale, not just whether they move together."
755
+ ),
756
+ "why": (
757
+ "Computed alongside weighted κ to check for absolute-scale "
758
+ "bias: a judge that ranks items correctly but is shifted or "
759
+ "compressed relative to human scores can still get a high "
760
+ "weighted κ / Spearman r while scoring poorly here."
761
+ ),
762
+ "interpretation": interp,
763
+ "example": example,
764
+ }
765
+
766
+ gap_est, gap_lo, gap_hi = _bootstrap_ci_gap(wk, _icc_21, llm, human, alpha=alpha, rng=rng)
767
+ metrics["_bias_check"] = _build_bias_check(
768
+ "Weighted Cohen's κ", metrics["weighted_kappa"]["estimate"],
769
+ icc_est, gap_est, gap_lo, gap_hi,
770
+ )
771
+
496
772
  else: # continuous / grade
497
773
  def pe(a, b):
498
774
  r, _ = pearsonr(a, b)
@@ -503,10 +779,11 @@ def _compute_alignment_metrics(
503
779
  return float(r)
504
780
 
505
781
  est, lo, hi = _bootstrap_ci_2(pe, llm, human, alpha=alpha, rng=rng)
506
- interp, example = _interpret_corr(est, lo, hi, len(llm), "Pearson r")
782
+ band, interp, example = _interpret_corr(est, lo, hi, len(llm), "Pearson r")
507
783
  metrics["pearson_r"] = {
508
784
  "estimate": est, "ci_low": lo, "ci_high": hi,
509
785
  "label": "Pearson r",
786
+ "band": band,
510
787
  "what": "Linear correlation coefficient between judge and human scores.",
511
788
  "why": (
512
789
  "Your judge produces continuous/numeric scores, so a correlation "
@@ -516,10 +793,11 @@ def _compute_alignment_metrics(
516
793
  "example": example,
517
794
  }
518
795
  est, lo, hi = _bootstrap_ci_2(sp, llm, human, alpha=alpha, rng=rng)
519
- interp, example = _interpret_corr(est, lo, hi, len(llm), "Spearman r")
796
+ band, interp, example = _interpret_corr(est, lo, hi, len(llm), "Spearman r")
520
797
  metrics["spearman_r"] = {
521
798
  "estimate": est, "ci_low": lo, "ci_high": hi,
522
799
  "label": "Spearman r",
800
+ "band": band,
523
801
  "what": "Rank correlation between judge and human scores.",
524
802
  "why": (
525
803
  "Reported alongside Pearson r to check whether agreement holds even "
@@ -530,6 +808,35 @@ def _compute_alignment_metrics(
530
808
  "example": example,
531
809
  }
532
810
 
811
+ icc_est, icc_lo, icc_hi = _bootstrap_ci_2(_icc_21, llm, human, alpha=alpha, rng=rng)
812
+ band, interp, example = _interpret_icc(icc_est, icc_lo, icc_hi, len(llm), "ICC(2,1)")
813
+ metrics["icc_21"] = {
814
+ "estimate": icc_est, "ci_low": icc_lo, "ci_high": icc_hi,
815
+ "label": "ICC(2,1)",
816
+ "band": band,
817
+ "what": (
818
+ "Two-way random-effects intraclass correlation for absolute "
819
+ "agreement (Shrout & Fleiss, 1979): unlike Pearson/Spearman r, "
820
+ "which are invariant to any linear rescaling of one variable, "
821
+ "this is sensitive to a systematic offset or scale mismatch "
822
+ "between the judge and human scale."
823
+ ),
824
+ "why": (
825
+ "Computed alongside Pearson r to check for absolute-scale bias: "
826
+ "a judge that is consistently shifted or compressed relative to "
827
+ "human scores can still score a perfect Pearson r while "
828
+ "disagreeing badly here."
829
+ ),
830
+ "interpretation": interp,
831
+ "example": example,
832
+ }
833
+
834
+ gap_est, gap_lo, gap_hi = _bootstrap_ci_gap(pe, _icc_21, llm, human, alpha=alpha, rng=rng)
835
+ metrics["_bias_check"] = _build_bias_check(
836
+ "Pearson r", metrics["pearson_r"]["estimate"],
837
+ icc_est, gap_est, gap_lo, gap_hi,
838
+ )
839
+
533
840
  return metrics
534
841
 
535
842
 
@@ -756,6 +1063,7 @@ def validate_alignment(
756
1063
  alignment_metrics = _compute_alignment_metrics(
757
1064
  llm_aligned, human_aligned, score_type, alpha=alpha, rng=rng
758
1065
  )
1066
+ bias_check = alignment_metrics.pop("_bias_check", None)
759
1067
 
760
1068
  # Representativeness: score distribution
761
1069
  rep: dict = {}
@@ -801,4 +1109,5 @@ def validate_alignment(
801
1109
  calibration=calibration,
802
1110
  alignment_metrics=alignment_metrics,
803
1111
  representativeness=rep,
1112
+ bias_check=bias_check,
804
1113
  )
@@ -719,8 +719,8 @@ def _bridge_to_io(
719
719
  # PPI alignment correction
720
720
  # ─────────────────────────────────────────────────────────────────────────────
721
721
 
722
- _PPI_PAIRWISE_SUPPORTED = ("tango", "t_interval", "bootstrap", "wilcoxon", "mannwhitney", "bootstrap_t", "bayes_bootstrap")
723
- _PPI_ROBUSTNESS_SUPPORTED = ("wilson", "bootstrap", "bootstrap_t")
722
+ _PPI_PAIRWISE_SUPPORTED = ("tango", "t_interval", "bootstrap", "wilcoxon", "mannwhitney", "bootstrap_t", "bayes_bootstrap", "ppi_t_interval", "ppi_logit_t")
723
+ _PPI_ROBUSTNESS_SUPPORTED = ("wilson", "bootstrap", "bootstrap_t", "ppi_t_interval", "ppi_logit_t")
724
724
 
725
725
 
726
726
  def _ppi_pairwise_dispatch(method: str, a, b, a_lab, b_lab, alpha: float, n_boot: int, rng):
@@ -729,10 +729,18 @@ def _ppi_pairwise_dispatch(method: str, a, b, a_lab, b_lab, alpha: float, n_boot
729
729
  Only methods with a validated PPI-corrected counterpart (see
730
730
  ``evalstats.tests``'s ``_ppi_paired_*``/``_ppi_two_sample`` functions,
731
731
  calibrated via ``simulations/harness --mode ppi``) are supported here.
732
+
733
+ "ppi_t_interval"/"ppi_logit_t" are DISTINCT method strings from the
734
+ existing bare "t_interval" (below) -- that one already maps to
735
+ ``_ppi_paired_arrays(..., np.mean, rectifier_func=np.mean)``, the
736
+ generic PPI-mean-diff bootstrap routine, not the closed-form analytic
737
+ construction these two use. Reusing "t_interval"/"logit_t" here would
738
+ silently collide with that existing mapping.
732
739
  """
733
740
  from evalstats.tests import (
734
741
  _ppi_paired_tango, _ppi_paired_bootstrap_t, _ppi_paired_bayes_bootstrap,
735
- _ppi_paired_arrays, _ppi_two_sample_midrank_corrected,
742
+ _ppi_paired_arrays, _ppi_two_sample, _p_x_gt_y_midrank,
743
+ _ppi_paired_t_interval, _ppi_paired_logit_t,
736
744
  )
737
745
  if method == "tango":
738
746
  return _ppi_paired_tango(a, b, a_lab, b_lab, alpha)
@@ -740,17 +748,30 @@ def _ppi_pairwise_dispatch(method: str, a, b, a_lab, b_lab, alpha: float, n_boot
740
748
  return _ppi_paired_bootstrap_t(a, b, a_lab, b_lab, alpha, n_boot, rng)
741
749
  if method == "bayes_bootstrap":
742
750
  return _ppi_paired_bayes_bootstrap(a, b, a_lab, b_lab, alpha, n_boot, rng)
751
+ if method == "ppi_t_interval":
752
+ return _ppi_paired_t_interval(a, b, a_lab, b_lab, alpha)
753
+ if method == "ppi_logit_t":
754
+ # lo/hi default (0.0, 1.0): this dispatch path has no score_range
755
+ # concept (see _run_alignment_ppi's is_bounded_01_scores check --
756
+ # "bounded_01" always means raw scores are literally in [0, 1] here).
757
+ return _ppi_paired_logit_t(a, b, a_lab, b_lab, alpha)
743
758
  if method in ("t_interval", "bootstrap"):
744
759
  return _ppi_paired_arrays(a, b, a_lab, b_lab, np.mean, alpha, n_boot, rng, rectifier_func=np.mean)
745
760
  if method == "wilcoxon":
746
761
  return _ppi_paired_arrays(a, b, a_lab, b_lab, np.median, alpha, n_boot, rng, rectifier_func=np.mean)
747
762
  if method == "mannwhitney":
748
- # Per-group, per-score-bin local rectifier -- NOT the single-global-
749
- # rectifier _ppi_two_sample -- see _ppi_two_sample_midrank_corrected's
750
- # docstring for why the naive global rectifier badly miscalibrates
751
- # this rank estimand under score-correlated labeling + real judge
752
- # bias (validated via simulations/harness --mode ppi).
753
- return _ppi_two_sample_midrank_corrected(a, b, a_lab, b_lab, alpha, n_boot, rng)
763
+ # Matches evalstats.tests.mannwhitney's method="global" default
764
+ # (reinstated 2026-08-02, a few hours after "ridge" -- see
765
+ # mannwhitney's `method` docstring's "REVERTED TO 'global'" note
766
+ # for the full six-default history). NOT a finding against
767
+ # "ridge" -- it remains the best-validated option by every number
768
+ # gathered -- reverted because the harness's --official-tests
769
+ # suite has always tested plain "global" under the name "mwu"
770
+ # (hardcoded, doesn't track mannwhitney()'s actual default), so
771
+ # production stays aligned with what's actually been exercised by
772
+ # that sanctioned pipeline until "ridge" gets its own official
773
+ # pass.
774
+ return _ppi_two_sample(a, b, a_lab, b_lab, lambda xa, ya: _p_x_gt_y_midrank(xa, ya) - 0.5, alpha, n_boot, rng)
754
775
  raise ValueError(
755
776
  f"PPI alignment correction has no validated implementation for pairwise "
756
777
  f"method {method!r}. Supported pairwise methods: "
@@ -917,11 +938,18 @@ def _max_t_from_joint_stats(
917
938
 
918
939
  def _ppi_robustness_dispatch(method: str, a, a_lab, alpha: float, n_boot: int, rng):
919
940
  """Dispatch to the PPI-corrected single-sample implementation of *method*."""
920
- from evalstats.tests import _ppi_single_wilson, _ppi_single_bootstrap_t
941
+ from evalstats.tests import (
942
+ _ppi_single_wilson, _ppi_single_bootstrap_t, _ppi_single_t_interval, _ppi_single_logit_t,
943
+ )
921
944
  if method == "wilson":
922
945
  return _ppi_single_wilson(a, a_lab, alpha)
923
946
  if method == "bootstrap_t":
924
947
  return _ppi_single_bootstrap_t(a, a_lab, alpha, n_boot, rng)
948
+ if method == "ppi_t_interval":
949
+ return _ppi_single_t_interval(a, a_lab, alpha)
950
+ if method == "ppi_logit_t":
951
+ # lo/hi default (0.0, 1.0) -- see _ppi_pairwise_dispatch's matching note.
952
+ return _ppi_single_logit_t(a, a_lab, alpha)
925
953
  if method == "bootstrap":
926
954
  from evalstats.ppi import correct as _ppi_correct
927
955
  mask = ~np.isnan(a_lab)