@tangle-network/agent-eval 0.133.1 → 0.133.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. package/CHANGELOG.md +165 -0
  2. package/dist/{analyze-runs-DZr7JW-m.d.ts → analyze-runs-BClW9OSe.d.ts} +3 -3
  3. package/dist/{analyze-runs-DZr7JW-m.d.ts.map → analyze-runs-BClW9OSe.d.ts.map} +1 -1
  4. package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
  5. package/dist/analyze-runs-qk8op0tN.js.map +1 -0
  6. package/dist/baseline-BaPxoROc.js +149 -0
  7. package/dist/baseline-BaPxoROc.js.map +1 -0
  8. package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
  9. package/dist/baseline-D_fT6277.d.ts.map +1 -0
  10. package/dist/benchmarks/index.d.ts +1 -1
  11. package/dist/benchmarks/index.js +1 -1
  12. package/dist/{benchmarks-BU7P6PCW.js → benchmarks-BP9sgMia.js} +3 -3
  13. package/dist/{benchmarks-BU7P6PCW.js.map → benchmarks-BP9sgMia.js.map} +1 -1
  14. package/dist/builder-eval/index.js +1 -1
  15. package/dist/campaign/index.d.ts +2 -2
  16. package/dist/campaign/index.js +2 -2
  17. package/dist/{campaign-CnzHQndg.js → campaign--V4ffEKR.js} +12 -6
  18. package/dist/{campaign-CnzHQndg.js.map → campaign--V4ffEKR.js.map} +1 -1
  19. package/dist/{client-D4F9hdzR.d.ts → client-Du7B81wW.d.ts} +28 -14
  20. package/dist/client-Du7B81wW.d.ts.map +1 -0
  21. package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
  22. package/dist/client-LIuo-KPv.js.map +1 -0
  23. package/dist/contract/index.d.ts +3 -3
  24. package/dist/contract/index.d.ts.map +1 -1
  25. package/dist/contract/index.js +9 -8
  26. package/dist/contract/index.js.map +1 -1
  27. package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
  28. package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
  29. package/dist/hosted/index.d.ts +1 -1
  30. package/dist/hosted/index.d.ts.map +1 -1
  31. package/dist/hosted/index.js +1 -1
  32. package/dist/index-3cdlURSk.d.ts.map +1 -1
  33. package/dist/{index-Wek5mU0y.d.ts → index-B5MNN1f1.d.ts} +3 -3
  34. package/dist/{index-Wek5mU0y.d.ts.map → index-B5MNN1f1.d.ts.map} +1 -1
  35. package/dist/{index-Ba636PKl.d.ts → index-DOqvIJ8I.d.ts} +27 -10
  36. package/dist/index-DOqvIJ8I.d.ts.map +1 -0
  37. package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
  38. package/dist/index-DSC51roc2.d.ts.map +1 -0
  39. package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
  40. package/dist/index-DuhJaaiH.d.ts.map +1 -0
  41. package/dist/index.d.ts +56 -10
  42. package/dist/index.d.ts.map +1 -1
  43. package/dist/index.js +39 -22
  44. package/dist/index.js.map +1 -1
  45. package/dist/ledger-core/index.d.ts +2 -2
  46. package/dist/ledger-core/index.js +2 -2
  47. package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
  48. package/dist/ledger-core-DAKFKRzi.js.map +1 -0
  49. package/dist/matrix/index.d.ts +1 -1
  50. package/dist/meta-eval/index.d.ts +1 -1
  51. package/dist/meta-eval/index.js +2 -2
  52. package/dist/multishot/index.d.ts +1 -1
  53. package/dist/openapi.json +1 -1
  54. package/dist/{opencode-sqlite-BGrHeDu3.js → opencode-sqlite-8r6WUfHc.js} +2 -3
  55. package/dist/opencode-sqlite-8r6WUfHc.js.map +1 -0
  56. package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
  57. package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
  58. package/dist/pipelines/index.d.ts +1 -1
  59. package/dist/pipelines/index.js +3 -2
  60. package/dist/pipelines/index.js.map +1 -1
  61. package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
  62. package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
  63. package/dist/{release-report-DfmKSIEE.d.ts → release-report-DKBtegGt.d.ts} +2 -2
  64. package/dist/{release-report-DfmKSIEE.d.ts.map → release-report-DKBtegGt.d.ts.map} +1 -1
  65. package/dist/reporting.d.ts +3 -3
  66. package/dist/reporting.js +4 -4
  67. package/dist/{researcher-DMimgHtN.d.ts → researcher-BtD5U1Up.d.ts} +2 -2
  68. package/dist/{researcher-DMimgHtN.d.ts.map → researcher-BtD5U1Up.d.ts.map} +1 -1
  69. package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
  70. package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
  71. package/dist/rl.d.ts +1 -1
  72. package/dist/rl.js +4 -4
  73. package/dist/rollout/index.js +2 -2
  74. package/dist/{rollout-CeTlDrf6.js → rollout-CreDz__7.js} +2 -2
  75. package/dist/{rollout-CeTlDrf6.js.map → rollout-CreDz__7.js.map} +1 -1
  76. package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
  77. package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
  78. package/dist/{skillopt-optimization-method-CAASpcS3.d.ts → skillopt-optimization-method-Dxr8pdZd.d.ts} +12 -7
  79. package/dist/{skillopt-optimization-method-CAASpcS3.d.ts.map → skillopt-optimization-method-Dxr8pdZd.d.ts.map} +1 -1
  80. package/dist/{skillopt-optimization-method-BoIzh7Dl.js → skillopt-optimization-method-vvJ4bMNI.js} +123 -24
  81. package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +1 -0
  82. package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
  83. package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
  84. package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
  85. package/dist/statistics-RwRNu2__.js.map +1 -0
  86. package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
  87. package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
  88. package/dist/{summary-report-DnUcjVpV.d.ts → summary-report-DyOhItws.d.ts} +4 -3
  89. package/dist/summary-report-DyOhItws.d.ts.map +1 -0
  90. package/dist/supervisor-run/index.js +1 -1
  91. package/dist/{supervisor-run-_lnTLM3z.js → supervisor-run-B7lUGoyZ.js} +2 -2
  92. package/dist/{supervisor-run-_lnTLM3z.js.map → supervisor-run-B7lUGoyZ.js.map} +1 -1
  93. package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
  94. package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
  95. package/docs/design/statistics-decisions.md +271 -0
  96. package/docs/design.md +1 -0
  97. package/docs/insight-report.md +1 -1
  98. package/docs/research-report-methodology.md +4 -1
  99. package/package.json +2 -1
  100. package/dist/analyze-runs-B-afTpCv.js.map +0 -1
  101. package/dist/baseline-DcX5hQDv.js.map +0 -1
  102. package/dist/baseline-hG3K85h4.d.ts.map +0 -1
  103. package/dist/client-CYzbdJOZ.js.map +0 -1
  104. package/dist/client-D4F9hdzR.d.ts.map +0 -1
  105. package/dist/index-Ba636PKl.d.ts.map +0 -1
  106. package/dist/index-DSC51roc.d.ts.map +0 -1
  107. package/dist/index-nhIYz9hn.d.ts.map +0 -1
  108. package/dist/ledger-core-CPZfcrC2.js.map +0 -1
  109. package/dist/opencode-sqlite-BGrHeDu3.js.map +0 -1
  110. package/dist/skillopt-optimization-method-BoIzh7Dl.js.map +0 -1
  111. package/dist/statistics-DWM_AyLe.js.map +0 -1
  112. package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
  113. package/dist/summary-report-DnUcjVpV.d.ts.map +0 -1
@@ -1,4 +1,3 @@
1
- import { F as studentTCdf, i as cohensD } from "./statistics-DWM_AyLe.js";
2
1
  import { i as hasCapturedToolArgs, l as toolSpans, n as argHash, r as groupBy } from "./query-Di7eEQ79.js";
3
2
  //#region src/failure-taxonomy.ts
4
3
  const DEFAULT_RULES = [
@@ -366,117 +365,6 @@ async function computeToolUseMetrics(store, runId, options = {}) {
366
365
  };
367
366
  }
368
367
  //#endregion
369
- //#region src/baseline.ts
370
- /**
371
- * Baseline regression detection.
372
- *
373
- * Lifted from ADC baseline.ts. Every promotion-blocking signal boils down
374
- * to: "is this run measurably worse than baseline?" — with enough
375
- * statistical rigor to distinguish noise from drift.
376
- *
377
- * Uses:
378
- * - Welch's t-test (unequal variance) for per-metric mean comparison
379
- * - Cohen's d for effect size magnitude
380
- * - IQR for stability flag (unstable samples can't be trusted for comparisons)
381
- *
382
- * Returns a structured verdict: improved | regressed | stable | unstable.
383
- */
384
- /**
385
- * Compare candidate samples against baseline per metric. Verdict logic:
386
- * - unstable: IQR/|mean| > threshold on either set — not enough signal
387
- * - improved: meaningful effect in the "better" direction AND p < alpha
388
- * - regressed: meaningful effect in the "worse" direction AND p < alpha
389
- * - stable: otherwise (no significant change)
390
- */
391
- function compareToBaseline(samples, options = {}) {
392
- const effectThreshold = options.effectThreshold ?? .5;
393
- const alpha = options.alpha ?? .05;
394
- const cvThreshold = options.unstableCvThreshold ?? .3;
395
- const metrics = samples.map((s) => {
396
- if (s.baseline.length < 2 || s.candidate.length < 2) throw new Error(`compareToBaseline: need ≥2 samples per side for "${s.metric}"`);
397
- const bMean = mean(s.baseline);
398
- const cMean = mean(s.candidate);
399
- const delta = cMean - bMean;
400
- const d = cohensD(s.baseline, s.candidate);
401
- const { t, df, p } = welchsTTest(s.baseline, s.candidate);
402
- const baselineIqr = iqr(s.baseline);
403
- const candidateIqr = iqr(s.candidate);
404
- const baselineStable = baselineIqr / Math.max(Math.abs(bMean), 1e-9) <= cvThreshold;
405
- const candidateStable = candidateIqr / Math.max(Math.abs(cMean), 1e-9) <= cvThreshold;
406
- const stable = baselineStable && candidateStable;
407
- const reportedIqr = Math.max(baselineIqr, candidateIqr);
408
- let verdict;
409
- if (!stable) verdict = "unstable";
410
- else if (p < alpha && Math.abs(d) >= effectThreshold) verdict = (s.higherIsBetter ? delta > 0 : delta < 0) ? "improved" : "regressed";
411
- else verdict = "stable";
412
- return {
413
- metric: s.metric,
414
- baselineMean: bMean,
415
- candidateMean: cMean,
416
- delta,
417
- cohensD: d,
418
- welchT: t,
419
- welchDf: df,
420
- welchP: p,
421
- stable,
422
- iqr: reportedIqr,
423
- verdict
424
- };
425
- });
426
- return {
427
- metrics,
428
- hasRegression: metrics.some((m) => m.verdict === "regressed"),
429
- hasUnstable: metrics.some((m) => m.verdict === "unstable")
430
- };
431
- }
432
- function mean(xs) {
433
- return xs.reduce((a, b) => a + b, 0) / xs.length;
434
- }
435
- /** Inter-quartile range; 0 when the sample has no spread. */
436
- function iqr(xs) {
437
- if (xs.length === 0) return 0;
438
- const sorted = [...xs].sort((a, b) => a - b);
439
- const q = (p) => {
440
- const idx = p * (sorted.length - 1);
441
- const lo = Math.floor(idx);
442
- const hi = Math.ceil(idx);
443
- return sorted[lo] + (sorted[hi] - sorted[lo]) * (idx - lo);
444
- };
445
- return q(.75) - q(.25);
446
- }
447
- /**
448
- * Welch's t-test — unequal-variance two-sample t. Uses the same Student-t
449
- * CDF as `pairedTTest` (via incomplete beta); falls back to normal tail
450
- * when df is large.
451
- */
452
- function welchsTTest(a, b) {
453
- if (a.length < 2 || b.length < 2) return {
454
- t: 0,
455
- df: 0,
456
- p: 1
457
- };
458
- const mA = mean(a);
459
- const mB = mean(b);
460
- const vA = variance(a, mA);
461
- const vB = variance(b, mB);
462
- const seSquared = vA / a.length + vB / b.length;
463
- if (seSquared === 0) return {
464
- t: mA === mB ? 0 : Infinity,
465
- df: 0,
466
- p: mA === mB ? 1 : 0
467
- };
468
- const t = (mB - mA) / Math.sqrt(seSquared);
469
- const df = seSquared * seSquared / ((vA / a.length) ** 2 / (a.length - 1) + (vB / b.length) ** 2 / (b.length - 1));
470
- return {
471
- t,
472
- df,
473
- p: 2 * (1 - studentTCdf(Math.abs(t), df))
474
- };
475
- }
476
- function variance(xs, m) {
477
- return xs.reduce((acc, x) => acc + (x - m) ** 2, 0) / (xs.length - 1);
478
- }
479
- //#endregion
480
- export { DEFAULT_RULES as a, computeToolUseMetrics as i, iqr as n, classifyFailure as o, welchsTTest as r, compareToBaseline as t };
368
+ export { DEFAULT_RULES as n, classifyFailure as r, computeToolUseMetrics as t };
481
369
 
482
- //# sourceMappingURL=baseline-DcX5hQDv.js.map
370
+ //# sourceMappingURL=tool-use-metrics-DEGMKycK.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"tool-use-metrics-DEGMKycK.js","names":[],"sources":["../src/failure-taxonomy.ts","../src/tool-use-metrics.ts"],"sourcesContent":["/**\n * Failure taxonomy — canonical classes + a default classifier.\n *\n * Every failed run should end up in a named class. The classifier here\n * is rule-based (fast, deterministic); an LLM fallback can be added by\n * the consumer for novel cases and trained into the rule base over time.\n *\n * Consumers call `classifyFailure(run, spans, events)` and persist the\n * returned class as `Run.outcome.failureClass`.\n */\n\nimport type { FailureClass, Run, Span, TraceEvent } from './trace/schema'\nimport { FAILURE_CLASSES } from './trace/schema'\n\nexport { FAILURE_CLASSES, type FailureClass }\n\nexport interface FailureContext {\n run: Run\n spans: Span[]\n events: TraceEvent[]\n}\n\nexport interface FailureClassification {\n failureClass: FailureClass\n reason: string\n triggerSpanId?: string\n triggerEventId?: string\n}\n\n/** Ordered rules — first match wins. */\nexport interface FailureRule {\n id: string\n match: (ctx: FailureContext) => {\n failureClass: FailureClass\n reason: string\n triggerSpanId?: string\n triggerEventId?: string\n } | null\n}\n\nexport const DEFAULT_RULES: FailureRule[] = [\n // Outcome already named? Respect it.\n {\n id: 'explicit-outcome',\n match: ({ run }) => {\n const fc = run.outcome?.failureClass\n if (fc && fc !== 'unknown')\n return { failureClass: fc, reason: 'outcome.failureClass set explicitly' }\n return null\n },\n },\n {\n id: 'knowledge-readiness-blocked',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'readiness_scored' &&\n e.payload.passed === false,\n )\n return event\n ? {\n failureClass: 'knowledge_readiness_blocked',\n reason: 'knowledge readiness report blocked execution',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'bad-integration-manifest',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_manifest_validated' && e.payload.valid === false) ||\n (e.payload.kind === 'integration_invoke_failed' &&\n e.payload.code === 'manifest_invalid')),\n )\n return event\n ? {\n failureClass: 'bad_integration_manifest',\n reason: 'integration manifest validation failed before launch',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-integration-connection',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_manifest_resolved' &&\n hasResolutionStatus(e.payload, 'missing_connection'),\n )\n return event\n ? {\n failureClass: 'missing_integration_connection',\n reason: 'required integration connection was missing',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-integration-scope',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_manifest_resolved' && hasMissingScopes(e.payload)) ||\n (e.payload.kind === 'integration_invoke_failed' && e.payload.code === 'scope_denied')),\n )\n return event\n ? {\n failureClass: 'missing_integration_scope',\n reason: 'integration grant or connection lacks required scopes',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-approval-required',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_invoke' && e.payload.status === 'approval_required') ||\n (e.payload.kind === 'integration_invoke_failed' &&\n e.payload.code === 'approval_required') ||\n e.payload.kind === 'integration_approval_required'),\n )\n return event\n ? {\n failureClass: 'integration_approval_required',\n reason: 'integration write paused for user approval',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-auth-expired',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n (e.payload.code === 'auth_expired' ||\n e.payload.code === 'connection_not_active' ||\n e.payload.code === 'capability_expired' ||\n e.payload.status === 'expired'),\n )\n return event\n ? {\n failureClass: 'integration_auth_expired',\n reason: 'integration connection or capability expired',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'unsafe-integration-write-denied',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n (e.payload.code === 'unsafe_write_denied' ||\n e.payload.code === 'policy_denied' ||\n e.payload.code === 'action_denied'),\n )\n return event\n ? {\n failureClass: 'unsafe_integration_write_denied',\n reason: 'integration write was denied by policy or capability scope',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-provider-failure',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n ![\n 'scope_denied',\n 'approval_required',\n 'auth_expired',\n 'connection_not_active',\n 'capability_expired',\n 'unsafe_write_denied',\n 'policy_denied',\n 'action_denied',\n 'manifest_invalid',\n ].includes(String(e.payload.code)),\n )\n return event\n ? {\n failureClass: 'integration_provider_failure',\n reason: 'integration provider invocation failed',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-credentials',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.category === 'credential_or_secret',\n )\n return event\n ? {\n failureClass: 'missing_credentials',\n reason: 'required credential or secret was missing',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'bad-retrieval',\n match: ({ run, spans }) => {\n if (run.outcome?.pass !== false) return null\n const retrieval = spans.find(\n (s) =>\n s.kind === 'retrieval' && (s.hits.length === 0 || s.hits.every((hit) => hit.score <= 0)),\n )\n return retrieval\n ? {\n failureClass: 'bad_retrieval',\n reason: 'retrieval returned no useful hits for a failed run',\n triggerSpanId: retrieval.spanId,\n }\n : null\n },\n },\n {\n id: 'insufficient-evidence',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.reason === 'insufficient_evidence',\n )\n return event\n ? {\n failureClass: 'insufficient_evidence',\n reason: 'task proceeded with insufficient supporting evidence',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'contradictory-evidence',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.reason === 'contradictory_evidence',\n )\n return event\n ? {\n failureClass: 'contradictory_evidence',\n reason: 'supporting evidence contradicted itself',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n // Budget breach events\n {\n id: 'budget-breach',\n match: ({ events }) => {\n const breach = events.find((e) => e.kind === 'budget_breach')\n return breach\n ? {\n failureClass: 'budget_exceeded',\n reason: `budget breached on ${breach.payload.dimension ?? 'unknown dimension'}`,\n triggerEventId: breach.eventId,\n }\n : null\n },\n },\n // Policy violations\n {\n id: 'policy-violation',\n match: ({ events }) => {\n const e = events.find((x) => x.kind === 'policy_violation')\n return e\n ? {\n failureClass: 'policy_violation',\n reason: 'policy_violation event emitted',\n triggerEventId: e.eventId,\n }\n : null\n },\n },\n // Sandbox non-zero exit code\n {\n id: 'sandbox-failure',\n match: ({ spans }) => {\n const s = spans.find(\n (x) => x.kind === 'sandbox' && typeof x.exitCode === 'number' && x.exitCode !== 0,\n )\n if (!s) return null\n return {\n failureClass: 'sandbox_failure',\n reason: `sandbox exited ${(s as Extract<Span, { kind: 'sandbox' }>).exitCode}`,\n triggerSpanId: s.spanId,\n }\n },\n },\n // Timeout: run aborted by external signal\n {\n id: 'timeout',\n match: ({ run, events }) => {\n if (run.status !== 'aborted') return null\n const hasTimeout = events.some(\n (e) =>\n e.kind === 'error' &&\n String(e.payload.reason ?? '')\n .toLowerCase()\n .includes('timeout'),\n )\n const note = (run.outcome?.notes ?? '').toLowerCase()\n if (hasTimeout || note.includes('timeout') || note.includes('deadline')) {\n return { failureClass: 'timeout', reason: 'timeout signal observed' }\n }\n return null\n },\n },\n // Tool recovery failure: many consecutive tool errors on the same tool\n {\n id: 'tool-recovery-failure',\n match: ({ spans }) => {\n const tools = spans.filter((s) => s.kind === 'tool')\n const byTool = new Map<string, Span[]>()\n for (const t of tools) {\n const name = (t as Extract<Span, { kind: 'tool' }>).toolName\n const arr = byTool.get(name) ?? []\n arr.push(t)\n byTool.set(name, arr)\n }\n for (const [name, arr] of byTool) {\n const errs = arr.filter((s) => s.status === 'error')\n if (errs.length >= 3 && errs.length === arr.length) {\n return {\n failureClass: 'tool_recovery_failure',\n reason: `${errs.length} consecutive errors on tool \"${name}\"`,\n triggerSpanId: errs[errs.length - 1]!.spanId,\n }\n }\n }\n return null\n },\n },\n // Tool selection error: the run failed and agent called zero tools despite having them\n {\n id: 'tool-selection-error',\n match: ({ run, spans }) => {\n if (run.outcome?.pass !== false) return null\n const hasToolsAvailable = spans.some(\n (s) =>\n s.kind === 'agent' &&\n (s.attributes?.toolsAvailable as number | undefined) !== undefined &&\n (s.attributes?.toolsAvailable as number) > 0,\n )\n const tools = spans.filter((s) => s.kind === 'tool')\n if (hasToolsAvailable && tools.length === 0) {\n return {\n failureClass: 'tool_selection_error',\n reason: 'tools were available but none were called',\n }\n }\n return null\n },\n },\n // Format drift: scored by a judge with dimension='format' below threshold\n {\n id: 'format-drift',\n match: ({ spans }) => {\n const judge = spans.find(\n (s) =>\n s.kind === 'judge' &&\n (s as Extract<Span, { kind: 'judge' }>).dimension === 'format' &&\n (s as Extract<Span, { kind: 'judge' }>).score < 0.5,\n )\n return judge\n ? {\n failureClass: 'format_drift',\n reason: 'format judge scored below 0.5',\n triggerSpanId: judge.spanId,\n }\n : null\n },\n },\n]\n\nfunction hasResolutionStatus(payload: Record<string, unknown>, status: string): boolean {\n if (status === 'missing_connection' && stringArray(payload.missingConnections).length > 0)\n return true\n return resolutionItems(payload).some((item) => item.status === status)\n}\n\nfunction hasMissingScopes(payload: Record<string, unknown>): boolean {\n if (stringArray(payload.missingScopes).length > 0) return true\n return resolutionItems(payload).some(\n (item) => Array.isArray(item.missingScopes) && item.missingScopes.length > 0,\n )\n}\n\nfunction resolutionItems(payload: Record<string, unknown>): Array<Record<string, unknown>> {\n return [\n ...records(payload.missing),\n ...records(payload.optionalMissing),\n ...records(payload.ready),\n ]\n}\n\nfunction records(value: unknown): Array<Record<string, unknown>> {\n if (!Array.isArray(value)) return []\n return value.filter(\n (item): item is Record<string, unknown> =>\n Boolean(item) && typeof item === 'object' && !Array.isArray(item),\n )\n}\n\nfunction stringArray(value: unknown): string[] {\n return Array.isArray(value)\n ? value.filter((item): item is string => typeof item === 'string')\n : []\n}\n\n/** Classify the failure mode of a run using an ordered rule list. */\nexport function classifyFailure(\n ctx: FailureContext,\n rules: FailureRule[] = DEFAULT_RULES,\n): FailureClassification {\n if (ctx.run.outcome?.pass !== false && ctx.run.status === 'completed') {\n return { failureClass: 'success', reason: 'run completed with pass=true (or no explicit fail)' }\n }\n for (const rule of rules) {\n const hit = rule.match(ctx)\n if (hit) return hit\n }\n return { failureClass: 'unknown', reason: 'no rule matched; run failed for unclassified reason' }\n}\n","/**\n * Tool-use metrics — derived purely from trace data.\n *\n * No scoring assumptions: consumers supply optional ground-truth tool\n * selections per turn + optional \"information used downstream\" signals.\n * Without those, we still compute descriptive metrics (error rate,\n * retry rate, duplicate-call rate) that are useful on their own.\n */\n\nimport { argHash, groupBy, hasCapturedToolArgs, toolSpans } from './trace/query'\nimport type { Span } from './trace/schema'\nimport type { TraceStore } from './trace/store'\n\nexport interface ToolUseMetrics {\n runId: string\n totalCalls: number\n /** Calls whose arguments were captured and can be compared for duplication. */\n callsWithCapturedArgs: number\n byTool: Record<string, ToolStats>\n errorRate: number\n /** Ratio of captured-argument calls already seen with the same tool name and arguments. */\n duplicateRate: number\n /** Ratio of error calls followed by ≥1 retry on same tool. */\n retryRate: number\n /** Optional: of the calls agent made, fraction the evaluator marked as \"correct selection\". */\n selectionAccuracy?: number\n}\n\nexport interface ToolStats {\n calls: number\n callsWithCapturedArgs: number\n errors: number\n avgLatencyMs: number\n duplicates: number\n}\n\nexport interface ToolUseOptions {\n /** Map of spanId → whether the evaluator judged the tool selection correct. Optional. */\n selectionLabels?: Record<string, boolean>\n}\n\nexport async function computeToolUseMetrics(\n store: TraceStore,\n runId: string,\n options: ToolUseOptions = {},\n): Promise<ToolUseMetrics> {\n const tools = await toolSpans(store, runId)\n if (tools.length === 0) {\n return {\n runId,\n totalCalls: 0,\n callsWithCapturedArgs: 0,\n byTool: {},\n errorRate: 0,\n duplicateRate: 0,\n retryRate: 0,\n }\n }\n\n const byTool: Record<string, ToolStats> = {}\n let totalErrors = 0\n let totalDuplicates = 0\n let callsWithCapturedArgs = 0\n const sortedTools = [...tools].sort((a, b) => a.startedAt - b.startedAt)\n const seenSignatures = new Set<string>()\n\n // duplicate detection + per-tool aggregation\n for (const t of sortedTools) {\n byTool[t.toolName] ??= {\n calls: 0,\n callsWithCapturedArgs: 0,\n errors: 0,\n avgLatencyMs: 0,\n duplicates: 0,\n }\n const stat = byTool[t.toolName]!\n stat.calls += 1\n if (t.status === 'error') {\n stat.errors += 1\n totalErrors += 1\n }\n if (typeof t.latencyMs === 'number') stat.avgLatencyMs += t.latencyMs\n if (hasCapturedToolArgs(t)) {\n callsWithCapturedArgs += 1\n stat.callsWithCapturedArgs += 1\n const sig = `${t.toolName}|${argHash(t.args)}`\n if (seenSignatures.has(sig)) {\n stat.duplicates += 1\n totalDuplicates += 1\n }\n seenSignatures.add(sig)\n }\n }\n\n for (const stat of Object.values(byTool)) {\n stat.avgLatencyMs = stat.calls > 0 ? stat.avgLatencyMs / stat.calls : 0\n }\n\n // retry detection: per-tool chronological adjacency where error → next same-tool call\n let retryOpportunities = 0\n let retriesFollowed = 0\n for (const [, arr] of groupBy(sortedTools, (t) => t.toolName)) {\n for (let i = 0; i < arr.length; i++) {\n if (arr[i]!.status !== 'error') continue\n retryOpportunities += 1\n if (arr[i + 1]) retriesFollowed += 1\n }\n }\n const retryRate = retryOpportunities > 0 ? retriesFollowed / retryOpportunities : 0\n\n let selectionAccuracy: number | undefined\n if (options.selectionLabels) {\n const labeled = sortedTools.filter((t) => t.spanId in options.selectionLabels!)\n if (labeled.length > 0) {\n selectionAccuracy =\n labeled.filter((t) => options.selectionLabels![t.spanId]).length / labeled.length\n }\n }\n\n return {\n runId,\n totalCalls: sortedTools.length,\n callsWithCapturedArgs,\n byTool,\n errorRate: totalErrors / sortedTools.length,\n duplicateRate: callsWithCapturedArgs > 0 ? totalDuplicates / callsWithCapturedArgs : 0,\n retryRate,\n selectionAccuracy,\n }\n}\n\nexport type { Span }\n"],"mappings":";;AAwCA,MAAa,gBAA+B;CAE1C;EACE,IAAI;EACJ,QAAQ,EAAE,UAAU;GAClB,MAAM,KAAK,IAAI,SAAS;GACxB,IAAI,MAAM,OAAO,WACf,OAAO;IAAE,cAAc;IAAI,QAAQ;GAAsC;GAC3E,OAAO;EACT;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,sBACnB,EAAE,QAAQ,WAAW,KACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,oCAAoC,EAAE,QAAQ,UAAU,SAC1E,EAAE,QAAQ,SAAS,+BAClB,EAAE,QAAQ,SAAS,mBAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mCACnB,oBAAoB,EAAE,SAAS,oBAAoB,CACvD;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,mCAAmC,iBAAiB,EAAE,OAAO,KAC/E,EAAE,QAAQ,SAAS,+BAA+B,EAAE,QAAQ,SAAS,eAC5E;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,wBAAwB,EAAE,QAAQ,WAAW,uBAC/D,EAAE,QAAQ,SAAS,+BAClB,EAAE,QAAQ,SAAS,uBACrB,EAAE,QAAQ,SAAS,gCACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,gCAClB,EAAE,QAAQ,SAAS,kBAClB,EAAE,QAAQ,SAAS,2BACnB,EAAE,QAAQ,SAAS,wBACnB,EAAE,QAAQ,WAAW,UAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,gCAClB,EAAE,QAAQ,SAAS,yBAClB,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,SAAS,gBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,+BACnB,CAAC;IACC;IACA;IACA;IACA;IACA;IACA;IACA;IACA;IACA;GACF,CAAC,CAAC,SAAS,OAAO,EAAE,QAAQ,IAAI,CAAC,CACrC;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,aAAa,sBAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,YAAY;GACzB,IAAI,IAAI,SAAS,SAAS,OAAO,OAAO;GACxC,MAAM,YAAY,MAAM,MACrB,MACC,EAAE,SAAS,gBAAgB,EAAE,KAAK,WAAW,KAAK,EAAE,KAAK,OAAO,QAAQ,IAAI,SAAS,CAAC,EAC1F;GACA,OAAO,YACH;IACE,cAAc;IACd,QAAQ;IACR,eAAe,UAAU;GAC3B,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,WAAW,uBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,WAAW,wBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,SAAS,OAAO,MAAM,MAAM,EAAE,SAAS,eAAe;GAC5D,OAAO,SACH;IACE,cAAc;IACd,QAAQ,sBAAsB,OAAO,QAAQ,aAAa;IAC1D,gBAAgB,OAAO;GACzB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,IAAI,OAAO,MAAM,MAAM,EAAE,SAAS,kBAAkB;GAC1D,OAAO,IACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,EAAE;GACpB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,IAAI,MAAM,MACb,MAAM,EAAE,SAAS,aAAa,OAAO,EAAE,aAAa,YAAY,EAAE,aAAa,CAClF;GACA,IAAI,CAAC,GAAG,OAAO;GACf,OAAO;IACL,cAAc;IACd,QAAQ,kBAAmB,EAAyC;IACpE,eAAe,EAAE;GACnB;EACF;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,aAAa;GAC1B,IAAI,IAAI,WAAW,WAAW,OAAO;GACrC,MAAM,aAAa,OAAO,MACvB,MACC,EAAE,SAAS,WACX,OAAO,EAAE,QAAQ,UAAU,EAAE,CAAC,CAC3B,YAAY,CAAC,CACb,SAAS,SAAS,CACzB;GACA,MAAM,QAAQ,IAAI,SAAS,SAAS,GAAA,CAAI,YAAY;GACpD,IAAI,cAAc,KAAK,SAAS,SAAS,KAAK,KAAK,SAAS,UAAU,GACpE,OAAO;IAAE,cAAc;IAAW,QAAQ;GAA0B;GAEtE,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,SAAS,MAAM;GACnD,MAAM,yBAAS,IAAI,IAAoB;GACvC,KAAK,MAAM,KAAK,OAAO;IACrB,MAAM,OAAQ,EAAsC;IACpD,MAAM,MAAM,OAAO,IAAI,IAAI,KAAK,CAAC;IACjC,IAAI,KAAK,CAAC;IACV,OAAO,IAAI,MAAM,GAAG;GACtB;GACA,KAAK,MAAM,CAAC,MAAM,QAAQ,QAAQ;IAChC,MAAM,OAAO,IAAI,QAAQ,MAAM,EAAE,WAAW,OAAO;IACnD,IAAI,KAAK,UAAU,KAAK,KAAK,WAAW,IAAI,QAC1C,OAAO;KACL,cAAc;KACd,QAAQ,GAAG,KAAK,OAAO,+BAA+B,KAAK;KAC3D,eAAe,KAAK,KAAK,SAAS,EAAE,CAAE;IACxC;GAEJ;GACA,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,YAAY;GACzB,IAAI,IAAI,SAAS,SAAS,OAAO,OAAO;GACxC,MAAM,oBAAoB,MAAM,MAC7B,MACC,EAAE,SAAS,WACV,EAAE,YAAY,mBAA0C,KAAA,KACxD,EAAE,YAAY,iBAA4B,CAC/C;GACA,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,SAAS,MAAM;GACnD,IAAI,qBAAqB,MAAM,WAAW,GACxC,OAAO;IACL,cAAc;IACd,QAAQ;GACV;GAEF,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,QAAQ,MAAM,MACjB,MACC,EAAE,SAAS,WACV,EAAuC,cAAc,YACrD,EAAuC,QAAQ,EACpD;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,eAAe,MAAM;GACvB,IACA;EACN;CACF;AACF;AAEA,SAAS,oBAAoB,SAAkC,QAAyB;CACtF,IAAI,WAAW,wBAAwB,YAAY,QAAQ,kBAAkB,CAAC,CAAC,SAAS,GACtF,OAAO;CACT,OAAO,gBAAgB,OAAO,CAAC,CAAC,MAAM,SAAS,KAAK,WAAW,MAAM;AACvE;AAEA,SAAS,iBAAiB,SAA2C;CACnE,IAAI,YAAY,QAAQ,aAAa,CAAC,CAAC,SAAS,GAAG,OAAO;CAC1D,OAAO,gBAAgB,OAAO,CAAC,CAAC,MAC7B,SAAS,MAAM,QAAQ,KAAK,aAAa,KAAK,KAAK,cAAc,SAAS,CAC7E;AACF;AAEA,SAAS,gBAAgB,SAAkE;CACzF,OAAO;EACL,GAAG,QAAQ,QAAQ,OAAO;EAC1B,GAAG,QAAQ,QAAQ,eAAe;EAClC,GAAG,QAAQ,QAAQ,KAAK;CAC1B;AACF;AAEA,SAAS,QAAQ,OAAgD;CAC/D,IAAI,CAAC,MAAM,QAAQ,KAAK,GAAG,OAAO,CAAC;CACnC,OAAO,MAAM,QACV,SACC,QAAQ,IAAI,KAAK,OAAO,SAAS,YAAY,CAAC,MAAM,QAAQ,IAAI,CACpE;AACF;AAEA,SAAS,YAAY,OAA0B;CAC7C,OAAO,MAAM,QAAQ,KAAK,IACtB,MAAM,QAAQ,SAAyB,OAAO,SAAS,QAAQ,IAC/D,CAAC;AACP;;AAGA,SAAgB,gBACd,KACA,QAAuB,eACA;CACvB,IAAI,IAAI,IAAI,SAAS,SAAS,SAAS,IAAI,IAAI,WAAW,aACxD,OAAO;EAAE,cAAc;EAAW,QAAQ;CAAqD;CAEjG,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,MAAM,KAAK,MAAM,GAAG;EAC1B,IAAI,KAAK,OAAO;CAClB;CACA,OAAO;EAAE,cAAc;EAAW,QAAQ;CAAsD;AAClG;;;;;;;;;;;ACpaA,eAAsB,sBACpB,OACA,OACA,UAA0B,CAAC,GACF;CACzB,MAAM,QAAQ,MAAM,UAAU,OAAO,KAAK;CAC1C,IAAI,MAAM,WAAW,GACnB,OAAO;EACL;EACA,YAAY;EACZ,uBAAuB;EACvB,QAAQ,CAAC;EACT,WAAW;EACX,eAAe;EACf,WAAW;CACb;CAGF,MAAM,SAAoC,CAAC;CAC3C,IAAI,cAAc;CAClB,IAAI,kBAAkB;CACtB,IAAI,wBAAwB;CAC5B,MAAM,cAAc,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;CACvE,MAAM,iCAAiB,IAAI,IAAY;CAGvC,KAAK,MAAM,KAAK,aAAa;EAC3B,OAAO,EAAE,cAAc;GACrB,OAAO;GACP,uBAAuB;GACvB,QAAQ;GACR,cAAc;GACd,YAAY;EACd;EACA,MAAM,OAAO,OAAO,EAAE;EACtB,KAAK,SAAS;EACd,IAAI,EAAE,WAAW,SAAS;GACxB,KAAK,UAAU;GACf,eAAe;EACjB;EACA,IAAI,OAAO,EAAE,cAAc,UAAU,KAAK,gBAAgB,EAAE;EAC5D,IAAI,oBAAoB,CAAC,GAAG;GAC1B,yBAAyB;GACzB,KAAK,yBAAyB;GAC9B,MAAM,MAAM,GAAG,EAAE,SAAS,GAAG,QAAQ,EAAE,IAAI;GAC3C,IAAI,eAAe,IAAI,GAAG,GAAG;IAC3B,KAAK,cAAc;IACnB,mBAAmB;GACrB;GACA,eAAe,IAAI,GAAG;EACxB;CACF;CAEA,KAAK,MAAM,QAAQ,OAAO,OAAO,MAAM,GACrC,KAAK,eAAe,KAAK,QAAQ,IAAI,KAAK,eAAe,KAAK,QAAQ;CAIxE,IAAI,qBAAqB;CACzB,IAAI,kBAAkB;CACtB,KAAK,MAAM,GAAG,QAAQ,QAAQ,cAAc,MAAM,EAAE,QAAQ,GAC1D,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;EACnC,IAAI,IAAI,EAAE,CAAE,WAAW,SAAS;EAChC,sBAAsB;EACtB,IAAI,IAAI,IAAI,IAAI,mBAAmB;CACrC;CAEF,MAAM,YAAY,qBAAqB,IAAI,kBAAkB,qBAAqB;CAElF,IAAI;CACJ,IAAI,QAAQ,iBAAiB;EAC3B,MAAM,UAAU,YAAY,QAAQ,MAAM,EAAE,UAAU,QAAQ,eAAgB;EAC9E,IAAI,QAAQ,SAAS,GACnB,oBACE,QAAQ,QAAQ,MAAM,QAAQ,gBAAiB,EAAE,OAAO,CAAC,CAAC,SAAS,QAAQ;CAEjF;CAEA,OAAO;EACL;EACA,YAAY,YAAY;EACxB;EACA;EACA,WAAW,cAAc,YAAY;EACrC,eAAe,wBAAwB,IAAI,kBAAkB,wBAAwB;EACrF;EACA;CACF;AACF"}
@@ -0,0 +1,271 @@
1
+ # Statistics: adoption decisions
2
+
3
+ This is the standing decision record for `src/statistics.ts` and the estimators layered on it.
4
+ It answers three questions: which statistic is trustworthy, where each number comes from, and what a caller is allowed to ask for at 3–10 repetitions per arm.
5
+ The API itself is documented in [`concepts.md`](../concepts.md); this file records *why* each statistic is kept, fixed, or refused.
6
+
7
+ ## Verdict
8
+
9
+ Keep the statistics in-house, fix them here, and take no statistics package as a runtime dependency.
10
+
11
+ Of the 38 exported statistics, 17 were correct, 15 were defective, and 6 are correct mathematics applied in a regime where they mislead.
12
+ The defects concentrated in exactly the functions a promotion gate reads: the paired rank tests, the bootstrap interval, and the multiple-comparison boundary.
13
+ All 15 are repaired; the 6 regime limits are now declared in the result rather than left for a caller to infer.
14
+ The largest single defect is a standard-normal CDF that mixed the arguments of the Abramowitz–Stegun error-function approximation, giving up to `3.7189e-2` absolute CDF error at `x = 0.567` where a correct implementation is bounded by `7.5e-8`.
15
+ That one line made every affected p-value too small by 26–36 % relative, so the module's real type-I error rate was 6.53 % at a nominal 5 %.
16
+
17
+ The reason to stay in-house is not preference.
18
+ Every statistic a JavaScript library gets right is one this module already gets right, and every statistic this module gets wrong at 3–10 reps is one no JavaScript library gets right either.
19
+ A survey of 13 installed packages, tested by execution rather than documentation, found no npm package that computes an exact rank test under ties, Cliff's delta, or a paired bootstrap interval.
20
+
21
+ ## How the numbers below were produced
22
+
23
+ Every measurement in this document was produced by executing the code, not by reading it.
24
+
25
+ The pristine baseline is `origin/main` blob `md5 15ca1ed9fbdd10215b5d0b682d99946c`, extracted with `git archive` and bundled with `esbuild` so that concurrent edits to the working tree could not move it.
26
+ The comparison build is the same bundle taken from the working tree.
27
+ References are `scipy 1.13.1` (`stats.norm`, `stats.t`, `stats.mannwhitneyu`, `stats.wilcoxon`, `stats.binomtest`, `stats.pearsonr`, `stats.spearmanr`) and, for the exact conditional null, an independent enumeration of every split of the observed data that agreed with scipy's permutation test to four decimals on 9 of 9 two-sample cases.
28
+ Monte Carlo figures are 4,000 trials per cell with a seeded generator unless stated otherwise.
29
+
30
+ Where a number in this document differs from one recorded elsewhere, the difference is almost always which build was measured.
31
+ The same call `mannWhitneyU([1,2,3],[4,5,6]).p` returns `0.03769147` on `origin/main` and `0.04953448` once the CDF is repaired; both are correctly reported values of different code.
32
+
33
+ ## Decision key
34
+
35
+ - **keep-and-fix** — the statistic belongs in this package, the implementation is wrong or incomplete, and the fix is ours to write.
36
+ - **keep-with-CI-oracle** — the implementation is correct and stays as-is, pinned against a scipy-generated golden fixture so it cannot silently regress.
37
+ - **replace-with-library** — a maintained package does this better than we do.
38
+ - **delete** — the statistic should not be callable.
39
+
40
+ No statistic in this module is marked **replace-with-library**.
41
+ That is the survey's conclusion, not an oversight, and the argument is in [Dependency verdict](#dependency-verdict).
42
+
43
+ ## Defective
44
+
45
+ These 15 produced a wrong number, hung, or fabricated a verdict.
46
+ All are repaired, each with a regression test that fails on the pristine blob and passes after.
47
+
48
+ | Statistic | Status | Decision | Measured evidence | Regime caveat |
49
+ | --- | --- | --- | --- | --- |
50
+ | `normalCdf` | defective, **fixed** | keep-and-fix | `t = 1/(1+p·|x|)` used the unscaled argument while the exponential used `exp(-x²/2) = exp(-z²)` with `z = x/√2`; max abs CDF error `3.7189e-2` at `x = 0.567`. Repaired to evaluate both halves at `z = |x|/√2`; worst deviation from `scipy.stats.norm.cdf` now `4e-10` at `x = 1.6` and `3.4e-8` at `x = 0.565`, inside the A&S 7.1.26 bound of `7.5e-8`. | None once fixed. |
51
+ | `normalCdf` duplicate in `src/baseline.ts` | defective, **fixed** | keep-and-fix | A byte-identical second copy of the same defect reached production verdicts through `welchsTTest → studentTCdf`. Deleted with its duplicated `studentTCdf`/`incompleteBeta`/`lnGamma`; the shared math now lives in `src/math/normal.ts`, `src/math/student-t.ts` and `src/math/special-functions.ts`. A `df = 298` case moved from `p = 0.010887` to the true Student-t `0.015150`. | Now one implementation repo-wide: `grep 0.3275911` returns a single site. |
52
+ | `mcnemarPower` | defective, **fixed** | keep-and-fix | Reached the normal distribution through the broken forward CDF while its documented inverse `mcnemarRequiredN` reached it through `zQuantile`, so the two were not inverses. `mcnemarPower({p10:0.2, p01:0.1, nPairs:200})` returned `0.773437` against a true `0.736522`; it now returns `0.736522`. | Power was **overstated** above 0.5 and understated below it, so pre-registered N read off this function was too small. |
53
+ | `studentTCdf` `df > 100` branch | defective, **deleted** | delete | The branch short-circuited to `normalCdf`, which changed a two-sided `df = 102, t = 1.98` result from the true `0.050398` to `0.047703` and flipped a 5% decision. | Every finite degree of freedom now uses the regularized incomplete beta. |
54
+ | `studentTCdf` / `incompleteBeta` at `df ≤ 100` | defective, **fixed** | keep-and-fix | The continued fraction omits the mandatory symmetry branch (`x < (a+1)/(a+b+2)`, else `1 − I(1−x, b, a)`), so it is evaluated outside its convergence domain for small `|t|` and collapses as `x → 1`. `studentTCdf(0.005, 100)` returns `0.89152130` against a true `0.50198972`, an error of `0.3895`. Through `pairedTTest` at `df = 7`: `t = 0.001` reports `p = 0.15130173` against a true `0.99923002`, and `t = 1e-6` reports `p = 0.00015251`. There is a discontinuity at zero — `t = 1e-8` returns exactly `p = 1.0` because `x ≥ 1` short-circuits. | A perfectly null paired result reports `p < 0.05`. The materially wrong band is `|t| ≲ 0.02` at `df = 2–10`, widening to `|t| ≲ 0.058` near `df = 100`; outside it `pairedTTest` matched `scipy.stats.ttest_rel` to `≤ 1e-6` relative across 166 cases. |
55
+ | `interRaterReliability` | defective, **fixed** | keep-and-fix | The grouping loop iterates judge-major and opens a new item bucket whenever the last holds `judgeScores.length` entries, so each bucket collects consecutive scores from the *same* judge. It measures within-judge spread, not between-judge agreement, and is anti-correlated with the truth. Two identical judges scoring `[0,100]` return **`−0.500000`** where the true α is `+1.0`; two maximally disagreeing judges (`[0,0]` versus `[100,100]`) return **`+1.000000`** where the true α is `−0.5`. Perfect agreement on three items returns `−0.250000`, on four items `+0.650000`. | The result is also unstable in the item count, because bucket boundaries depend on how the item count divides by the judge count. Zero test coverage repo-wide; consumed by `src/pipelines/judge-agreement.ts:73`. |
56
+ | `mannWhitneyU` (NaN input) | defective, **fixed** | keep-and-fix | The tie-grouping loop advances via `while (j < len && combined[j].v === combined[i].v) j++` then `i = j`. `NaN === NaN` is false, so `j` never leaves `i` and the process spins forever. `mannWhitneyU([1,2,3,4,5,6],[NaN,2,3,4,5,6])` did not return within 8 s and blocked the event loop hard enough that a 4 s `setTimeout` never fired. | A single NaN score hangs a campaign rather than failing it. `ranks` is safe only because it uses `i = j + 1`. |
57
+ | `wilcoxonSignedRank` (NaN input) | defective, **fixed** | keep-and-fix | The same construct at the absolute-rank grouping loop, same hang. | Reachable from `held-out-gate.ts:278`, so a NaN in a holdout set hangs the gate. |
58
+ | `mannWhitneyU` (tie and continuity correction) | defective, **fixed** | keep-and-fix | The variance uses the no-ties formula `n₁n₂(n₁+n₂+1)/12` and there is no continuity correction. `[1,2,3]` versus `[4,5,6]` (zero ties) and `[0,0,0]` versus `[1,1,1]` (maximal ties) both return the byte-identical `p = 0.04953448`. scipy's tie-and-continuity-corrected asymptotic separates them at `0.080856` and `0.046854`. | Both omissions push p downward, so they compound the CDF defect rather than offsetting it. `u` is returned as `min(u1,u2)`, which discards the direction of the effect. |
59
+ | `wilcoxonSignedRank` (tie and continuity correction) | defective, **fixed** | keep-and-fix | The variance uses `n(n+1)(2n+1)/24` with no tie term `−(1/48)Σ(t³−t)` and no continuity correction, despite the function computing average ranks for ties. | Same compounding direction. The statistic convention also differs from scipy without being documented: this returns `W⁺`, scipy returns `min(W⁺, W⁻)`. |
60
+ | `wilcoxonSignedRank` (`n < 6` branch) | defective, **fixed** | **branch deleted** | Line 219 hard-returns `{w: 0, p: 1}` below six non-zero differences, with no exception, no flag, and nothing in the result to distinguish it from a measured null. A clean `+0.5` on all five of five pairs returns `p = 1` where the exact answer is `0.0625`. Because zero deltas are dropped first, ties silently push `n` under the threshold: ten pairs with five zero deltas also returned `p = 1`. | This is the single most consequential open defect at 3–10 reps, and it violates the package's own *no fallbacks, fail loud* rule. It is a false-negative generator precisely in the target regime. |
61
+ | `bonferroni` | defective, **fixed** | keep-and-fix | Adjusted values are correct (`min(1, p·k)` matched R `p.adjust` on 8 case sets), but the rejection boundary uses strict `<` where the rule is `p ≤ α/k`. With `p = [0.0125]×4` and `α = 0.05` it returns `significant = [false,false,false,false]`; `holm` on the identical input returns `[true,true,true,true]`. It also has no input validation, unlike `holm`: `bonferroni([-0.1, 0.2], 0.05)` returns `{adjusted: [-0.2, 0.4], significant: [true, false]}` — a negative p-value declared significant. | Two corrections in one module disagree at their shared boundary. |
62
+ | `benjaminiHochberg` | defective, **fixed** | keep-and-fix | q-values are correct (matched R `p.adjust('BH')` on 8 case sets including the 15-value worked example and ties), same two rule defects. `benjaminiHochberg([0.05, 0.05], 0.05)` returns `significant = [false, false]` where BH rejects iff `q ≤ α`. `benjaminiHochberg([-0.1, 0.2])` returns `qValues = [-0.2, 0.2], significant = [true, false]`. | Drives `rl/contamination.ts:162` and `summary-report.ts:159`. |
63
+ | `pairedTTest` (zero-variance branch) | defective, **fixed** | keep-and-fix | Line 199 returns `p = 0` when the standard error is zero. `pairedTTest([0, 0, 0.5], [0.5, 0.5, 1])` — a constant `+0.5` shift on three pairs — returns `{t: null, df: 2, p: 0}`. That is absolute certainty from three observations where the exact signed-rank floor at `n = 3` is `0.25`, and the object is internally inconsistent (`t` serialized as `null` alongside `p = 0`). | `pairedCohensDz` handles the identical condition correctly by returning `null` and documents why. The module should answer the degenerate case once, the same way, everywhere. |
64
+ | `cohensD` (degenerate branches) | defective, **fixed** | keep-and-fix | Returns a silent `0` twice: when either sample has fewer than two observations, and when the pooled standard deviation is zero. `cohensD([1,1,1],[2,2,2])` returns `0` — "no effect" for a maximal, zero-variance separation. | Third distinct answer to the same degenerate condition, after `pairedTTest`'s `p = 0` and `pairedCohensDz`'s `null`. Only `pairedCohensDz` is right. |
65
+ | `mulberry32` | defective, **fixed** | keep-and-fix | `let s = seed \| 0 \|\| 0x9e3779b9` collapses seed `0` to the golden-ratio constant. `mulberry32(0)` and `mulberry32(0x9e3779b9 \| 0)` both emit `[0.3588899802, 0.1059032613, 0.6752904793]`. | Seed `0` is a common default, so two runs a caller believes are independent replicates are the same run. |
66
+ | `makeRng` / bootstrap seeding | defective, **fixed** | keep-and-fix | `makeRng` returns raw `Math.random` when `opts.seed` is undefined. Two back-to-back `confidenceInterval` calls on identical input returned `lower = 0.30000000000000004` and `lower = 0.3`, `upper = 0.6749999999999999` and `upper = 0.6499999999999999`. Seeding works when supplied (`seed: 7` reproduced exactly). | This contradicts `mulberry32`'s own docstring, which states that a seed is required because unseeded randomness in gate verdicts is non-reproducible by construction. `src/contract/analyze-runs.ts:908` is the live call site that passes no seed; the gates in `held-out-gate.ts`, `promotion-policy.ts`, `statistical-heldout.ts`, `measured-comparison.ts` and `summary-report.ts` all thread one through. |
67
+
68
+ ## Correct mathematics, misleading regime
69
+
70
+ These 5 compute what they claim.
71
+ They mislead at 3–10 repetitions because the asymptotic assumption does not hold there, and no change to the implementation fixes that.
72
+
73
+ | Statistic | Status | Decision | Measured evidence | Regime caveat |
74
+ | --- | --- | --- | --- | --- |
75
+ | `mannWhitneyU` small-n | approximation limit, **exact path added** | keep-and-fix (add exact path) | The normal approximation was applied unconditionally, including `n = 1`, with no switchover threshold. At three per group the exact minimum attainable two-sided p is `0.100000`, yet complete separation returned `p = 0.04953448`. The default now selects exact computation from the dynamic program's actual state and work, so balanced 12+12 and imbalanced 1+24 designs are both exact. Above that limit, the automatic permutation seed is invariant to observation order and group order; the prior seed changed a two-sided result from `0.04960` to `0.04420` when the groups were swapped. | Repairing the CDF did not fix this; only an exact test does. Discreteness is the binding constraint, and tie-aware `pFloor` now reports it on every result. |
76
+ | `wilcoxonSignedRank` small-n | approximation limit, **exact path added** | keep-and-fix (add exact path) | Where the approximation ran it was anti-conservative at `n = 6–7` (minimum attainable `p = 0.0209` against an exact `0.0312` at `n = 6`) and conservative from `n ≥ 8`. The default is now exact by sign-flip enumeration at `n ≤ 20`. | The `n < 6` hard return is gone. `pFloor = 2^(1−n)` is reported on every result, so a design that cannot reach alpha says so. |
77
+ | `confidenceInterval` / `pairedBootstrap` | approximation limit, **floor declared** | keep-and-fix (add an n floor) | Percentile bootstrap, correctly constructed. The gate-relevant quantity is `P(low > 0)` under a true null against a nominal 2.5 %. Measured over 4,000 seeded trials on a continuous null: **13.53 % at `n = 3`**, 3.33 % at `n = 5`, 4.45 % at `n = 8`, 3.52 % at `n = 10`, 3.10 % at `n = 20` for the median statistic; 13.85 %, 7.95 %, 5.80 %, 4.90 %, 3.80 % for the mean statistic. On a five-value discrete grid resembling judge scores: 6.02 % at `n = 3` (median), 6.68 % (mean), still 3.43 % at `n = 10` (mean). | This is an intrinsic limit of bootstrapping three points, not an implementation error — scipy's BCa on the same `n = 3` data gives 16.0 %. The defect is the docstring, which states that `low > threshold` means the gain is real at the confidence level. At `n = 3` that claim is wrong by more than 5×. Consumed by `promotion-policy.ts:133`, `held-out-gate.ts:272`, `statistical-heldout.ts:174`, `measured-comparison.ts:995`, `analyze-runs.ts:908`. `pairedBootstrap` now returns `gateEligible`, false below `BOOTSTRAP_GATE_MIN_N = 20`. |
78
+ | `pairedRiskDifference` | approximation limit | keep-and-fix | The point estimate and the variance formula both matched the closed form exactly on 6 configurations, but the interval is Wald: empirical coverage against a nominal 95 % is 73.80 % at `n = 5`, 86.48 % at `n = 10`, 93.80 % at `n = 20`, 94.73 % at `n = 200`. With no discordant pairs the variance is exactly zero, so `pairedRiskDifference([0,0,0],[1,1,1])` returns `riskDifference = 1, lower = 1, upper = 1` — a zero-width 95 % interval asserting certainty from three observations, and ten concordant pairs return `[0, 0]`. | The module already ships `wilson` and its own comment block argues against exactly this Wald approach for proportions. A Wilson-style or Tango score interval is the standard fix. |
79
+ | `requiredSampleSize`, `requiredPairedSampleSize`, `pairedMde` | approximation limit, **documented** | keep-with-CI-oracle | All three match their stated normal-approximation formulas exactly against `scipy.stats.norm.ppf` closed forms, and are numerically unchanged by the CDF fix because they route through `zQuantile` rather than the forward CDF (`requiredSampleSize({effect: 0.5}) = 63` before and after). They use normal quantiles with no t correction, so they understate required n where n is small: `requiredPairedSampleSize({effect: 0.5})` returns **32** against an exact t-based **34**, and `{effect: 0.8}` returns **13** against **15**. | A 6–13 % shortfall, precisely in the range a caller consults to decide whether 3–10 reps suffice. Both docstrings now say "treat as a lower bound". |
80
+
81
+ ## Correct
82
+
83
+ These 17 matched their references with zero mismatches and stay as they are, pinned against a golden fixture so they cannot regress silently.
84
+
85
+ | Statistic | Decision | Measured evidence |
86
+ | --- | --- | --- |
87
+ | `pairedTTest` (normal range) | keep-with-CI-oracle | Across 166 cases the t statistic matched `scipy.stats.ttest_rel` to `≤ 1.5e-15` and p to `≤ 1e-6` relative in all but 4. Type-I under a true null over 20,000 replicates: 5.20 % at `n = 3`, 4.74 % at `n = 4`, 4.97 % at `n = 6`, 5.17 % at `n = 8`, 5.03 % at `n = 10` — correctly calibrated at this package's actual sample sizes. |
88
+ | `mcnemar` | keep-with-CI-oracle | Exact two-sided binomial on discordant pairs; matched `scipy.stats.binomtest` two-sided to `< 1e-9` on all 13 `(b,c)` configurations including `(0,0)`, `(10,0)`, `(100,70)`. A `b = 1, c = 5` case returns `pValue = 0.21875`. Log-space accumulation via `lnGamma` keeps it stable at large counts. |
89
+ | `pairedSignTest` | keep-with-CI-oracle | Exact one-sided binomial; matched `scipy.stats.binomtest(..., alternative='greater')` to `< 1e-12` on all 9 cases. `pairedSignTest([0.5,0.5,0.5], 'greater')` returns `0.125`. Ties are excluded from the denominator and still reported; `alternative` must be passed explicitly and an invalid value throws, which prevents post-hoc direction selection. |
90
+ | `wilson` | keep-with-CI-oracle | Matched the closed form to `< 1e-8` on all 13 `(successes, n)` pairs including `0/1`, `10/10`, `1/1000`, `0/0`. Correctly asymmetric at the boundaries: `wilson(0, 10)` returns `[0, 0.27753280]`. |
91
+ | `passAtK` | keep-with-CI-oracle | Chen et al. 2021 unbiased estimator, exhaustively verified for every `(n, c, k)` with `n = 1..6` against exact integer `math.comb`: zero mismatches. Stable at scale (`n=1000, c=3, k=100` agreed to `1.11e-16`). `passAtK(10, 3, 5) = 0.9166666667`. |
92
+ | `corpusInterRaterAgreement` | keep-with-CI-oracle | The ICC(2,1) it surfaces matched a hand-derived two-way random-effects ANOVA reference to `< 1e-9` on 5 matrices, including the inverted case at `−1.959459`. This is a genuinely different and correct computation from `interRaterReliability`: it pivots to a proper items × judges matrix and delegates to `continuousAgreement`. Its fail-loud contract behaves — empty input, fewer than two judges, fewer than two common items, duplicate records, and absent dimensions all throw `ValidationError`. |
93
+ | `eProcess` | keep-with-CI-oracle | Betting test-martingale. Empirical type-I over 4,000 sequences of 200 observations at `α = 0.05`: 2.40 % at the null boundary, 0.00 % in the null interior, 1.33 % on continuous uniform — all inside Ville's bound. Power at `E[x] = 0.7` is 99.775 %. The predictability invariant holds: the first update leaves wealth at exactly 1. |
94
+ | `holm` | keep-with-CI-oracle | Matched `statsmodels.multipletests` to `1e-9` on a 6-value reference vector, and uses `≤` at the boundary, which is the correct rule. It also validates both `alpha` and the p range, which `bonferroni` does not. |
95
+ | `zQuantile` | keep-with-CI-oracle | Acklam inverse-normal, structurally independent of `normalCdf`. This independence is why the sample-size functions were untouched by the CDF defect, and why `mcnemarPower`'s failure to invert `mcnemarRequiredN` was a valid detector of it. |
96
+ | `mcnemarRequiredN` | keep-with-CI-oracle | Matched the Lachin closed form exactly on all 4 parameter sets: `234`, `77`, `155`, `Infinity`. Unchanged by the CDF fix. Round-trip against the repaired `mcnemarPower` now holds across 16 configurations: power at `requiredN` meets the target with overshoot `≤ 0.0049`, and power at `requiredN − 1` is below target in all 16. |
97
+ | `ranks` | keep-with-CI-oracle | Correct average-rank-with-ties. `ranks([3,1,1,2])` returns `[4, 1.5, 1.5, 3]`, matching `scipy.stats.rankdata`. Safe against NaN input because it advances with `i = j + 1`. |
98
+ | `pearsonR` | keep-with-CI-oracle | Matched `scipy.stats.pearsonr` to 8 decimals (`0.99061012` on an 8-point case). |
99
+ | `spearmanR` | keep-with-CI-oracle | Matched `scipy.stats.spearmanr` to 8 decimals (`0.99402980` on the same case). |
100
+ | `cliffsDelta` | keep-with-CI-oracle | Matched a hand-computed `(#gt − #lt)/(n₁n₂)` reference exactly (`0.312500`). Note the orientation is *after over before*, the opposite sign to the textbook `δ` written over `(a, b)`; the docstring states this and the parameter names `(before, after)` carry it. |
101
+ | `pairedCohensDz` | keep-with-CI-oracle | Matched `mean(d)/sd(d)` exactly (`2.184070`). It is the only function in the module that handles the zero-variance degenerate case correctly, returning `null` with a docstring explaining that the standardized effect is undefined rather than an arbitrarily large finite number. Treat it as the reference behaviour the other degenerate branches should adopt. |
102
+ | `weightedMean`, `partialCredit`, `weightedComposite`, `interpretCliffs`, `normalizeScores` | keep-with-CI-oracle | Arithmetic and thresholding, verified by direct evaluation (`weightedMean` of `[{1,w2},{4,w1}]` is `2.000000`; `partialCredit(3,4)` is `0.750000`). |
103
+ | `welchsTTest`, `compareToBaseline` (`src/baseline.ts`) | keep-with-CI-oracle, **covered** | Correct, and previously unguarded: no test file imported either, which is the structural reason a duplicated broken CDF survived here. It is exported publicly and it gates improved / regressed / stable verdicts. Now pinned against `scipy.stats.ttest_ind(equal_var=False)` in the oracle fixture and exercised through `compareToBaseline` in `tests/statistics.test.ts`. |
104
+
105
+ ## Public surface change
106
+
107
+ `normalCdf` and `studentTCdf` were private on `origin/main` and are now exported with explicit accuracy contracts.
108
+ `baseline.ts` owns the single Welch implementation, and `contract/analyze-runs.ts` consumes that result instead of carrying another normal-approximation copy.
109
+ That is the right trade: one implementation with one accuracy contract beats three copies of which two were wrong.
110
+ It also means both functions are now public API and owe callers a stated accuracy bound.
111
+ `normalCdf` is documented at `7.5e-8` absolute.
112
+ `studentTCdf` is exact to the incomplete beta's own precision at every finite degree of freedom and matches `scipy.stats.t.cdf` to `1e-9` across the pinned oracle cases; the residual floor near `t = 0` is float64 cancellation in `x = df/(df + t²)`, not the approximation.
113
+
114
+ ## Dependency verdict
115
+
116
+ **Take no statistics package as a runtime dependency.**
117
+ **Use scipy as a CI oracle.**
118
+ **Implement the exact small-n rank tests here, because nothing else does.**
119
+
120
+ ### What we must implement ourselves
121
+
122
+ Four things the survey found no correct implementation of in npm, at any package, under ties:
123
+
124
+ | Needed | Library that does it correctly |
125
+ | --- | --- |
126
+ | Exact two-sample rank test under ties | none |
127
+ | Exact paired signed-rank test under ties | none |
128
+ | Cliff's delta | no package exists in npm at all |
129
+ | Paired bootstrap confidence interval | none |
130
+
131
+ Those four rows are the entire bottleneck, and they are the reason this decision is not close.
132
+ `lib-r-math.js` comes closest — it is the only source of the exact two-sample rank-sum null distribution in JavaScript, and it reproduces every exact floor we care about (`pwilcox(0, 3, 3) = 0.100000`, `psignrank` at `n = 5` gives `0.062500`, at `n = 8` gives `0.007813`) at a cost of 2 packages and 1.2 MB.
133
+ But its signature is `(q, m, n)` with no tie vector, so ties are structurally unrepresentable, exactly as in R where `wilcox.test` warns and falls back.
134
+ It ships distributions, not tests, so the test wrapper, the tie conditioning, and the effect sizes remain ours regardless.
135
+
136
+ The cost of writing them ourselves is small and was measured, not estimated.
137
+ Enumerating every split of the observed data takes 0.1 ms at 3 versus 3 (20 splits), **3.7 ms at 10 versus 10** (184,756 splits), and 33 ms at 12 versus 12 (2,704,156 splits).
138
+ Paired sign-flip enumeration is `2ⁿ`: 1,024 at `n = 10`, about 1 M at `n = 20`.
139
+ The entire stated 3–10 repetition regime runs exact in under 4 ms.
140
+ This is roughly 150 lines, not a research project.
141
+
142
+ ### What we must not take as a runtime dependency
143
+
144
+ The libraries that are correct are correct at things this module already gets right.
145
+
146
+ `@stdlib/stats-padjust` matches `statsmodels.multipletests` to `1e-9` on `holm`, `bh`, and `bonferroni`, and so does this module's own `holm`, `benjaminiHochberg`, and `bonferroni` on the same frozen reference vectors.
147
+ It was removed even as a development dependency because it bought zero additional coverage for 190 locked packages.
148
+ `@stdlib/stats-ranks` and `jstat.rank` both do average-rank-with-ties correctly, and so does `ranks` here.
149
+ `@sipemu/anofox-statistics`'s count-based exact tests are correct (`fisherExact([[3,0],[0,3]]) = 0.09999999999999992`, `binomTest(3,3,0.5) = 0.25000000000000006`, `mcnemarExact = 0.21875000000000008`), and so are `mcnemar` and `pairedSignTest` here.
150
+
151
+ Meanwhile the libraries that cover the missing statistics are wrong in this regime, sometimes worse than the incumbent:
152
+
153
+ - `@stdlib/stats-kruskal-test` matches `scipy.kruskal` to `1e-9` but is anti-conservative against the exact conditional truth in 7 of 8 cases: `p = 0.0495` against an exact `0.1000` at 3 versus 3, and `p = 0.4945` against an exact `1.0000` on a binary grid — off by `0.5055`. It returns a silent `NaN` on all-tied input rather than throwing. Measured cost: 286 packages, 11,997,061 bytes, 4,033 files.
154
+ - `@sipemu/anofox-statistics`'s `mannWhitneyU` returns `p_value = 0` for two **identical** samples, verified on four inputs including `x = y = [0.5,0.5,0.5]`, where the truth is `p = 1.0`. Its documented `exact` flag is a silent no-op under ties: all five tied cases returned byte-identical p for `exact: true` and `exact: false`.
155
+ - `@stdlib/stats-wilcoxon` has a genuine exact path but silently switches to the normal approximation when the differences contain ties, with nothing in the result to signal it — the `method` string is identical either way. Holding the statistic constant at `W = 15, n = 5`, untied input returns the exact `0.062500` and tied input returns `0.053337`, a p-value below the exact floor and therefore one that cannot exist at `n = 5`.
156
+ - `simple-statistics`'s `wilcoxonRankSum` returns the wrong rank sum on 5 of 9 regime cases; every case with two or more tie groups is corrupted. The tie accumulator is reset only in the single-element branch, so a multi-element group's average spans from the previous group's start. Upstream PR #809 carries a fix and a regression test, opened 2026-07-08, still open with no maintainer response.
157
+ - `mann-whitney-utest` and `@tainakanchu/mann-whitney-utest` (byte-identical source) compare a U statistic against a z-score in their significance decision, so the comparison is dimensionally meaningless. Complete separation at 3 versus 3 reports `significant = true` where no result at that size can reach `α = 0.05`.
158
+ - `jstat` (last published 2022-11-21) has no Mann-Whitney, no Wilcoxon, no Kruskal, no p-adjust. `science.js` (last published 2015-08-20) has zero hypothesis tests.
159
+
160
+ Dependency weight is a real cost, not a stylistic one.
161
+ This package's entire runtime dependency set is 7 packages.
162
+ Adding 190–286 for statistics we already compute correctly would be inherited by every consumer of `@tangle-network/agent-eval`.
163
+
164
+ ### What scipy is for
165
+
166
+ scipy is the CI oracle and never ships.
167
+
168
+ Pin scipy-generated golden values as a JSON fixture under `tests/`, regenerated by a checked-in script, and assert every **keep-with-CI-oracle** statistic against it.
169
+ That is `scripts/generate-statistics-oracle.py` → `tests/fixtures/statistics-oracle.json`, asserted by `tests/statistics-oracle.test.ts`: 154 cases across 22 statistics, each carrying its own tolerance (`7.5e-8` where A&S bounds it, `1e-8` where Acklam's inverse normal does, `1e-12` elsewhere).
170
+ scipy 1.13.1 was the reference for every number in this document and it catches exactly the class of defect found here: a hand-rolled approximation that is plausible on inspection and wrong by `3.7e-2`.
171
+ `lib-r-math.js` is a **devDependency only**, cross-checking the untied exact null distributions in `tests/statistics-library-crosscheck.test.ts`.
172
+ The three correction functions are independently covered by the statsmodels-generated fixture.
173
+ Neither reference implementation is imported by `src/`.
174
+ `fast-check` is already a devDependency, so the invariants that no fixture can express — a p-value never below the attainable floor, monotonicity of p in the statistic, an exact and an asymptotic path agreeing as `n` grows — belong there.
175
+
176
+ ## Exact versus asymptotic policy
177
+
178
+ The governing fact is combinatorial, not numerical.
179
+ At 3 versus 3 there are only 20 possible splits, so the attainable two-sided p-grid is `{0.1, 0.2, …}` and `0.05` is unreachable.
180
+ `[1,2,3]` versus `[4,5,6]` and `[0,0,0]` versus `[1,1,1]` both have an exact `p = 0.1000`, while scipy's tie-corrected asymptotic gives `0.080856` and `0.046854` — **adding the tie correction makes the answer worse.**
181
+ No better approximation reaches the right answer here; only an exact test does.
182
+
183
+ ### Switchover thresholds
184
+
185
+ | Test | Exact by enumeration | Seeded Monte Carlo permutation | Asymptotic |
186
+ | --- | --- | --- | --- |
187
+ | Two-sample rank (`mannWhitneyU`) | dynamic program up to 8,192 cells and 250,000 transitions; includes 12 v 12 and 1 v 24 | above that, default 100,000 permutations | never the default; only on explicit request above the exact work limits |
188
+ | Paired signed-rank (`wilcoxonSignedRank`) | `n ≤ 20` — `2²⁰ = 1,048,576` sign flips | above that, default 100,000 sign flips | same |
189
+ | Paired sign test (`pairedSignTest`) | always exact — binomial, already correct | — | never |
190
+ | McNemar (`mcnemar`) | always exact — binomial, already correct | — | never |
191
+ | Paired mean difference (`pairedTTest`) | — | — | valid from `n ≥ 3`, now that the `incompleteBeta` symmetry branch is in place |
192
+ | Bootstrap interval (`confidenceInterval`, `pairedBootstrap`) | — | — | **not a valid gate below `n = 20`** |
193
+
194
+ The bootstrap row is the strongest recommendation here and the one most likely to be resisted.
195
+ Measured `P(low > 0)` under a true null against a nominal 2.5 % never reaches nominal in the tested range: 13.53 % at `n = 3`, 3.52 % at `n = 10`, 3.10 % at `n = 20` for the median statistic, and 13.85 % / 4.90 % / 3.80 % for the mean.
196
+ Below `n = 20` the bootstrap interval should be reported as descriptive spread and must not be the leg a promotion turns on; the exact sign test or exact signed-rank should carry the decision instead.
197
+
198
+ ### What to do when the caller asks for a misleading number
199
+
200
+ Refuse, loudly, in the package's own idiom.
201
+
202
+ Every rank test takes `method: 'exact' | 'asymptotic' | 'auto'`, defaulting to `'auto'`.
203
+ `'auto'` selects exact whenever the design is inside the enumeration threshold, Monte Carlo permutation above it, and asymptotic never.
204
+ An explicit `method: 'asymptotic'` inside the exact-feasible range **throws a `ValidationError`** naming the smallest attainable p at that design and the exact work limits.
205
+ An explicit `method: 'exact'` ABOVE the threshold throws too, rather than enumerating a distribution whose cost is unbounded.
206
+
207
+ This is a refusal, not a warning, and the reason is the package's own doctrine.
208
+ A warning on `stderr` does not reach the JSON a gate reads, does not reach a CI log a human skims, and does not survive serialization into a run record.
209
+ An anti-conservative p that a gate silently believes is exactly the class of silent fallback that *no fallbacks, fail loud* exists to prevent, and the `wilcoxonSignedRank` `n < 6` branch was the proof: it returned `p = 1` for real effects from v0.1.0 to 0.133.0 and nothing downstream could tell.
210
+
211
+ The cost of the refusal is real and is stated here rather than discovered later: an explicit-asymptotic caller at 3 v 3 who was reading `0.0495` now gets an exception, and the same design read through the default now reports `0.1000`.
212
+ A historical verdict that turned on the difference was never valid.
213
+
214
+ Two supporting requirements make the refusal usable rather than merely obstructive.
215
+
216
+ Every rank-test result carries `method` and `pFloor` — the method actually used and the smallest attainable p at that design — so a downstream gate can see the discreteness rather than infer it.
217
+ This is the field `@stdlib/stats-wilcoxon` omits, which is why its silent exact-to-approximate switch is undetectable by a caller.
218
+ `pairedBootstrap` carries the same signal as `gateEligible`.
219
+
220
+ Still to do: every gate that consumes a rank test should state its minimum n at construction and fail its own precondition check when the data is smaller, rather than accepting whatever the test returns.
221
+ A gate that cannot reach its alpha at the n it was handed should report *underpowered*, which is a true statement about the experiment, not *not significant*, which is a false statement about the effect.
222
+ `pFloor` and `gateEligible` make that check expressible.
223
+ Promotion paths now use `pairedDeltaTest`: an exact one-sided sign test carries decisions from 6 through 19 pairs, and the bootstrap interval carries them from 20 onward.
224
+
225
+ ## Consumer notice
226
+
227
+ Every published version from **0.1.0 (2026-04-20)** through **0.133.0 (2026-07-27)** reported p-values that are too small from the functions listed below.
228
+ The defect entered at the initial commit (`7d5032b`, `src/statistics.ts:747`) and was present in every release since.
229
+
230
+ The normal CDF itself was corrected in 0.133.1; every other row below was still open at that release.
231
+
232
+ ### Which functions, and by how much
233
+
234
+ | Function | Path to the defect | Direction | Measured |
235
+ | --- | --- | --- | --- |
236
+ | `mannWhitneyU` | `normalCdf`, then the asymptotic path itself | p too small | `[1,2,3]` v `[4,5,6]`: reported `0.03769147`. Repairing the CDF alone gives `0.04953448` (1.31×); the shipped exact answer is `0.10000000` (2.65×), and `0.05` is unreachable at 3 v 3 in the first place. `[1..5]` v `[10..14]`: reported `0.00671001`, shipped exact `0.00793651` (1.18×). |
237
+ | `wilcoxonSignedRank` | `normalCdf`, then the asymptotic path itself | p too small, or fabricated as 1 | `n = 8` constant shift: reported `0.00873623`, CDF-repaired `0.01171872`, shipped exact `0.00781250`. Below six non-zero differences the old code returned `p = 1` regardless of the data: a clean 5-of-5 shift reported `1.0` where the exact answer is `0.0625`. |
238
+ | `pairedTTest` at `df > 100` | deleted `studentTCdf` normal shortcut | p too small | A `df = 298` case: reported `0.010887`, correct Student-t `0.015150`. |
239
+ | `welchsTTest`, `compareToBaseline` | duplicate `normalCdf` in `baseline.ts` | p too small | Same `df = 298` case, same shift. This drives the improved / regressed / stable verdict at `baseline.ts:97`. |
240
+ | `mcnemarPower` | `normalCdf` | power **overstated** above 0.5, understated below | `{p10: 0.2, p01: 0.1, nPairs: 200}`: reported `0.773437`, correct `0.736522`. At `n = 80`: `0.9368` against `0.9197`. At `n = 20`: `0.3294` against a correct `0.3627`. |
241
+ | `pairedTTest` at `df ≤ 100`, small `\|t\|` | `incompleteBeta` — **fixed** | p wildly too small near `t = 0` | `df = 7, t = 0.001`: reported `0.15130173`, correct `0.99923002`. `df = 100, t = 1e-6`: reported `0.00004411`, correct ≈ `1.0`. |
242
+ | `interRaterReliability` | grouping loop — **fixed** | sign inverted | Identical judges return `−0.500000`; maximally disagreeing judges return `+1.000000`. |
243
+
244
+ `requiredSampleSize`, `requiredPairedSampleSize`, `pairedMde`, `mcnemarRequiredN`, `mcnemar`, `pairedSignTest`, `wilson`, `passAtK`, `corpusInterRaterAgreement`, `eProcess`, `holm`, `ranks`, `pearsonR`, `spearmanR`, `cliffsDelta`, and `pairedCohensDz` are **unaffected** — verified numerically identical before and after (`requiredSampleSize({effect: 0.5}) = 63`, `mcnemarRequiredN({p10: 0.2, p01: 0.1}) = 234` both ways).
245
+
246
+ ### How to re-check a decision you already made
247
+
248
+ The defect is monotone in `|z|`, so the affected band is exact and narrow.
249
+
250
+ The broken code crossed `p = 0.05` at `|z| = 1.843031` instead of the correct `1.959964`, and at that true critical value it reported `p = 0.038053`.
251
+ Therefore:
252
+
253
+ - **Any recorded p in `[0.038053, 0.050000)` from an affected function crossed a 5 % gate that it should not have crossed.**
254
+ - At `α = 0.01` the band is `[0.007443, 0.010000)`.
255
+ - At `α = 0.10` the band is `[0.077398, 0.100000)`.
256
+ - A recorded p below `0.038053` was significant at 5 % either way; a recorded p at or above `0.05` was not significant either way. Neither needs re-checking.
257
+
258
+ The practical size of the error: the module's real type-I error rate was **6.53 % at a nominal 5 %** and **1.34 % at a nominal 1 %**, confirmed by 20,000 null replicates at `n = 8` per group where `mannWhitneyU` rejected at 6.47 % as shipped against 4.86 % with the same U and z fed a correct CDF.
259
+
260
+ Three further cautions for anyone auditing an old verdict.
261
+
262
+ A promotion that turned on a `pairedBootstrap` `low > 0` check at fewer than 10 pairs was never valid at the stated confidence, independent of this defect — the measured false-positive rate is 13.53 % at `n = 3` against a nominal 2.5 %.
263
+
264
+ A `wilcoxonSignedRank` leg that reported `p = 1` on fewer than six non-zero differences measured nothing.
265
+ It is not evidence of no effect, and re-running it on this release returns a real exact p.
266
+ Note that exact ties are dropped before ranking, so ten pairs with five tied deltas also fell into that branch.
267
+
268
+ Any bootstrap interval recorded from a release at or before 0.133.0 through `analyze-runs.ts` is not reproducible: that call site passed no seed and `makeRng` fell back to `Math.random`.
269
+ Re-running it against the old release will not give the same interval. From this release the seed is derived from the data when the caller supplies none, so it is reproducible either way — but an interval recorded earlier cannot be reconstructed.
270
+
271
+ Mirror this section into `CHANGELOG.md` at the release that carries the fix, with the affected version range and the re-check bands, so a consumer who never reads this file still gets the notice.
package/docs/design.md CHANGED
@@ -65,5 +65,6 @@ They are not adoption reference:
65
65
 
66
66
  - [`building-doctrine.md`](./building-doctrine.md): conventions our agents follow when consuming this package (reachable model defaults, probe-before-debug, experiment integrity checklist)
67
67
  - [`design/loop-taxonomy.md`](./design/loop-taxonomy.md): the internal vocabulary for execution drivers, workers, measurements, and proposers
68
+ - [`design/statistics-decisions.md`](./design/statistics-decisions.md): per-statistic trust status, the no-runtime-dependency verdict, and the exact-versus-asymptotic policy at 3–10 repetitions
68
69
  - [`research-report-methodology.md`](./research-report-methodology.md): the evidence standard our own research reports are held to
69
70
  - [`.claude/skills/agent-eval/SKILL.md`](../.claude/skills/agent-eval/SKILL.md): directives for LLM agents writing integration code, encoding bug classes we have already shipped and fixed once
@@ -247,7 +247,7 @@ Populated when baseline + candidate candidates are present (auto-detected from t
247
247
  "candidateMean": 0.65,
248
248
  "delta": 0.07,
249
249
  "ci95": [0.04, 0.10], // bootstrap CI on the delta
250
- "pValue": 0.0008, // paired t-test
250
+ "pValue": 0.0008, // paired t-test; null when the delta is a non-zero constant
251
251
  "n": 40, // paired observations
252
252
  "unpairedBaselineRuns": 2,
253
253
  "unpairedCandidateRuns": 1,
@@ -62,9 +62,10 @@ In order: first match wins:
62
62
  |---|---|---|
63
63
  | Marginal CI on score mean | `confidenceInterval` | `statistics.ts` |
64
64
  | Paired Cohen's dz vs comparator | `pairedCohensDz` | `statistics.ts` |
65
- | Wilcoxon signed-rank (paired) | `wilcoxonSignedRank` | `statistics.ts` |
65
+ | Wilcoxon signed-rank (paired), exact at n ≤ 20 | `wilcoxonSignedRank` | `statistics.ts` |
66
66
  | BH-FDR q-values | `benjaminiHochberg` | `statistics.ts` |
67
67
  | Paired bootstrap CI on median delta | `pairedBootstrap` | `statistics.ts` |
68
+ | Smallest p a rank-test design can produce | `pFloor` on the result | `statistics.ts` |
68
69
  | Bayesian-bootstrap Pr(Δ>0), Pr(Δ∈ROPE) | `bayesianBootstrapMeanSamples` | `summary-report.ts` (private) |
69
70
  | Minimum detectable paired effect | `pairedMde` | `statistics.ts` |
70
71
  | Run fingerprint | `hashJson(canonicalize(...))` | `pre-registration.ts` |
@@ -73,6 +74,8 @@ The Pr(Δ>0) and Pr(Δ∈ROPE) summaries use Rubin's Bayesian bootstrap.
73
74
  Each posterior draw assigns the observed paired deltas Dirichlet(1, ..., 1) weights, implemented as normalized independent Exponential(1) draws.
74
75
  The posterior summaries apply to the **mean** delta.
75
76
  The separate frequentist bootstrap CI applies to the **median** delta because the median is more robust to heavy-tailed agent scores.
77
+ It is descriptive spread below `BOOTSTRAP_GATE_MIN_N = 20` pairs, which the result reports as `gateEligible: false`; the exact signed-rank or sign test carries the decision there.
78
+ See [`design/statistics-decisions.md`](./design/statistics-decisions.md) for the measured false-positive rates and the exact-versus-asymptotic policy.
76
79
 
77
80
  ## MDE
78
81
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.133.1",
3
+ "version": "0.133.3",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {
@@ -180,6 +180,7 @@
180
180
  "esbuild": "^0.28.1",
181
181
  "fast-check": "^4.9.0",
182
182
  "husky": "^9.1.7",
183
+ "lib-r-math.js": "^3.0.2",
183
184
  "lint-staged": "^17.2.0",
184
185
  "openapi3-ts": "^4.6.0",
185
186
  "oxc-parser": "^0.141.0",