@tangle-network/agent-eval 0.161.0 → 0.163.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/CHANGELOG.md +51 -0
  2. package/dist/{active-curriculum-CD5TU2yW.js → active-curriculum-OjIWgrUJ.js} +2 -2
  3. package/dist/{active-curriculum-CD5TU2yW.js.map → active-curriculum-OjIWgrUJ.js.map} +1 -1
  4. package/dist/analyst/index.d.ts +2 -2
  5. package/dist/analyst/index.d.ts.map +1 -1
  6. package/dist/analyst/index.js +3 -3
  7. package/dist/{baseline-BhPRQBVn.js → baseline-BC-eBZ7U.js} +2 -2
  8. package/dist/{baseline-BhPRQBVn.js.map → baseline-BC-eBZ7U.js.map} +1 -1
  9. package/dist/{benchmark-command-BVtaq_ve.js → benchmark-command-CF-4GEWZ.js} +9 -8
  10. package/dist/benchmark-command-CF-4GEWZ.js.map +1 -0
  11. package/dist/benchmarks/index.d.ts +1 -1
  12. package/dist/benchmarks/index.js +3 -3
  13. package/dist/builder-eval/index.d.ts.map +1 -1
  14. package/dist/builder-eval/index.js +22 -8
  15. package/dist/builder-eval/index.js.map +1 -1
  16. package/dist/campaign/index.js +8 -8
  17. package/dist/{campaign-BSmOwskD.js → campaign-DQZmc2Dq.js} +12 -12
  18. package/dist/{campaign-BSmOwskD.js.map → campaign-DQZmc2Dq.js.map} +1 -1
  19. package/dist/{chat-json-call-6g5sJobJ.js → chat-json-call-5Jxna-aV.js} +2 -2
  20. package/dist/{chat-json-call-6g5sJobJ.js.map → chat-json-call-5Jxna-aV.js.map} +1 -1
  21. package/dist/cli.js +3 -3
  22. package/dist/contract/index.js +7 -7
  23. package/dist/{counterfactual-D_VWavVm.js → counterfactual-Bjq1mlUu.js} +2 -2
  24. package/dist/{counterfactual-D_VWavVm.js.map → counterfactual-Bjq1mlUu.js.map} +1 -1
  25. package/dist/{define-agent-eval-h-s-sI-v.js → define-agent-eval-D08pWIJb.js} +13 -13
  26. package/dist/{define-agent-eval-h-s-sI-v.js.map → define-agent-eval-D08pWIJb.js.map} +1 -1
  27. package/dist/{descriptive-jDOuI6mz.js → descriptive-1V17A-qa.js} +2 -2
  28. package/dist/{descriptive-jDOuI6mz.js.map → descriptive-1V17A-qa.js.map} +1 -1
  29. package/dist/{dspy-rlm-engine-DptEII26.js → dspy-rlm-engine-DhA9qKIm.js} +2 -2
  30. package/dist/{dspy-rlm-engine-DptEII26.js.map → dspy-rlm-engine-DhA9qKIm.js.map} +1 -1
  31. package/dist/emitter-D_jYSGRd.d.ts.map +1 -1
  32. package/dist/{emitter-BpYFQPj4.js → emitter-DeQHiDMm.js} +13 -6
  33. package/dist/emitter-DeQHiDMm.js.map +1 -0
  34. package/dist/{eval-campaign-BsXWL2-2.js → eval-campaign-BfohKmzx.js} +3 -3
  35. package/dist/{eval-campaign-BsXWL2-2.js.map → eval-campaign-BfohKmzx.js.map} +1 -1
  36. package/dist/experiment/index.js +8 -8
  37. package/dist/{experiment-tracker-Ym6rEQT1.js → experiment-tracker-BKEumQug.js} +2 -2
  38. package/dist/{experiment-tracker-Ym6rEQT1.js.map → experiment-tracker-BKEumQug.js.map} +1 -1
  39. package/dist/{external-optimizer-process-WosTBChy.js → external-optimizer-process-BFmh36vW.js} +2 -2
  40. package/dist/{external-optimizer-process-WosTBChy.js.map → external-optimizer-process-BFmh36vW.js.map} +1 -1
  41. package/dist/{external-optimizer-subprocess-BIWbHpgD.js → external-optimizer-subprocess-CqLMW3nh.js} +2 -2
  42. package/dist/{external-optimizer-subprocess-BIWbHpgD.js.map → external-optimizer-subprocess-CqLMW3nh.js.map} +1 -1
  43. package/dist/fuzz.d.ts +1 -2
  44. package/dist/fuzz.d.ts.map +1 -1
  45. package/dist/fuzz.js +4 -10
  46. package/dist/fuzz.js.map +1 -1
  47. package/dist/index.d.ts +3 -3
  48. package/dist/index.js +26 -26
  49. package/dist/index.js.map +1 -1
  50. package/dist/{internal-BDHPCnjk.js → internal-BMFSR8Ns.js} +3 -19
  51. package/dist/internal-BMFSR8Ns.js.map +1 -0
  52. package/dist/{judge-calibration-zZjLz8hr.js → judge-calibration-BnpVKtnb.js} +3 -13
  53. package/dist/judge-calibration-BnpVKtnb.js.map +1 -0
  54. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -1
  55. package/dist/{llm-client-hgDieDNN.js → llm-client-BFMRpmqb.js} +8 -11
  56. package/dist/llm-client-BFMRpmqb.js.map +1 -0
  57. package/dist/{llm-judge-BhasIPFT.js → llm-judge-Du7WQPh7.js} +7 -7
  58. package/dist/{llm-judge-BhasIPFT.js.map → llm-judge-Du7WQPh7.js.map} +1 -1
  59. package/dist/meta-eval/index.d.ts +2 -1
  60. package/dist/meta-eval/index.d.ts.map +1 -1
  61. package/dist/meta-eval/index.js +14 -22
  62. package/dist/meta-eval/index.js.map +1 -1
  63. package/dist/openapi.json +1 -1
  64. package/dist/{paired-arms-D-XRF_fy.js → paired-arms-D4aeIHUy.js} +3 -3
  65. package/dist/{paired-arms-D-XRF_fy.js.map → paired-arms-D4aeIHUy.js.map} +1 -1
  66. package/dist/{paired-tests-BHIhYVdu.js → paired-tests-C8iCsioC.js} +3 -3
  67. package/dist/{paired-tests-BHIhYVdu.js.map → paired-tests-C8iCsioC.js.map} +1 -1
  68. package/dist/pipelines/index.d.ts +3 -2
  69. package/dist/pipelines/index.d.ts.map +1 -1
  70. package/dist/pipelines/index.js +4 -19
  71. package/dist/pipelines/index.js.map +1 -1
  72. package/dist/{power-and-mde-CHIrXJll.js → power-and-mde-B8F2RdcD.js} +3 -3
  73. package/dist/{power-and-mde-CHIrXJll.js.map → power-and-mde-B8F2RdcD.js.map} +1 -1
  74. package/dist/{power-preflight-DEw-uC7q.js → power-preflight-CFXm0Vjo.js} +3 -3
  75. package/dist/{power-preflight-DEw-uC7q.js.map → power-preflight-CFXm0Vjo.js.map} +1 -1
  76. package/dist/{produced-state-DZ89riy5.js → produced-state-Be0BK3RN.js} +3 -3
  77. package/dist/{produced-state-DZ89riy5.js.map → produced-state-Be0BK3RN.js.map} +1 -1
  78. package/dist/{promotion-policy-xzA40Evo.js → promotion-policy-LY9mVQ7W.js} +3 -3
  79. package/dist/{promotion-policy-xzA40Evo.js.map → promotion-policy-LY9mVQ7W.js.map} +1 -1
  80. package/dist/{query-DxPYqpmT.d.ts → query-CwnHlu5p.d.ts} +19 -2
  81. package/dist/query-CwnHlu5p.d.ts.map +1 -0
  82. package/dist/{query-CHmMP42p.js → query-_5g6re3_.js} +44 -2
  83. package/dist/query-_5g6re3_.js.map +1 -0
  84. package/dist/random-Dn5fPWkt.js +21 -0
  85. package/dist/random-Dn5fPWkt.js.map +1 -0
  86. package/dist/record-id-DUgsK5qp.js +17 -0
  87. package/dist/record-id-DUgsK5qp.js.map +1 -0
  88. package/dist/{release-confidence-DKfD2RYU.js → release-confidence-nGDJiiwc.js} +2 -2
  89. package/dist/{release-confidence-DKfD2RYU.js.map → release-confidence-nGDJiiwc.js.map} +1 -1
  90. package/dist/reporting.js +6 -6
  91. package/dist/{reward-hacking-t4lB1yt8.js → reward-hacking-O5zKDANP.js} +2 -2
  92. package/dist/{reward-hacking-t4lB1yt8.js.map → reward-hacking-O5zKDANP.js.map} +1 -1
  93. package/dist/rl.d.ts.map +1 -1
  94. package/dist/rl.js +9 -15
  95. package/dist/rl.js.map +1 -1
  96. package/dist/{rubric-predictive-validity-CK8SCOg-.js → rubric-predictive-validity-2D5Gw9z9.js} +3 -3
  97. package/dist/{rubric-predictive-validity-CK8SCOg-.js.map → rubric-predictive-validity-2D5Gw9z9.js.map} +1 -1
  98. package/dist/{semantic-concept-judge-BSkKKHeq.js → semantic-concept-judge-BsDMOwJr.js} +2 -2
  99. package/dist/{semantic-concept-judge-BSkKKHeq.js.map → semantic-concept-judge-BsDMOwJr.js.map} +1 -1
  100. package/dist/{sequential-rYW-Ophm.js → sequential-BLMbdrD7.js} +3 -3
  101. package/dist/{sequential-rYW-Ophm.js.map → sequential-BLMbdrD7.js.map} +1 -1
  102. package/dist/{server-BtFd4uzB.js → server-BjYiJHoJ.js} +2 -2
  103. package/dist/{server-BtFd4uzB.js.map → server-BjYiJHoJ.js.map} +1 -1
  104. package/dist/{skillopt-optimization-method-DbaekMcn.js → skillopt-optimization-method-UArRo-nr.js} +6 -6
  105. package/dist/{skillopt-optimization-method-DbaekMcn.js.map → skillopt-optimization-method-UArRo-nr.js.map} +1 -1
  106. package/dist/{store-tool-spans-Dlh9vkFK.js → store-tool-spans-BVga3c37.js} +21 -17
  107. package/dist/store-tool-spans-BVga3c37.js.map +1 -0
  108. package/dist/store-tool-spans-DPUG7UUY.d.ts.map +1 -1
  109. package/dist/{student-t-CvBq2mve.js → student-t-BA-Uy51p.js} +2 -2
  110. package/dist/{student-t-CvBq2mve.js.map → student-t-BA-Uy51p.js.map} +1 -1
  111. package/dist/{summary-report-BI5hUtvK.js → summary-report-BXeQ5Ues.js} +6 -6
  112. package/dist/{summary-report-BI5hUtvK.js.map → summary-report-BXeQ5Ues.js.map} +1 -1
  113. package/dist/supervisor-run/index.js +1 -1
  114. package/dist/{tool-waste-BqzmVdJk.js → tool-waste-8BQiUc8K.js} +4 -4
  115. package/dist/{tool-waste-BqzmVdJk.js.map → tool-waste-8BQiUc8K.js.map} +1 -1
  116. package/dist/{tool-waste-Dro0gJi3.d.ts → tool-waste-CKc7bYIg.d.ts} +2 -2
  117. package/dist/{tool-waste-Dro0gJi3.d.ts.map → tool-waste-CKc7bYIg.d.ts.map} +1 -1
  118. package/dist/trace-repair/index.js +1 -1
  119. package/dist/traces.d.ts +2 -2
  120. package/dist/traces.d.ts.map +1 -1
  121. package/dist/traces.js +8 -17
  122. package/dist/traces.js.map +1 -1
  123. package/dist/trajectory-replay/index.d.ts.map +1 -1
  124. package/dist/trajectory-replay/index.js +3 -13
  125. package/dist/trajectory-replay/index.js.map +1 -1
  126. package/dist/{types-BPb2Kf_C2.d.ts → types-BPb2Kf_C.d.ts} +1 -1
  127. package/dist/types-BPb2Kf_C.d.ts.map +1 -0
  128. package/dist/types-Bfk0uxRj.d.ts.map +1 -1
  129. package/dist/wire/index.js +1 -1
  130. package/package.json +2 -2
  131. package/dist/benchmark-command-BVtaq_ve.js.map +0 -1
  132. package/dist/emitter-BpYFQPj4.js.map +0 -1
  133. package/dist/internal-BDHPCnjk.js.map +0 -1
  134. package/dist/judge-calibration-zZjLz8hr.js.map +0 -1
  135. package/dist/llm-client-hgDieDNN.js.map +0 -1
  136. package/dist/query-CHmMP42p.js.map +0 -1
  137. package/dist/query-DxPYqpmT.d.ts.map +0 -1
  138. package/dist/store-tool-spans-Dlh9vkFK.js.map +0 -1
  139. package/dist/types-BPb2Kf_C2.d.ts.map +0 -1
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","names":[],"sources":["../../src/pipelines/first-divergence.ts","../../src/pipelines/regression.ts","../../src/pipelines/stuck-loop.ts"],"sourcesContent":["/**\n * FirstDivergenceView — aligns two trajectories by step index, reports\n * the first step where they differ.\n *\n * \"Differ\" is configurable — default is (kind, toolName if tool, model\n * if llm). Use this view to attribute \"why is variant B better?\" to a\n * specific step rather than an aggregate mean delta.\n */\n\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\nexport interface DivergenceReport {\n runA: string\n runB: string\n firstDivergenceIndex: number | null\n aStep?: TrajectoryStep\n bStep?: TrajectoryStep\n reason?: string\n /** Common prefix length (steps that matched). */\n commonPrefixLen: number\n}\n\nexport interface DivergenceOptions {\n /** Returns true if two steps are considered equal. Default: kind + tool/model match. */\n stepEquals?: (a: TrajectoryStep, b: TrajectoryStep) => boolean\n}\n\nexport async function firstDivergenceView(\n store: TraceStore,\n runA: string,\n runB: string,\n options: DivergenceOptions = {},\n): Promise<DivergenceReport> {\n const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)])\n const eq = options.stepEquals ?? defaultStepEquals\n const minLen = Math.min(a.steps.length, b.steps.length)\n for (let i = 0; i < minLen; i++) {\n const aStep = a.steps[i]!\n const bStep = b.steps[i]!\n if (!eq(aStep, bStep)) {\n return {\n runA,\n runB,\n firstDivergenceIndex: i,\n aStep,\n bStep,\n reason: describeDifference(aStep, bStep),\n commonPrefixLen: i,\n }\n }\n }\n if (a.steps.length === b.steps.length) {\n return { runA, runB, firstDivergenceIndex: null, commonPrefixLen: minLen }\n }\n const longer: Trajectory = a.steps.length > b.steps.length ? a : b\n const extra = longer.steps.length - minLen\n // minLen === 0 means one trajectory is empty: divergence is the first step\n // itself (index 0), and the absent side has no step to surface.\n const reason =\n minLen === 0\n ? `one trajectory is empty; the other has ${extra} step(s) starting at index 0`\n : `one trajectory has ${extra} more step(s) after index ${minLen - 1}`\n return {\n runA,\n runB,\n firstDivergenceIndex: minLen,\n aStep: a.steps[minLen],\n bStep: b.steps[minLen],\n reason,\n commonPrefixLen: minLen,\n }\n}\n\nfunction defaultStepEquals(a: TrajectoryStep, b: TrajectoryStep): boolean {\n if (a.span.kind !== b.span.kind) return false\n if (a.span.kind === 'tool' && b.span.kind === 'tool') return a.span.toolName === b.span.toolName\n if (a.span.kind === 'llm' && b.span.kind === 'llm') return a.span.model === b.span.model\n if (a.span.kind === 'judge' && b.span.kind === 'judge')\n return a.span.dimension === b.span.dimension\n return a.span.name === b.span.name\n}\n\nfunction describeDifference(a: TrajectoryStep, b: TrajectoryStep): string {\n if (a.span.kind !== b.span.kind) return `kind ${a.span.kind} vs ${b.span.kind}`\n if (a.span.kind === 'tool' && b.span.kind === 'tool' && a.span.toolName !== b.span.toolName) {\n return `tool ${a.span.toolName} vs ${b.span.toolName}`\n }\n if (a.span.kind === 'llm' && b.span.kind === 'llm' && a.span.model !== b.span.model) {\n return `model ${a.span.model} vs ${b.span.model}`\n }\n return `name \"${a.span.name}\" vs \"${b.span.name}\"`\n}\n","/**\n * RegressionView — compares a candidate slice to a baseline slice on a\n * named metric. Delegates the statistics (Welch's t-test, Cohen's d,\n * IQR stability) to `baseline.ts`.\n *\n * This is the entry point for CI regression gates: \"given runs tagged\n * release=A and release=B, did any metric regress?\"\n */\n\nimport { type BaselineOptions, type BaselineReport, compareToBaseline } from '../baseline'\nimport { aggregateLlm, llmSpans, runFailureClass } from '../trace/query'\nimport type { Run } from '../trace/schema'\nimport type { RunFilter, TraceStore } from '../trace/store'\n\nexport interface RegressionSpec {\n metric: string\n higherIsBetter: boolean\n /** Extract a scalar from a run. Default extractors handle common metrics. */\n extract?: (run: Run, store: TraceStore) => Promise<number | null>\n}\n\nexport interface RegressionOptions extends BaselineOptions {\n baseline: RunFilter\n candidate: RunFilter\n}\n\nexport async function regressionView(\n store: TraceStore,\n metrics: RegressionSpec[],\n options: RegressionOptions,\n): Promise<BaselineReport> {\n const baselineRuns = await store.listRuns(options.baseline)\n const candidateRuns = await store.listRuns(options.candidate)\n const samples = await Promise.all(\n metrics.map(async (m) => {\n const extract = m.extract ?? defaultExtract(m.metric)\n const baseline = await extractAll(baselineRuns, extract, store)\n const candidate = await extractAll(candidateRuns, extract, store)\n return { metric: m.metric, higherIsBetter: m.higherIsBetter, baseline, candidate }\n }),\n )\n return compareToBaseline(samples, options)\n}\n\nasync function extractAll(\n runs: Run[],\n extract: (r: Run, s: TraceStore) => Promise<number | null>,\n store: TraceStore,\n): Promise<number[]> {\n const out: number[] = []\n for (const r of runs) {\n const v = await extract(r, store)\n if (v !== null && Number.isFinite(v)) out.push(v)\n }\n return out\n}\n\nfunction defaultExtract(metric: string): (run: Run, store: TraceStore) => Promise<number | null> {\n return async (run, store) => {\n switch (metric) {\n case 'score':\n case 'overallScore':\n return run.outcome?.score ?? null\n case 'pass':\n return run.outcome?.pass === true ? 1 : 0\n case 'durationMs':\n return run.endedAt && run.startedAt ? run.endedAt - run.startedAt : null\n case 'costUsd': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).costUsd\n }\n case 'inputTokens': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).inputTokens\n }\n case 'outputTokens': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).outputTokens\n }\n case 'failureClass': {\n return runFailureClass(run) === 'success' ? 1 : 0\n }\n default:\n return null\n }\n }\n}\n","/**\n * StuckLoopView — detects when an agent calls the same tool with the\n * same (or structurally similar) arguments ≥ N times in a short window.\n *\n * Rationale: agents that loop are the number-one production failure\n * mode on long-horizon flows. The view returns (runId, toolName,\n * argHash, occurrences, windowMs) for each detected loop plus a\n * fraction of runs affected.\n */\n\nimport { executionTrackByLane } from '../trace/execution-tracks'\nimport { argHash, hasCapturedToolArgs } from '../trace/query'\nimport { isToolSpan, type Span, type ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nconst DEFAULT_MAX_WINDOW_MS = 60_000\nconst DEFAULT_MAX_INTERVENING_TOOL_CALLS = 0\n\ninterface ScopedToolCall {\n span: ToolSpan\n scopeSpanId: string | null\n laneSpanId: string | null\n}\n\ninterface IndexedToolCall extends ScopedToolCall {\n toolCallIndex: number\n trackId: string\n}\n\nexport interface StuckLoopFinding {\n runId: string\n toolName: string\n argHash: string\n /** Calls in this episode's densest qualifying interval, not the whole-run total. */\n occurrences: number\n spanIds: string[]\n /** Nearest agent ancestor, or the direct parent when ancestry is incomplete. */\n scopeSpanId?: string\n /** Milliseconds between first and last call in the loop. */\n windowMs: number\n}\n\nexport interface StuckLoopReport {\n findings: StuckLoopFinding[]\n affectedRunRatio: number\n totalRuns: number\n}\n\nexport interface StuckLoopOptions {\n /** Minimum call count to flag a loop (default 3). */\n minOccurrences?: number\n /** Maximum time between the first and last repeated call (default 60 seconds). */\n maxWindowMs?: number\n /**\n * Maximum other tool calls allowed between adjacent repeats (default 0).\n * Set to 1 to detect alternating patterns such as A,B,A,B,A.\n */\n maxInterveningToolCalls?: number\n /** Filter to a specific runId; omit to scan the entire corpus. */\n runId?: string\n}\n\nexport async function stuckLoopView(\n store: TraceStore,\n options: StuckLoopOptions = {},\n): Promise<StuckLoopReport> {\n const minOccurrences = options.minOccurrences ?? 3\n const maxWindowMs = options.maxWindowMs ?? DEFAULT_MAX_WINDOW_MS\n const maxInterveningToolCalls =\n options.maxInterveningToolCalls ?? DEFAULT_MAX_INTERVENING_TOOL_CALLS\n if (!Number.isInteger(minOccurrences) || minOccurrences < 1) {\n throw new RangeError('minOccurrences must be a positive integer')\n }\n if (!Number.isFinite(maxWindowMs) || maxWindowMs < 0) {\n throw new RangeError('maxWindowMs must be a finite non-negative number')\n }\n if (!Number.isInteger(maxInterveningToolCalls) || maxInterveningToolCalls < 0) {\n throw new RangeError('maxInterveningToolCalls must be a non-negative integer')\n }\n const runs = options.runId\n ? [{ runId: options.runId }]\n : (await store.listRuns()).map((r) => ({ runId: r.runId }))\n\n const findings: StuckLoopFinding[] = []\n for (const { runId } of runs) {\n const spans = await store.spans({ runId })\n const spansById = new Map(spans.map((span) => [span.spanId, span]))\n const scopedTools: ScopedToolCall[] = spans\n .filter(isToolSpan)\n .map((span, sourceIndex) => ({ span, sourceIndex }))\n .sort((a, b) => a.span.startedAt - b.span.startedAt || a.sourceIndex - b.sourceIndex)\n .map(({ span }) => ({ span, ...executionScope(span, spansById) }))\n const trackByLane = executionTrackByLane(\n scopedTools.map((call) => {\n const direct = call.laneSpanId === null || call.laneSpanId === call.scopeSpanId\n const timed = direct\n ? call.span\n : call.laneSpanId\n ? spansById.get(call.laneSpanId)\n : undefined\n return {\n key: executionKey(call),\n scopeKey: JSON.stringify(call.scopeSpanId),\n start: timed?.startedAt ?? null,\n end: timed?.endedAt ?? null,\n }\n }),\n )\n const nextToolIndexByTrack = new Map<string, number>()\n const orderedTools: IndexedToolCall[] = scopedTools.map((call) => {\n const trackId = trackByLane.get(executionKey(call))!\n const toolCallIndex = nextToolIndexByTrack.get(trackId) ?? 0\n nextToolIndexByTrack.set(trackId, toolCallIndex + 1)\n return { ...call, toolCallIndex, trackId }\n })\n const byKey = new Map<\n string,\n {\n calls: IndexedToolCall[]\n argHash: string\n toolName: string\n scopeSpanId: string | null\n }\n >()\n for (const call of orderedTools) {\n if (!hasCapturedToolArgs(call.span)) continue\n const h = argHash(call.span.args)\n const key = JSON.stringify([call.trackId, call.span.toolName, h])\n const bucket = byKey.get(key) ?? {\n calls: [],\n argHash: h,\n toolName: call.span.toolName,\n scopeSpanId: call.scopeSpanId,\n }\n bucket.calls.push(call)\n byKey.set(key, bucket)\n }\n for (const { calls, argHash: h, toolName, scopeSpanId } of byKey.values()) {\n if (calls.length < minOccurrences) continue\n let episodeStart = 0\n for (let episodeEnd = 1; episodeEnd <= calls.length; episodeEnd += 1) {\n const previous = calls[episodeEnd - 1]!\n const next = calls[episodeEnd]\n const episodeEnded =\n next === undefined ||\n next.span.startedAt - previous.span.startedAt > maxWindowMs ||\n next.toolCallIndex - previous.toolCallIndex - 1 > maxInterveningToolCalls ||\n !callsAreSerial(previous, next)\n if (!episodeEnded) continue\n\n const episode = calls.slice(episodeStart, episodeEnd)\n let left = 0\n let bestStart = 0\n let bestEnd = -1\n for (let right = 0; right < episode.length; right += 1) {\n while (episode[right]!.span.startedAt - episode[left]!.span.startedAt > maxWindowMs) {\n left += 1\n }\n if (right - left > bestEnd - bestStart) {\n bestStart = left\n bestEnd = right\n }\n }\n if (bestEnd - bestStart + 1 >= minOccurrences) {\n const loop = episode.slice(bestStart, bestEnd + 1)\n const first = loop[0]!.span.startedAt\n const last = loop[loop.length - 1]!.span.startedAt\n findings.push({\n runId,\n toolName,\n argHash: h,\n occurrences: loop.length,\n spanIds: loop.map((call) => call.span.spanId),\n ...(scopeSpanId ? { scopeSpanId } : {}),\n windowMs: last - first,\n })\n }\n episodeStart = episodeEnd\n }\n }\n }\n\n const affectedRuns = new Set(findings.map((f) => f.runId))\n return {\n findings,\n affectedRunRatio: runs.length > 0 ? affectedRuns.size / runs.length : 0,\n totalRuns: runs.length,\n }\n}\n\nfunction laneKey(call: Pick<ScopedToolCall, 'scopeSpanId' | 'laneSpanId'>): string {\n return JSON.stringify([call.scopeSpanId, call.laneSpanId])\n}\n\nfunction executionKey(call: ScopedToolCall): string {\n return call.laneSpanId === null || call.laneSpanId === call.scopeSpanId\n ? JSON.stringify([call.scopeSpanId, call.span.spanId])\n : laneKey(call)\n}\n\nfunction executionScope(\n span: ToolSpan,\n spansById: ReadonlyMap<string, Span>,\n): { scopeSpanId: string | null; laneSpanId: string | null } {\n const directParent = span.parentSpanId\n if (!directParent) return { scopeSpanId: null, laneSpanId: null }\n\n let currentId: string | undefined = directParent\n let laneSpanId: string | null = null\n const seen = new Set<string>()\n while (currentId && !seen.has(currentId)) {\n seen.add(currentId)\n const current = spansById.get(currentId)\n if (!current) return { scopeSpanId: directParent, laneSpanId: directParent }\n if (current.kind === 'agent') {\n return { scopeSpanId: current.spanId, laneSpanId: laneSpanId ?? current.spanId }\n }\n laneSpanId = current.spanId\n currentId = current.parentSpanId\n }\n return { scopeSpanId: directParent, laneSpanId: directParent }\n}\n\nfunction callsAreSerial(previous: IndexedToolCall, next: IndexedToolCall): boolean {\n return previous.span.endedAt !== undefined && previous.span.endedAt <= next.span.startedAt\n}\n"],"mappings":";;;;;;;AA4BA,eAAsB,oBACpB,OACA,MACA,MACA,UAA6B,CAAC,GACH;CAC3B,MAAM,CAAC,GAAG,KAAK,MAAM,QAAQ,IAAI,CAAC,gBAAgB,OAAO,IAAI,GAAG,gBAAgB,OAAO,IAAI,CAAC,CAAC;CAC7F,MAAM,KAAK,QAAQ,cAAc;CACjC,MAAM,SAAS,KAAK,IAAI,EAAE,MAAM,QAAQ,EAAE,MAAM,MAAM;CACtD,KAAK,IAAI,IAAI,GAAG,IAAI,QAAQ,KAAK;EAC/B,MAAM,QAAQ,EAAE,MAAM;EACtB,MAAM,QAAQ,EAAE,MAAM;EACtB,IAAI,CAAC,GAAG,OAAO,KAAK,GAClB,OAAO;GACL;GACA;GACA,sBAAsB;GACtB;GACA;GACA,QAAQ,mBAAmB,OAAO,KAAK;GACvC,iBAAiB;EACnB;CAEJ;CACA,IAAI,EAAE,MAAM,WAAW,EAAE,MAAM,QAC7B,OAAO;EAAE;EAAM;EAAM,sBAAsB;EAAM,iBAAiB;CAAO;CAG3E,MAAM,SADqB,EAAE,MAAM,SAAS,EAAE,MAAM,SAAS,IAAI,EAAA,CAC5C,MAAM,SAAS;CAGpC,MAAM,SACJ,WAAW,IACP,0CAA0C,MAAM,gCAChD,sBAAsB,MAAM,4BAA4B,SAAS;CACvE,OAAO;EACL;EACA;EACA,sBAAsB;EACtB,OAAO,EAAE,MAAM;EACf,OAAO,EAAE,MAAM;EACf;EACA,iBAAiB;CACnB;AACF;AAEA,SAAS,kBAAkB,GAAmB,GAA4B;CACxE,IAAI,EAAE,KAAK,SAAS,EAAE,KAAK,MAAM,OAAO;CACxC,IAAI,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,SAAS,QAAQ,OAAO,EAAE,KAAK,aAAa,EAAE,KAAK;CACxF,IAAI,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,OAAO,OAAO,EAAE,KAAK,UAAU,EAAE,KAAK;CACnF,IAAI,EAAE,KAAK,SAAS,WAAW,EAAE,KAAK,SAAS,SAC7C,OAAO,EAAE,KAAK,cAAc,EAAE,KAAK;CACrC,OAAO,EAAE,KAAK,SAAS,EAAE,KAAK;AAChC;AAEA,SAAS,mBAAmB,GAAmB,GAA2B;CACxE,IAAI,EAAE,KAAK,SAAS,EAAE,KAAK,MAAM,OAAO,QAAQ,EAAE,KAAK,KAAK,MAAM,EAAE,KAAK;CACzE,IAAI,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,aAAa,EAAE,KAAK,UACjF,OAAO,QAAQ,EAAE,KAAK,SAAS,MAAM,EAAE,KAAK;CAE9C,IAAI,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,UAAU,EAAE,KAAK,OAC5E,OAAO,SAAS,EAAE,KAAK,MAAM,MAAM,EAAE,KAAK;CAE5C,OAAO,SAAS,EAAE,KAAK,KAAK,QAAQ,EAAE,KAAK,KAAK;AAClD;;;;;;;;;;;AClEA,eAAsB,eACpB,OACA,SACA,SACyB;CACzB,MAAM,eAAe,MAAM,MAAM,SAAS,QAAQ,QAAQ;CAC1D,MAAM,gBAAgB,MAAM,MAAM,SAAS,QAAQ,SAAS;CAS5D,OAAO,kBAAkB,MARH,QAAQ,IAC5B,QAAQ,IAAI,OAAO,MAAM;EACvB,MAAM,UAAU,EAAE,WAAW,eAAe,EAAE,MAAM;EACpD,MAAM,WAAW,MAAM,WAAW,cAAc,SAAS,KAAK;EAC9D,MAAM,YAAY,MAAM,WAAW,eAAe,SAAS,KAAK;EAChE,OAAO;GAAE,QAAQ,EAAE;GAAQ,gBAAgB,EAAE;GAAgB;GAAU;EAAU;CACnF,CAAC,CACH,GACkC,OAAO;AAC3C;AAEA,eAAe,WACb,MACA,SACA,OACmB;CACnB,MAAM,MAAgB,CAAC;CACvB,KAAK,MAAM,KAAK,MAAM;EACpB,MAAM,IAAI,MAAM,QAAQ,GAAG,KAAK;EAChC,IAAI,MAAM,QAAQ,OAAO,SAAS,CAAC,GAAG,IAAI,KAAK,CAAC;CAClD;CACA,OAAO;AACT;AAEA,SAAS,eAAe,QAAyE;CAC/F,OAAO,OAAO,KAAK,UAAU;EAC3B,QAAQ,QAAR;GACE,KAAK;GACL,KAAK,gBACH,OAAO,IAAI,SAAS,SAAS;GAC/B,KAAK,QACH,OAAO,IAAI,SAAS,SAAS,OAAO,IAAI;GAC1C,KAAK,cACH,OAAO,IAAI,WAAW,IAAI,YAAY,IAAI,UAAU,IAAI,YAAY;GACtE,KAAK,WAEH,OAAO,aAAa,MADF,SAAS,OAAO,IAAI,KAAK,CACpB,CAAC,CAAC;GAE3B,KAAK,eAEH,OAAO,aAAa,MADF,SAAS,OAAO,IAAI,KAAK,CACpB,CAAC,CAAC;GAE3B,KAAK,gBAEH,OAAO,aAAa,MADF,SAAS,OAAO,IAAI,KAAK,CACpB,CAAC,CAAC;GAE3B,KAAK,gBACH,OAAO,gBAAgB,GAAG,MAAM,YAAY,IAAI;GAElD,SACE,OAAO;EACX;CACF;AACF;;;;;;;;;;;;ACvEA,MAAM,wBAAwB;AAC9B,MAAM,qCAAqC;AA8C3C,eAAsB,cACpB,OACA,UAA4B,CAAC,GACH;CAC1B,MAAM,iBAAiB,QAAQ,kBAAkB;CACjD,MAAM,cAAc,QAAQ,eAAe;CAC3C,MAAM,0BACJ,QAAQ,2BAA2B;CACrC,IAAI,CAAC,OAAO,UAAU,cAAc,KAAK,iBAAiB,GACxD,MAAM,IAAI,WAAW,2CAA2C;CAElE,IAAI,CAAC,OAAO,SAAS,WAAW,KAAK,cAAc,GACjD,MAAM,IAAI,WAAW,kDAAkD;CAEzE,IAAI,CAAC,OAAO,UAAU,uBAAuB,KAAK,0BAA0B,GAC1E,MAAM,IAAI,WAAW,wDAAwD;CAE/E,MAAM,OAAO,QAAQ,QACjB,CAAC,EAAE,OAAO,QAAQ,MAAM,CAAC,KACxB,MAAM,MAAM,SAAS,EAAA,CAAG,KAAK,OAAO,EAAE,OAAO,EAAE,MAAM,EAAE;CAE5D,MAAM,WAA+B,CAAC;CACtC,KAAK,MAAM,EAAE,WAAW,MAAM;EAC5B,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,MAAM,CAAC;EACzC,MAAM,YAAY,IAAI,IAAI,MAAM,KAAK,SAAS,CAAC,KAAK,QAAQ,IAAI,CAAC,CAAC;EAClE,MAAM,cAAgC,MACnC,OAAO,UAAU,CAAC,CAClB,KAAK,MAAM,iBAAiB;GAAE;GAAM;EAAY,EAAE,CAAC,CACnD,MAAM,GAAG,MAAM,EAAE,KAAK,YAAY,EAAE,KAAK,aAAa,EAAE,cAAc,EAAE,WAAW,CAAC,CACpF,KAAK,EAAE,YAAY;GAAE;GAAM,GAAG,eAAe,MAAM,SAAS;EAAE,EAAE;EACnE,MAAM,cAAc,qBAClB,YAAY,KAAK,SAAS;GAExB,MAAM,QADS,KAAK,eAAe,QAAQ,KAAK,eAAe,KAAK,cAEhE,KAAK,OACL,KAAK,aACH,UAAU,IAAI,KAAK,UAAU,IAC7B,KAAA;GACN,OAAO;IACL,KAAK,aAAa,IAAI;IACtB,UAAU,KAAK,UAAU,KAAK,WAAW;IACzC,OAAO,OAAO,aAAa;IAC3B,KAAK,OAAO,WAAW;GACzB;EACF,CAAC,CACH;EACA,MAAM,uCAAuB,IAAI,IAAoB;EACrD,MAAM,eAAkC,YAAY,KAAK,SAAS;GAChE,MAAM,UAAU,YAAY,IAAI,aAAa,IAAI,CAAC;GAClD,MAAM,gBAAgB,qBAAqB,IAAI,OAAO,KAAK;GAC3D,qBAAqB,IAAI,SAAS,gBAAgB,CAAC;GACnD,OAAO;IAAE,GAAG;IAAM;IAAe;GAAQ;EAC3C,CAAC;EACD,MAAM,wBAAQ,IAAI,IAQhB;EACF,KAAK,MAAM,QAAQ,cAAc;GAC/B,IAAI,CAAC,oBAAoB,KAAK,IAAI,GAAG;GACrC,MAAM,IAAI,QAAQ,KAAK,KAAK,IAAI;GAChC,MAAM,MAAM,KAAK,UAAU;IAAC,KAAK;IAAS,KAAK,KAAK;IAAU;GAAC,CAAC;GAChE,MAAM,SAAS,MAAM,IAAI,GAAG,KAAK;IAC/B,OAAO,CAAC;IACR,SAAS;IACT,UAAU,KAAK,KAAK;IACpB,aAAa,KAAK;GACpB;GACA,OAAO,MAAM,KAAK,IAAI;GACtB,MAAM,IAAI,KAAK,MAAM;EACvB;EACA,KAAK,MAAM,EAAE,OAAO,SAAS,GAAG,UAAU,iBAAiB,MAAM,OAAO,GAAG;GACzE,IAAI,MAAM,SAAS,gBAAgB;GACnC,IAAI,eAAe;GACnB,KAAK,IAAI,aAAa,GAAG,cAAc,MAAM,QAAQ,cAAc,GAAG;IACpE,MAAM,WAAW,MAAM,aAAa;IACpC,MAAM,OAAO,MAAM;IAMnB,IAAI,EAJF,SAAS,KAAA,KACT,KAAK,KAAK,YAAY,SAAS,KAAK,YAAY,eAChD,KAAK,gBAAgB,SAAS,gBAAgB,IAAI,2BAClD,CAAC,eAAe,UAAU,IAAI,IACb;IAEnB,MAAM,UAAU,MAAM,MAAM,cAAc,UAAU;IACpD,IAAI,OAAO;IACX,IAAI,YAAY;IAChB,IAAI,UAAU;IACd,KAAK,IAAI,QAAQ,GAAG,QAAQ,QAAQ,QAAQ,SAAS,GAAG;KACtD,OAAO,QAAQ,MAAM,CAAE,KAAK,YAAY,QAAQ,KAAK,CAAE,KAAK,YAAY,aACtE,QAAQ;KAEV,IAAI,QAAQ,OAAO,UAAU,WAAW;MACtC,YAAY;MACZ,UAAU;KACZ;IACF;IACA,IAAI,UAAU,YAAY,KAAK,gBAAgB;KAC7C,MAAM,OAAO,QAAQ,MAAM,WAAW,UAAU,CAAC;KACjD,MAAM,QAAQ,KAAK,EAAE,CAAE,KAAK;KAC5B,MAAM,OAAO,KAAK,KAAK,SAAS,EAAE,CAAE,KAAK;KACzC,SAAS,KAAK;MACZ;MACA;MACA,SAAS;MACT,aAAa,KAAK;MAClB,SAAS,KAAK,KAAK,SAAS,KAAK,KAAK,MAAM;MAC5C,GAAI,cAAc,EAAE,YAAY,IAAI,CAAC;MACrC,UAAU,OAAO;KACnB,CAAC;IACH;IACA,eAAe;GACjB;EACF;CACF;CAEA,MAAM,eAAe,IAAI,IAAI,SAAS,KAAK,MAAM,EAAE,KAAK,CAAC;CACzD,OAAO;EACL;EACA,kBAAkB,KAAK,SAAS,IAAI,aAAa,OAAO,KAAK,SAAS;EACtE,WAAW,KAAK;CAClB;AACF;AAEA,SAAS,QAAQ,MAAkE;CACjF,OAAO,KAAK,UAAU,CAAC,KAAK,aAAa,KAAK,UAAU,CAAC;AAC3D;AAEA,SAAS,aAAa,MAA8B;CAClD,OAAO,KAAK,eAAe,QAAQ,KAAK,eAAe,KAAK,cACxD,KAAK,UAAU,CAAC,KAAK,aAAa,KAAK,KAAK,MAAM,CAAC,IACnD,QAAQ,IAAI;AAClB;AAEA,SAAS,eACP,MACA,WAC2D;CAC3D,MAAM,eAAe,KAAK;CAC1B,IAAI,CAAC,cAAc,OAAO;EAAE,aAAa;EAAM,YAAY;CAAK;CAEhE,IAAI,YAAgC;CACpC,IAAI,aAA4B;CAChC,MAAM,uBAAO,IAAI,IAAY;CAC7B,OAAO,aAAa,CAAC,KAAK,IAAI,SAAS,GAAG;EACxC,KAAK,IAAI,SAAS;EAClB,MAAM,UAAU,UAAU,IAAI,SAAS;EACvC,IAAI,CAAC,SAAS,OAAO;GAAE,aAAa;GAAc,YAAY;EAAa;EAC3E,IAAI,QAAQ,SAAS,SACnB,OAAO;GAAE,aAAa,QAAQ;GAAQ,YAAY,cAAc,QAAQ;EAAO;EAEjF,aAAa,QAAQ;EACrB,YAAY,QAAQ;CACtB;CACA,OAAO;EAAE,aAAa;EAAc,YAAY;CAAa;AAC/D;AAEA,SAAS,eAAe,UAA2B,MAAgC;CACjF,OAAO,SAAS,KAAK,YAAY,KAAA,KAAa,SAAS,KAAK,WAAW,KAAK,KAAK;AACnF"}
1
+ {"version":3,"file":"index.js","names":[],"sources":["../../src/pipelines/first-divergence.ts","../../src/pipelines/regression.ts","../../src/pipelines/stuck-loop.ts"],"sourcesContent":["/**\n * FirstDivergenceView — aligns two trajectories by step index, reports\n * the first step where they differ.\n *\n * \"Differ\" is configurable — default is (kind, toolName if tool, model\n * if llm). Use this view to attribute \"why is variant B better?\" to a\n * specific step rather than an aggregate mean delta.\n */\n\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\nexport interface DivergenceReport {\n runA: string\n runB: string\n firstDivergenceIndex: number | null\n aStep?: TrajectoryStep\n bStep?: TrajectoryStep\n reason?: string\n /** Common prefix length (steps that matched). */\n commonPrefixLen: number\n}\n\nexport interface DivergenceOptions {\n /** Returns true if two steps are considered equal. Default: kind + tool/model match. */\n stepEquals?: (a: TrajectoryStep, b: TrajectoryStep) => boolean\n}\n\nexport async function firstDivergenceView(\n store: TraceStore,\n runA: string,\n runB: string,\n options: DivergenceOptions = {},\n): Promise<DivergenceReport> {\n const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)])\n const eq = options.stepEquals ?? defaultStepEquals\n const minLen = Math.min(a.steps.length, b.steps.length)\n for (let i = 0; i < minLen; i++) {\n const aStep = a.steps[i]!\n const bStep = b.steps[i]!\n if (!eq(aStep, bStep)) {\n return {\n runA,\n runB,\n firstDivergenceIndex: i,\n aStep,\n bStep,\n reason: describeDifference(aStep, bStep),\n commonPrefixLen: i,\n }\n }\n }\n if (a.steps.length === b.steps.length) {\n return { runA, runB, firstDivergenceIndex: null, commonPrefixLen: minLen }\n }\n const longer: Trajectory = a.steps.length > b.steps.length ? a : b\n const extra = longer.steps.length - minLen\n // minLen === 0 means one trajectory is empty: divergence is the first step\n // itself (index 0), and the absent side has no step to surface.\n const reason =\n minLen === 0\n ? `one trajectory is empty; the other has ${extra} step(s) starting at index 0`\n : `one trajectory has ${extra} more step(s) after index ${minLen - 1}`\n return {\n runA,\n runB,\n firstDivergenceIndex: minLen,\n aStep: a.steps[minLen],\n bStep: b.steps[minLen],\n reason,\n commonPrefixLen: minLen,\n }\n}\n\nfunction defaultStepEquals(a: TrajectoryStep, b: TrajectoryStep): boolean {\n if (a.span.kind !== b.span.kind) return false\n if (a.span.kind === 'tool' && b.span.kind === 'tool') return a.span.toolName === b.span.toolName\n if (a.span.kind === 'llm' && b.span.kind === 'llm') return a.span.model === b.span.model\n if (a.span.kind === 'judge' && b.span.kind === 'judge')\n return a.span.dimension === b.span.dimension\n return a.span.name === b.span.name\n}\n\nfunction describeDifference(a: TrajectoryStep, b: TrajectoryStep): string {\n if (a.span.kind !== b.span.kind) return `kind ${a.span.kind} vs ${b.span.kind}`\n if (a.span.kind === 'tool' && b.span.kind === 'tool' && a.span.toolName !== b.span.toolName) {\n return `tool ${a.span.toolName} vs ${b.span.toolName}`\n }\n if (a.span.kind === 'llm' && b.span.kind === 'llm' && a.span.model !== b.span.model) {\n return `model ${a.span.model} vs ${b.span.model}`\n }\n return `name \"${a.span.name}\" vs \"${b.span.name}\"`\n}\n","/**\n * RegressionView — compares a candidate slice to a baseline slice on a\n * named metric. Delegates the statistics (Welch's t-test, Cohen's d,\n * IQR stability) to `baseline.ts`.\n *\n * This is the entry point for CI regression gates: \"given runs tagged\n * release=A and release=B, did any metric regress?\"\n */\n\nimport { type BaselineOptions, type BaselineReport, compareToBaseline } from '../baseline'\nimport { runMetricExtractor } from '../trace/query'\nimport type { Run } from '../trace/schema'\nimport type { RunFilter, TraceStore } from '../trace/store'\n\nexport interface RegressionSpec {\n metric: string\n higherIsBetter: boolean\n /** Extract a scalar from a run. Omit it and `metric` must name one of\n * `RUN_METRICS`; any other name is refused. */\n extract?: (run: Run, store: TraceStore) => Promise<number | null>\n}\n\nexport interface RegressionOptions extends BaselineOptions {\n baseline: RunFilter\n candidate: RunFilter\n}\n\nexport async function regressionView(\n store: TraceStore,\n metrics: RegressionSpec[],\n options: RegressionOptions,\n): Promise<BaselineReport> {\n const baselineRuns = await store.listRuns(options.baseline)\n const candidateRuns = await store.listRuns(options.candidate)\n const samples = await Promise.all(\n metrics.map(async (m) => {\n const extract = m.extract ?? runMetricExtractor(m.metric)\n const baseline = await extractAll(baselineRuns, extract, store)\n const candidate = await extractAll(candidateRuns, extract, store)\n return { metric: m.metric, higherIsBetter: m.higherIsBetter, baseline, candidate }\n }),\n )\n return compareToBaseline(samples, options)\n}\n\nasync function extractAll(\n runs: Run[],\n extract: (r: Run, s: TraceStore) => Promise<number | null>,\n store: TraceStore,\n): Promise<number[]> {\n const out: number[] = []\n for (const r of runs) {\n const v = await extract(r, store)\n if (v !== null && Number.isFinite(v)) out.push(v)\n }\n return out\n}\n","/**\n * StuckLoopView — detects when an agent calls the same tool with the\n * same (or structurally similar) arguments ≥ N times in a short window.\n *\n * Rationale: agents that loop are the number-one production failure\n * mode on long-horizon flows. The view returns (runId, toolName,\n * argHash, occurrences, windowMs) for each detected loop plus a\n * fraction of runs affected.\n */\n\nimport { executionTrackByLane } from '../trace/execution-tracks'\nimport { argHash, hasCapturedToolArgs } from '../trace/query'\nimport { isToolSpan, type Span, type ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nconst DEFAULT_MAX_WINDOW_MS = 60_000\nconst DEFAULT_MAX_INTERVENING_TOOL_CALLS = 0\n\ninterface ScopedToolCall {\n span: ToolSpan\n scopeSpanId: string | null\n laneSpanId: string | null\n}\n\ninterface IndexedToolCall extends ScopedToolCall {\n toolCallIndex: number\n trackId: string\n}\n\nexport interface StuckLoopFinding {\n runId: string\n toolName: string\n argHash: string\n /** Calls in this episode's densest qualifying interval, not the whole-run total. */\n occurrences: number\n spanIds: string[]\n /** Nearest agent ancestor, or the direct parent when ancestry is incomplete. */\n scopeSpanId?: string\n /** Milliseconds between first and last call in the loop. */\n windowMs: number\n}\n\nexport interface StuckLoopReport {\n findings: StuckLoopFinding[]\n affectedRunRatio: number\n totalRuns: number\n}\n\nexport interface StuckLoopOptions {\n /** Minimum call count to flag a loop (default 3). */\n minOccurrences?: number\n /** Maximum time between the first and last repeated call (default 60 seconds). */\n maxWindowMs?: number\n /**\n * Maximum other tool calls allowed between adjacent repeats (default 0).\n * Set to 1 to detect alternating patterns such as A,B,A,B,A.\n */\n maxInterveningToolCalls?: number\n /** Filter to a specific runId; omit to scan the entire corpus. */\n runId?: string\n}\n\nexport async function stuckLoopView(\n store: TraceStore,\n options: StuckLoopOptions = {},\n): Promise<StuckLoopReport> {\n const minOccurrences = options.minOccurrences ?? 3\n const maxWindowMs = options.maxWindowMs ?? DEFAULT_MAX_WINDOW_MS\n const maxInterveningToolCalls =\n options.maxInterveningToolCalls ?? DEFAULT_MAX_INTERVENING_TOOL_CALLS\n if (!Number.isInteger(minOccurrences) || minOccurrences < 1) {\n throw new RangeError('minOccurrences must be a positive integer')\n }\n if (!Number.isFinite(maxWindowMs) || maxWindowMs < 0) {\n throw new RangeError('maxWindowMs must be a finite non-negative number')\n }\n if (!Number.isInteger(maxInterveningToolCalls) || maxInterveningToolCalls < 0) {\n throw new RangeError('maxInterveningToolCalls must be a non-negative integer')\n }\n const runs = options.runId\n ? [{ runId: options.runId }]\n : (await store.listRuns()).map((r) => ({ runId: r.runId }))\n\n const findings: StuckLoopFinding[] = []\n for (const { runId } of runs) {\n const spans = await store.spans({ runId })\n const spansById = new Map(spans.map((span) => [span.spanId, span]))\n const scopedTools: ScopedToolCall[] = spans\n .filter(isToolSpan)\n .map((span, sourceIndex) => ({ span, sourceIndex }))\n .sort((a, b) => a.span.startedAt - b.span.startedAt || a.sourceIndex - b.sourceIndex)\n .map(({ span }) => ({ span, ...executionScope(span, spansById) }))\n const trackByLane = executionTrackByLane(\n scopedTools.map((call) => {\n const direct = call.laneSpanId === null || call.laneSpanId === call.scopeSpanId\n const timed = direct\n ? call.span\n : call.laneSpanId\n ? spansById.get(call.laneSpanId)\n : undefined\n return {\n key: executionKey(call),\n scopeKey: JSON.stringify(call.scopeSpanId),\n start: timed?.startedAt ?? null,\n end: timed?.endedAt ?? null,\n }\n }),\n )\n const nextToolIndexByTrack = new Map<string, number>()\n const orderedTools: IndexedToolCall[] = scopedTools.map((call) => {\n const trackId = trackByLane.get(executionKey(call))!\n const toolCallIndex = nextToolIndexByTrack.get(trackId) ?? 0\n nextToolIndexByTrack.set(trackId, toolCallIndex + 1)\n return { ...call, toolCallIndex, trackId }\n })\n const byKey = new Map<\n string,\n {\n calls: IndexedToolCall[]\n argHash: string\n toolName: string\n scopeSpanId: string | null\n }\n >()\n for (const call of orderedTools) {\n if (!hasCapturedToolArgs(call.span)) continue\n const h = argHash(call.span.args)\n const key = JSON.stringify([call.trackId, call.span.toolName, h])\n const bucket = byKey.get(key) ?? {\n calls: [],\n argHash: h,\n toolName: call.span.toolName,\n scopeSpanId: call.scopeSpanId,\n }\n bucket.calls.push(call)\n byKey.set(key, bucket)\n }\n for (const { calls, argHash: h, toolName, scopeSpanId } of byKey.values()) {\n if (calls.length < minOccurrences) continue\n let episodeStart = 0\n for (let episodeEnd = 1; episodeEnd <= calls.length; episodeEnd += 1) {\n const previous = calls[episodeEnd - 1]!\n const next = calls[episodeEnd]\n const episodeEnded =\n next === undefined ||\n next.span.startedAt - previous.span.startedAt > maxWindowMs ||\n next.toolCallIndex - previous.toolCallIndex - 1 > maxInterveningToolCalls ||\n !callsAreSerial(previous, next)\n if (!episodeEnded) continue\n\n const episode = calls.slice(episodeStart, episodeEnd)\n let left = 0\n let bestStart = 0\n let bestEnd = -1\n for (let right = 0; right < episode.length; right += 1) {\n while (episode[right]!.span.startedAt - episode[left]!.span.startedAt > maxWindowMs) {\n left += 1\n }\n if (right - left > bestEnd - bestStart) {\n bestStart = left\n bestEnd = right\n }\n }\n if (bestEnd - bestStart + 1 >= minOccurrences) {\n const loop = episode.slice(bestStart, bestEnd + 1)\n const first = loop[0]!.span.startedAt\n const last = loop[loop.length - 1]!.span.startedAt\n findings.push({\n runId,\n toolName,\n argHash: h,\n occurrences: loop.length,\n spanIds: loop.map((call) => call.span.spanId),\n ...(scopeSpanId ? { scopeSpanId } : {}),\n windowMs: last - first,\n })\n }\n episodeStart = episodeEnd\n }\n }\n }\n\n const affectedRuns = new Set(findings.map((f) => f.runId))\n return {\n findings,\n affectedRunRatio: runs.length > 0 ? affectedRuns.size / runs.length : 0,\n totalRuns: runs.length,\n }\n}\n\nfunction laneKey(call: Pick<ScopedToolCall, 'scopeSpanId' | 'laneSpanId'>): string {\n return JSON.stringify([call.scopeSpanId, call.laneSpanId])\n}\n\nfunction executionKey(call: ScopedToolCall): string {\n return call.laneSpanId === null || call.laneSpanId === call.scopeSpanId\n ? JSON.stringify([call.scopeSpanId, call.span.spanId])\n : laneKey(call)\n}\n\nfunction executionScope(\n span: ToolSpan,\n spansById: ReadonlyMap<string, Span>,\n): { scopeSpanId: string | null; laneSpanId: string | null } {\n const directParent = span.parentSpanId\n if (!directParent) return { scopeSpanId: null, laneSpanId: null }\n\n let currentId: string | undefined = directParent\n let laneSpanId: string | null = null\n const seen = new Set<string>()\n while (currentId && !seen.has(currentId)) {\n seen.add(currentId)\n const current = spansById.get(currentId)\n if (!current) return { scopeSpanId: directParent, laneSpanId: directParent }\n if (current.kind === 'agent') {\n return { scopeSpanId: current.spanId, laneSpanId: laneSpanId ?? current.spanId }\n }\n laneSpanId = current.spanId\n currentId = current.parentSpanId\n }\n return { scopeSpanId: directParent, laneSpanId: directParent }\n}\n\nfunction callsAreSerial(previous: IndexedToolCall, next: IndexedToolCall): boolean {\n return previous.span.endedAt !== undefined && previous.span.endedAt <= next.span.startedAt\n}\n"],"mappings":";;;;;;;AA4BA,eAAsB,oBACpB,OACA,MACA,MACA,UAA6B,CAAC,GACH;CAC3B,MAAM,CAAC,GAAG,KAAK,MAAM,QAAQ,IAAI,CAAC,gBAAgB,OAAO,IAAI,GAAG,gBAAgB,OAAO,IAAI,CAAC,CAAC;CAC7F,MAAM,KAAK,QAAQ,cAAc;CACjC,MAAM,SAAS,KAAK,IAAI,EAAE,MAAM,QAAQ,EAAE,MAAM,MAAM;CACtD,KAAK,IAAI,IAAI,GAAG,IAAI,QAAQ,KAAK;EAC/B,MAAM,QAAQ,EAAE,MAAM;EACtB,MAAM,QAAQ,EAAE,MAAM;EACtB,IAAI,CAAC,GAAG,OAAO,KAAK,GAClB,OAAO;GACL;GACA;GACA,sBAAsB;GACtB;GACA;GACA,QAAQ,mBAAmB,OAAO,KAAK;GACvC,iBAAiB;EACnB;CAEJ;CACA,IAAI,EAAE,MAAM,WAAW,EAAE,MAAM,QAC7B,OAAO;EAAE;EAAM;EAAM,sBAAsB;EAAM,iBAAiB;CAAO;CAG3E,MAAM,SADqB,EAAE,MAAM,SAAS,EAAE,MAAM,SAAS,IAAI,EAAA,CAC5C,MAAM,SAAS;CAGpC,MAAM,SACJ,WAAW,IACP,0CAA0C,MAAM,gCAChD,sBAAsB,MAAM,4BAA4B,SAAS;CACvE,OAAO;EACL;EACA;EACA,sBAAsB;EACtB,OAAO,EAAE,MAAM;EACf,OAAO,EAAE,MAAM;EACf;EACA,iBAAiB;CACnB;AACF;AAEA,SAAS,kBAAkB,GAAmB,GAA4B;CACxE,IAAI,EAAE,KAAK,SAAS,EAAE,KAAK,MAAM,OAAO;CACxC,IAAI,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,SAAS,QAAQ,OAAO,EAAE,KAAK,aAAa,EAAE,KAAK;CACxF,IAAI,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,OAAO,OAAO,EAAE,KAAK,UAAU,EAAE,KAAK;CACnF,IAAI,EAAE,KAAK,SAAS,WAAW,EAAE,KAAK,SAAS,SAC7C,OAAO,EAAE,KAAK,cAAc,EAAE,KAAK;CACrC,OAAO,EAAE,KAAK,SAAS,EAAE,KAAK;AAChC;AAEA,SAAS,mBAAmB,GAAmB,GAA2B;CACxE,IAAI,EAAE,KAAK,SAAS,EAAE,KAAK,MAAM,OAAO,QAAQ,EAAE,KAAK,KAAK,MAAM,EAAE,KAAK;CACzE,IAAI,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,aAAa,EAAE,KAAK,UACjF,OAAO,QAAQ,EAAE,KAAK,SAAS,MAAM,EAAE,KAAK;CAE9C,IAAI,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,UAAU,EAAE,KAAK,OAC5E,OAAO,SAAS,EAAE,KAAK,MAAM,MAAM,EAAE,KAAK;CAE5C,OAAO,SAAS,EAAE,KAAK,KAAK,QAAQ,EAAE,KAAK,KAAK;AAClD;;;;;;;;;;;ACjEA,eAAsB,eACpB,OACA,SACA,SACyB;CACzB,MAAM,eAAe,MAAM,MAAM,SAAS,QAAQ,QAAQ;CAC1D,MAAM,gBAAgB,MAAM,MAAM,SAAS,QAAQ,SAAS;CAS5D,OAAO,kBAAkB,MARH,QAAQ,IAC5B,QAAQ,IAAI,OAAO,MAAM;EACvB,MAAM,UAAU,EAAE,WAAW,mBAAmB,EAAE,MAAM;EACxD,MAAM,WAAW,MAAM,WAAW,cAAc,SAAS,KAAK;EAC9D,MAAM,YAAY,MAAM,WAAW,eAAe,SAAS,KAAK;EAChE,OAAO;GAAE,QAAQ,EAAE;GAAQ,gBAAgB,EAAE;GAAgB;GAAU;EAAU;CACnF,CAAC,CACH,GACkC,OAAO;AAC3C;AAEA,eAAe,WACb,MACA,SACA,OACmB;CACnB,MAAM,MAAgB,CAAC;CACvB,KAAK,MAAM,KAAK,MAAM;EACpB,MAAM,IAAI,MAAM,QAAQ,GAAG,KAAK;EAChC,IAAI,MAAM,QAAQ,OAAO,SAAS,CAAC,GAAG,IAAI,KAAK,CAAC;CAClD;CACA,OAAO;AACT;;;;;;;;;;;;ACzCA,MAAM,wBAAwB;AAC9B,MAAM,qCAAqC;AA8C3C,eAAsB,cACpB,OACA,UAA4B,CAAC,GACH;CAC1B,MAAM,iBAAiB,QAAQ,kBAAkB;CACjD,MAAM,cAAc,QAAQ,eAAe;CAC3C,MAAM,0BACJ,QAAQ,2BAA2B;CACrC,IAAI,CAAC,OAAO,UAAU,cAAc,KAAK,iBAAiB,GACxD,MAAM,IAAI,WAAW,2CAA2C;CAElE,IAAI,CAAC,OAAO,SAAS,WAAW,KAAK,cAAc,GACjD,MAAM,IAAI,WAAW,kDAAkD;CAEzE,IAAI,CAAC,OAAO,UAAU,uBAAuB,KAAK,0BAA0B,GAC1E,MAAM,IAAI,WAAW,wDAAwD;CAE/E,MAAM,OAAO,QAAQ,QACjB,CAAC,EAAE,OAAO,QAAQ,MAAM,CAAC,KACxB,MAAM,MAAM,SAAS,EAAA,CAAG,KAAK,OAAO,EAAE,OAAO,EAAE,MAAM,EAAE;CAE5D,MAAM,WAA+B,CAAC;CACtC,KAAK,MAAM,EAAE,WAAW,MAAM;EAC5B,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,MAAM,CAAC;EACzC,MAAM,YAAY,IAAI,IAAI,MAAM,KAAK,SAAS,CAAC,KAAK,QAAQ,IAAI,CAAC,CAAC;EAClE,MAAM,cAAgC,MACnC,OAAO,UAAU,CAAC,CAClB,KAAK,MAAM,iBAAiB;GAAE;GAAM;EAAY,EAAE,CAAC,CACnD,MAAM,GAAG,MAAM,EAAE,KAAK,YAAY,EAAE,KAAK,aAAa,EAAE,cAAc,EAAE,WAAW,CAAC,CACpF,KAAK,EAAE,YAAY;GAAE;GAAM,GAAG,eAAe,MAAM,SAAS;EAAE,EAAE;EACnE,MAAM,cAAc,qBAClB,YAAY,KAAK,SAAS;GAExB,MAAM,QADS,KAAK,eAAe,QAAQ,KAAK,eAAe,KAAK,cAEhE,KAAK,OACL,KAAK,aACH,UAAU,IAAI,KAAK,UAAU,IAC7B,KAAA;GACN,OAAO;IACL,KAAK,aAAa,IAAI;IACtB,UAAU,KAAK,UAAU,KAAK,WAAW;IACzC,OAAO,OAAO,aAAa;IAC3B,KAAK,OAAO,WAAW;GACzB;EACF,CAAC,CACH;EACA,MAAM,uCAAuB,IAAI,IAAoB;EACrD,MAAM,eAAkC,YAAY,KAAK,SAAS;GAChE,MAAM,UAAU,YAAY,IAAI,aAAa,IAAI,CAAC;GAClD,MAAM,gBAAgB,qBAAqB,IAAI,OAAO,KAAK;GAC3D,qBAAqB,IAAI,SAAS,gBAAgB,CAAC;GACnD,OAAO;IAAE,GAAG;IAAM;IAAe;GAAQ;EAC3C,CAAC;EACD,MAAM,wBAAQ,IAAI,IAQhB;EACF,KAAK,MAAM,QAAQ,cAAc;GAC/B,IAAI,CAAC,oBAAoB,KAAK,IAAI,GAAG;GACrC,MAAM,IAAI,QAAQ,KAAK,KAAK,IAAI;GAChC,MAAM,MAAM,KAAK,UAAU;IAAC,KAAK;IAAS,KAAK,KAAK;IAAU;GAAC,CAAC;GAChE,MAAM,SAAS,MAAM,IAAI,GAAG,KAAK;IAC/B,OAAO,CAAC;IACR,SAAS;IACT,UAAU,KAAK,KAAK;IACpB,aAAa,KAAK;GACpB;GACA,OAAO,MAAM,KAAK,IAAI;GACtB,MAAM,IAAI,KAAK,MAAM;EACvB;EACA,KAAK,MAAM,EAAE,OAAO,SAAS,GAAG,UAAU,iBAAiB,MAAM,OAAO,GAAG;GACzE,IAAI,MAAM,SAAS,gBAAgB;GACnC,IAAI,eAAe;GACnB,KAAK,IAAI,aAAa,GAAG,cAAc,MAAM,QAAQ,cAAc,GAAG;IACpE,MAAM,WAAW,MAAM,aAAa;IACpC,MAAM,OAAO,MAAM;IAMnB,IAAI,EAJF,SAAS,KAAA,KACT,KAAK,KAAK,YAAY,SAAS,KAAK,YAAY,eAChD,KAAK,gBAAgB,SAAS,gBAAgB,IAAI,2BAClD,CAAC,eAAe,UAAU,IAAI,IACb;IAEnB,MAAM,UAAU,MAAM,MAAM,cAAc,UAAU;IACpD,IAAI,OAAO;IACX,IAAI,YAAY;IAChB,IAAI,UAAU;IACd,KAAK,IAAI,QAAQ,GAAG,QAAQ,QAAQ,QAAQ,SAAS,GAAG;KACtD,OAAO,QAAQ,MAAM,CAAE,KAAK,YAAY,QAAQ,KAAK,CAAE,KAAK,YAAY,aACtE,QAAQ;KAEV,IAAI,QAAQ,OAAO,UAAU,WAAW;MACtC,YAAY;MACZ,UAAU;KACZ;IACF;IACA,IAAI,UAAU,YAAY,KAAK,gBAAgB;KAC7C,MAAM,OAAO,QAAQ,MAAM,WAAW,UAAU,CAAC;KACjD,MAAM,QAAQ,KAAK,EAAE,CAAE,KAAK;KAC5B,MAAM,OAAO,KAAK,KAAK,SAAS,EAAE,CAAE,KAAK;KACzC,SAAS,KAAK;MACZ;MACA;MACA,SAAS;MACT,aAAa,KAAK;MAClB,SAAS,KAAK,KAAK,SAAS,KAAK,KAAK,MAAM;MAC5C,GAAI,cAAc,EAAE,YAAY,IAAI,CAAC;MACrC,UAAU,OAAO;KACnB,CAAC;IACH;IACA,eAAe;GACjB;EACF;CACF;CAEA,MAAM,eAAe,IAAI,IAAI,SAAS,KAAK,MAAM,EAAE,KAAK,CAAC;CACzD,OAAO;EACL;EACA,kBAAkB,KAAK,SAAS,IAAI,aAAa,OAAO,KAAK,SAAS;EACtE,WAAW,KAAK;CAClB;AACF;AAEA,SAAS,QAAQ,MAAkE;CACjF,OAAO,KAAK,UAAU,CAAC,KAAK,aAAa,KAAK,UAAU,CAAC;AAC3D;AAEA,SAAS,aAAa,MAA8B;CAClD,OAAO,KAAK,eAAe,QAAQ,KAAK,eAAe,KAAK,cACxD,KAAK,UAAU,CAAC,KAAK,aAAa,KAAK,KAAK,MAAM,CAAC,IACnD,QAAQ,IAAI;AAClB;AAEA,SAAS,eACP,MACA,WAC2D;CAC3D,MAAM,eAAe,KAAK;CAC1B,IAAI,CAAC,cAAc,OAAO;EAAE,aAAa;EAAM,YAAY;CAAK;CAEhE,IAAI,YAAgC;CACpC,IAAI,aAA4B;CAChC,MAAM,uBAAO,IAAI,IAAY;CAC7B,OAAO,aAAa,CAAC,KAAK,IAAI,SAAS,GAAG;EACxC,KAAK,IAAI,SAAS;EAClB,MAAM,UAAU,UAAU,IAAI,SAAS;EACvC,IAAI,CAAC,SAAS,OAAO;GAAE,aAAa;GAAc,YAAY;EAAa;EAC3E,IAAI,QAAQ,SAAS,SACnB,OAAO;GAAE,aAAa,QAAQ;GAAQ,YAAY,cAAc,QAAQ;EAAO;EAEjF,aAAa,QAAQ;EACrB,YAAY,QAAQ;CACtB;CACA,OAAO;EAAE,aAAa;EAAc,YAAY;CAAa;AAC/D;AAEA,SAAS,eAAe,UAA2B,MAAgC;CACjF,OAAO,SAAS,KAAK,YAAY,KAAA,KAAa,SAAS,KAAK,WAAW,KAAK,KAAK;AACnF"}
@@ -1,6 +1,6 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
- import { s as zQuantile } from "./internal-BDHPCnjk.js";
3
- import { u as normalCdf } from "./paired-arms-D-XRF_fy.js";
2
+ import { s as zQuantile } from "./internal-BMFSR8Ns.js";
3
+ import { u as normalCdf } from "./paired-arms-D4aeIHUy.js";
4
4
  //#region src/statistics/multiplicity.ts
5
5
  /**
6
6
  * Multiple-comparison corrections: family-wise error (Bonferroni, Holm) and
@@ -192,4 +192,4 @@ function mcnemarPower(opts) {
192
192
  //#endregion
193
193
  export { requiredSampleSize as a, holm as c, requiredPairedSampleSize as i, mcnemarRequiredN as n, benjaminiHochberg as o, pairedMde as r, bonferroni as s, mcnemarPower as t };
194
194
 
195
- //# sourceMappingURL=power-and-mde-CHIrXJll.js.map
195
+ //# sourceMappingURL=power-and-mde-B8F2RdcD.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"power-and-mde-CHIrXJll.js","names":[],"sources":["../src/statistics/multiplicity.ts","../src/statistics/power-and-mde.ts"],"sourcesContent":["/**\n * Multiple-comparison corrections: family-wise error (Bonferroni, Holm) and\n * false discovery rate (Benjamini-Hochberg), with inclusive rejection\n * boundaries throughout.\n */\n\nimport { ValidationError } from '../errors'\n\n/**\n * Bonferroni adjustment: multiply every p-value by the test count, clamp at 1.\n *\n * Rejects at `p_adjusted ≤ alpha` — the boundary is inclusive, matching\n * {@link holm}, which uniformly dominates this correction and must therefore\n * never reject less. Validates its inputs on the same terms.\n */\nexport function bonferroni(\n pValues: readonly number[],\n alpha = 0.05,\n): { adjusted: number[]; significant: boolean[] } {\n assertAlpha('bonferroni', 'alpha', alpha)\n assertPValues('bonferroni', pValues)\n const k = pValues.length\n const adjusted = pValues.map((p) => Math.min(1, p * k))\n return { adjusted, significant: adjusted.map((p) => p <= alpha) }\n}\n\n/**\n * Holm step-down family-wise error adjustment.\n *\n * P-values are sorted from smallest to largest, multiplied by their remaining\n * hypothesis count, and made monotonically non-decreasing before being mapped\n * back to input order. This uniformly dominates plain Bonferroni while keeping\n * strong family-wise error control under arbitrary dependence.\n */\nexport function holm(\n pValues: readonly number[],\n alpha = 0.05,\n): { adjusted: number[]; significant: boolean[] } {\n assertAlpha('holm', 'alpha', alpha)\n assertPValues('holm', pValues)\n const count = pValues.length\n if (count === 0) return { adjusted: [], significant: [] }\n\n const ordered = pValues\n .map((pValue, index) => ({ pValue, index }))\n .sort((a, b) => a.pValue - b.pValue || a.index - b.index)\n const adjusted = new Array<number>(count)\n let previous = 0\n for (let rank = 0; rank < count; rank++) {\n const entry = ordered[rank]!\n const stepAdjusted = Math.min(1, entry.pValue * (count - rank))\n previous = Math.max(previous, stepAdjusted)\n adjusted[entry.index] = previous\n }\n // Holm's rejection rule is inclusive at the adjusted alpha boundary.\n return { adjusted, significant: adjusted.map((pValue) => pValue <= alpha) }\n}\n\n/**\n * Benjamini–Hochberg false discovery rate. Returns adjusted q-values and\n * significance at the target FDR; handles ties and preserves q monotonicity.\n *\n * Rejects at `q ≤ fdr` — the BH rule is inclusive at the boundary, so an\n * exactly-`fdr` q-value is a discovery.\n */\nexport function benjaminiHochberg(\n pValues: readonly number[],\n fdr = 0.05,\n): { qValues: number[]; significant: boolean[] } {\n assertAlpha('benjaminiHochberg', 'fdr', fdr)\n assertPValues('benjaminiHochberg', pValues)\n const n = pValues.length\n if (n === 0) return { qValues: [], significant: [] }\n const indexed = pValues.map((p, i) => ({ p, i })).sort((a, b) => a.p - b.p)\n const q = new Array<number>(n)\n let minRight = 1\n for (let k = n - 1; k >= 0; k--) {\n const rank = k + 1\n const entry = indexed[k]!\n // (n/rank)·p, not (p·n)/rank: the latter lands one ULP above `fdr` at an\n // exact boundary (p = 0.05, n = 3, rank = 3 gives 0.05000000000000001),\n // which silently turns a discovery into a non-discovery. This is R's\n // `p.adjust(method = \"BH\")` formulation.\n const raw = (n / rank) * entry.p\n const bounded = Math.min(minRight, raw)\n minRight = bounded\n q[entry.i] = Math.min(1, bounded)\n }\n return { qValues: q, significant: q.map((v) => v <= fdr) }\n}\n\nfunction assertAlpha(fn: string, label: string, value: number): void {\n if (!Number.isFinite(value) || value <= 0 || value >= 1) {\n throw new ValidationError(`${fn}: ${label} must be in (0,1), got ${value}`)\n }\n}\n\nfunction assertPValues(fn: string, pValues: readonly number[]): void {\n for (const [index, pValue] of pValues.entries()) {\n if (!Number.isFinite(pValue) || pValue < 0 || pValue > 1) {\n throw new ValidationError(`${fn}: pValues[${index}] must be in [0,1], got ${pValue}`)\n }\n }\n}\n","import { normalCdf } from '../math/normal'\nimport { zQuantile } from './internal'\n\n// ── Power analysis + minimum detectable effect ───────────────────────\n\n/**\n * Required N per arm for a two-sample comparison at target effect size,\n * alpha, and power. Normal-approximation formula:\n * n = 2 * ( (z_{1-α/2} + z_{1-β}) / d )^2\n * where d is Cohen's d. Returns Infinity for effect ≤ 0.\n */\nexport function requiredSampleSize(opts: {\n effect: number\n alpha?: number\n power?: number\n twoSided?: boolean\n}): number {\n const effect = opts.effect\n if (!Number.isFinite(effect) || effect <= 0) return Infinity\n const alpha = opts.alpha ?? 0.05\n const power = opts.power ?? 0.8\n const twoSided = opts.twoSided ?? true\n const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha)\n const zBeta = zQuantile(power)\n const n = 2 * ((zAlpha + zBeta) / effect) ** 2\n return Math.ceil(n)\n}\n\n/**\n * Required number of paired observations for a target Cohen's dz.\n * Unlike the independent-groups formula, this has no two-arm factor of two.\n *\n * Normal quantiles with no t correction, so treat the result as a LOWER bound:\n * it returns 32 where the exact t-based answer is 34 at dz = 0.5, and 13 where\n * it is 15 at dz = 0.8 — a 6–13 % shortfall precisely in the range a caller\n * consults to decide whether 3–10 repetitions suffice.\n */\nexport function requiredPairedSampleSize(opts: {\n effect: number\n alpha?: number\n power?: number\n twoSided?: boolean\n}): number {\n const effect = opts.effect\n if (!Number.isFinite(effect) || effect <= 0) return Infinity\n const alpha = opts.alpha ?? 0.05\n const power = opts.power ?? 0.8\n const twoSided = opts.twoSided ?? true\n const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha)\n const zBeta = zQuantile(power)\n return Math.ceil(((zAlpha + zBeta) / effect) ** 2)\n}\n\n/**\n * Minimum detectable paired effect (standardised units) for a target paired\n * sample size: d_min = (z_{1-α/2} + z_β) / sqrt(n_paired). Multiply by\n * sd(deltas) for score units; treat as a lower bound — Wilcoxon and bootstrap\n * have asymptotic relative efficiency below 1 vs the t-test on heavy tails.\n */\nexport function pairedMde(opts: {\n nPaired: number\n alpha?: number\n power?: number\n twoSided?: boolean\n}): number {\n if (!Number.isFinite(opts.nPaired) || opts.nPaired <= 0) return Infinity\n const alpha = opts.alpha ?? 0.05\n const power = opts.power ?? 0.8\n const twoSided = opts.twoSided ?? true\n const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha)\n const zBeta = zQuantile(power)\n return (zAlpha + zBeta) / Math.sqrt(opts.nPaired)\n}\n\n/**\n * Number of paired observations needed for a McNemar test to reach a target\n * power — the pre-registration companion to {@link mcnemar}. Parametrised by the\n * expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and\n * `p01` (P[control wins]); concordant pairs carry no information, so the count\n * is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal\n * approximation: with discordant rate `pDisc = p10 + p01` and marginal effect\n * `δ = p10 − p01`,\n * n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².\n * Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the\n * tiny discordant counts where the exact {@link mcnemar} differs from the normal\n * approximation, treat the result as a lower bound and prefer the discordant-pair\n * floor.\n */\nexport function mcnemarRequiredN(opts: {\n p10: number\n p01: number\n alpha?: number\n power?: number\n twoSided?: boolean\n}): number {\n const { p10, p01 } = opts\n if (p10 < 0 || p01 < 0 || p10 + p01 > 1) {\n throw new Error(`mcnemarRequiredN: require p10,p01 ≥ 0 and p10+p01 ≤ 1 (got ${p10}, ${p01})`)\n }\n const delta = p10 - p01\n if (delta === 0) return Infinity\n const alpha = opts.alpha ?? 0.05\n const power = opts.power ?? 0.8\n const twoSided = opts.twoSided ?? true\n const pDisc = p10 + p01\n const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha)\n const zBeta = zQuantile(power)\n const n =\n (zAlpha * Math.sqrt(pDisc) + zBeta * Math.sqrt(Math.max(0, pDisc - delta * delta))) ** 2 /\n (delta * delta)\n return Math.ceil(n)\n}\n\n/**\n * Power of a McNemar test at a given number of paired observations, the inverse\n * of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).\n * Returns a value in [0, 1]; equals `alpha` when there is no effect.\n */\nexport function mcnemarPower(opts: {\n p10: number\n p01: number\n nPairs: number\n alpha?: number\n twoSided?: boolean\n}): number {\n const { p10, p01, nPairs } = opts\n if (p10 < 0 || p01 < 0 || p10 + p01 > 1) {\n throw new Error(`mcnemarPower: require p10,p01 ≥ 0 and p10+p01 ≤ 1 (got ${p10}, ${p01})`)\n }\n const alpha = opts.alpha ?? 0.05\n const twoSided = opts.twoSided ?? true\n const delta = p10 - p01\n if (delta === 0 || nPairs <= 0) return alpha\n const pDisc = p10 + p01\n const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha)\n const denom = Math.sqrt(Math.max(1e-12, pDisc - delta * delta))\n const zBeta = (Math.sqrt(nPairs) * Math.abs(delta) - zAlpha * Math.sqrt(pDisc)) / denom\n return Math.min(1, Math.max(0, normalCdf(zBeta)))\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,SAAgB,WACd,SACA,QAAQ,KACwC;CAChD,YAAY,cAAc,SAAS,KAAK;CACxC,cAAc,cAAc,OAAO;CACnC,MAAM,IAAI,QAAQ;CAClB,MAAM,WAAW,QAAQ,KAAK,MAAM,KAAK,IAAI,GAAG,IAAI,CAAC,CAAC;CACtD,OAAO;EAAE;EAAU,aAAa,SAAS,KAAK,MAAM,KAAK,KAAK;CAAE;AAClE;;;;;;;;;AAUA,SAAgB,KACd,SACA,QAAQ,KACwC;CAChD,YAAY,QAAQ,SAAS,KAAK;CAClC,cAAc,QAAQ,OAAO;CAC7B,MAAM,QAAQ,QAAQ;CACtB,IAAI,UAAU,GAAG,OAAO;EAAE,UAAU,CAAC;EAAG,aAAa,CAAC;CAAE;CAExD,MAAM,UAAU,QACb,KAAK,QAAQ,WAAW;EAAE;EAAQ;CAAM,EAAE,CAAC,CAC3C,MAAM,GAAG,MAAM,EAAE,SAAS,EAAE,UAAU,EAAE,QAAQ,EAAE,KAAK;CAC1D,MAAM,WAAW,IAAI,MAAc,KAAK;CACxC,IAAI,WAAW;CACf,KAAK,IAAI,OAAO,GAAG,OAAO,OAAO,QAAQ;EACvC,MAAM,QAAQ,QAAQ;EACtB,MAAM,eAAe,KAAK,IAAI,GAAG,MAAM,UAAU,QAAQ,KAAK;EAC9D,WAAW,KAAK,IAAI,UAAU,YAAY;EAC1C,SAAS,MAAM,SAAS;CAC1B;CAEA,OAAO;EAAE;EAAU,aAAa,SAAS,KAAK,WAAW,UAAU,KAAK;CAAE;AAC5E;;;;;;;;AASA,SAAgB,kBACd,SACA,MAAM,KACyC;CAC/C,YAAY,qBAAqB,OAAO,GAAG;CAC3C,cAAc,qBAAqB,OAAO;CAC1C,MAAM,IAAI,QAAQ;CAClB,IAAI,MAAM,GAAG,OAAO;EAAE,SAAS,CAAC;EAAG,aAAa,CAAC;CAAE;CACnD,MAAM,UAAU,QAAQ,KAAK,GAAG,OAAO;EAAE;EAAG;CAAE,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,IAAI,EAAE,CAAC;CAC1E,MAAM,IAAI,IAAI,MAAc,CAAC;CAC7B,IAAI,WAAW;CACf,KAAK,IAAI,IAAI,IAAI,GAAG,KAAK,GAAG,KAAK;EAC/B,MAAM,OAAO,IAAI;EACjB,MAAM,QAAQ,QAAQ;EAKtB,MAAM,MAAO,IAAI,OAAQ,MAAM;EAC/B,MAAM,UAAU,KAAK,IAAI,UAAU,GAAG;EACtC,WAAW;EACX,EAAE,MAAM,KAAK,KAAK,IAAI,GAAG,OAAO;CAClC;CACA,OAAO;EAAE,SAAS;EAAG,aAAa,EAAE,KAAK,MAAM,KAAK,GAAG;CAAE;AAC3D;AAEA,SAAS,YAAY,IAAY,OAAe,OAAqB;CACnE,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,SAAS,KAAK,SAAS,GACpD,MAAM,IAAI,gBAAgB,GAAG,GAAG,IAAI,MAAM,yBAAyB,OAAO;AAE9E;AAEA,SAAS,cAAc,IAAY,SAAkC;CACnE,KAAK,MAAM,CAAC,OAAO,WAAW,QAAQ,QAAQ,GAC5C,IAAI,CAAC,OAAO,SAAS,MAAM,KAAK,SAAS,KAAK,SAAS,GACrD,MAAM,IAAI,gBAAgB,GAAG,GAAG,YAAY,MAAM,0BAA0B,QAAQ;AAG1F;;;;;;;;;AC5FA,SAAgB,mBAAmB,MAKxB;CACT,MAAM,SAAS,KAAK;CACpB,IAAI,CAAC,OAAO,SAAS,MAAM,KAAK,UAAU,GAAG,OAAO;CACpD,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,QAAQ,KAAK,SAAS;CAI5B,MAAM,IAAI,MAFK,UADE,KAAK,YAAY,OACE,IAAI,QAAQ,IAAI,IAAI,KAEnC,IADP,UAAU,KACK,KAAK,WAAW;CAC7C,OAAO,KAAK,KAAK,CAAC;AACpB;;;;;;;;;;AAWA,SAAgB,yBAAyB,MAK9B;CACT,MAAM,SAAS,KAAK;CACpB,IAAI,CAAC,OAAO,SAAS,MAAM,KAAK,UAAU,GAAG,OAAO;CACpD,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,QAAQ,KAAK,SAAS;CAE5B,MAAM,SAAS,UADE,KAAK,YAAY,OACE,IAAI,QAAQ,IAAI,IAAI,KAAK;CAC7D,MAAM,QAAQ,UAAU,KAAK;CAC7B,OAAO,KAAK,OAAO,SAAS,SAAS,WAAW,CAAC;AACnD;;;;;;;AAQA,SAAgB,UAAU,MAKf;CACT,IAAI,CAAC,OAAO,SAAS,KAAK,OAAO,KAAK,KAAK,WAAW,GAAG,OAAO;CAChE,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,QAAQ,KAAK,SAAS;CAI5B,QAFe,UADE,KAAK,YAAY,OACE,IAAI,QAAQ,IAAI,IAAI,KAE3C,IADC,UAAU,KACH,KAAK,KAAK,KAAK,KAAK,OAAO;AAClD;;;;;;;;;;;;;;;AAgBA,SAAgB,iBAAiB,MAMtB;CACT,MAAM,EAAE,KAAK,QAAQ;CACrB,IAAI,MAAM,KAAK,MAAM,KAAK,MAAM,MAAM,GACpC,MAAM,IAAI,MAAM,8DAA8D,IAAI,IAAI,IAAI,EAAE;CAE9F,MAAM,QAAQ,MAAM;CACpB,IAAI,UAAU,GAAG,OAAO;CACxB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,WAAW,KAAK,YAAY;CAClC,MAAM,QAAQ,MAAM;CACpB,MAAM,SAAS,UAAU,WAAW,IAAI,QAAQ,IAAI,IAAI,KAAK;CAC7D,MAAM,QAAQ,UAAU,KAAK;CAC7B,MAAM,KACH,SAAS,KAAK,KAAK,KAAK,IAAI,QAAQ,KAAK,KAAK,KAAK,IAAI,GAAG,QAAQ,QAAQ,KAAK,CAAC,MAAM,KACtF,QAAQ;CACX,OAAO,KAAK,KAAK,CAAC;AACpB;;;;;;AAOA,SAAgB,aAAa,MAMlB;CACT,MAAM,EAAE,KAAK,KAAK,WAAW;CAC7B,IAAI,MAAM,KAAK,MAAM,KAAK,MAAM,MAAM,GACpC,MAAM,IAAI,MAAM,0DAA0D,IAAI,IAAI,IAAI,EAAE;CAE1F,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,WAAW,KAAK,YAAY;CAClC,MAAM,QAAQ,MAAM;CACpB,IAAI,UAAU,KAAK,UAAU,GAAG,OAAO;CACvC,MAAM,QAAQ,MAAM;CACpB,MAAM,SAAS,UAAU,WAAW,IAAI,QAAQ,IAAI,IAAI,KAAK;CAC7D,MAAM,QAAQ,KAAK,KAAK,KAAK,IAAI,OAAO,QAAQ,QAAQ,KAAK,CAAC;CAC9D,MAAM,SAAS,KAAK,KAAK,MAAM,IAAI,KAAK,IAAI,KAAK,IAAI,SAAS,KAAK,KAAK,KAAK,KAAK;CAClF,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,UAAU,KAAK,CAAC,CAAC;AAClD"}
1
+ {"version":3,"file":"power-and-mde-B8F2RdcD.js","names":[],"sources":["../src/statistics/multiplicity.ts","../src/statistics/power-and-mde.ts"],"sourcesContent":["/**\n * Multiple-comparison corrections: family-wise error (Bonferroni, Holm) and\n * false discovery rate (Benjamini-Hochberg), with inclusive rejection\n * boundaries throughout.\n */\n\nimport { ValidationError } from '../errors'\n\n/**\n * Bonferroni adjustment: multiply every p-value by the test count, clamp at 1.\n *\n * Rejects at `p_adjusted ≤ alpha` — the boundary is inclusive, matching\n * {@link holm}, which uniformly dominates this correction and must therefore\n * never reject less. Validates its inputs on the same terms.\n */\nexport function bonferroni(\n pValues: readonly number[],\n alpha = 0.05,\n): { adjusted: number[]; significant: boolean[] } {\n assertAlpha('bonferroni', 'alpha', alpha)\n assertPValues('bonferroni', pValues)\n const k = pValues.length\n const adjusted = pValues.map((p) => Math.min(1, p * k))\n return { adjusted, significant: adjusted.map((p) => p <= alpha) }\n}\n\n/**\n * Holm step-down family-wise error adjustment.\n *\n * P-values are sorted from smallest to largest, multiplied by their remaining\n * hypothesis count, and made monotonically non-decreasing before being mapped\n * back to input order. This uniformly dominates plain Bonferroni while keeping\n * strong family-wise error control under arbitrary dependence.\n */\nexport function holm(\n pValues: readonly number[],\n alpha = 0.05,\n): { adjusted: number[]; significant: boolean[] } {\n assertAlpha('holm', 'alpha', alpha)\n assertPValues('holm', pValues)\n const count = pValues.length\n if (count === 0) return { adjusted: [], significant: [] }\n\n const ordered = pValues\n .map((pValue, index) => ({ pValue, index }))\n .sort((a, b) => a.pValue - b.pValue || a.index - b.index)\n const adjusted = new Array<number>(count)\n let previous = 0\n for (let rank = 0; rank < count; rank++) {\n const entry = ordered[rank]!\n const stepAdjusted = Math.min(1, entry.pValue * (count - rank))\n previous = Math.max(previous, stepAdjusted)\n adjusted[entry.index] = previous\n }\n // Holm's rejection rule is inclusive at the adjusted alpha boundary.\n return { adjusted, significant: adjusted.map((pValue) => pValue <= alpha) }\n}\n\n/**\n * Benjamini–Hochberg false discovery rate. Returns adjusted q-values and\n * significance at the target FDR; handles ties and preserves q monotonicity.\n *\n * Rejects at `q ≤ fdr` — the BH rule is inclusive at the boundary, so an\n * exactly-`fdr` q-value is a discovery.\n */\nexport function benjaminiHochberg(\n pValues: readonly number[],\n fdr = 0.05,\n): { qValues: number[]; significant: boolean[] } {\n assertAlpha('benjaminiHochberg', 'fdr', fdr)\n assertPValues('benjaminiHochberg', pValues)\n const n = pValues.length\n if (n === 0) return { qValues: [], significant: [] }\n const indexed = pValues.map((p, i) => ({ p, i })).sort((a, b) => a.p - b.p)\n const q = new Array<number>(n)\n let minRight = 1\n for (let k = n - 1; k >= 0; k--) {\n const rank = k + 1\n const entry = indexed[k]!\n // (n/rank)·p, not (p·n)/rank: the latter lands one ULP above `fdr` at an\n // exact boundary (p = 0.05, n = 3, rank = 3 gives 0.05000000000000001),\n // which silently turns a discovery into a non-discovery. This is R's\n // `p.adjust(method = \"BH\")` formulation.\n const raw = (n / rank) * entry.p\n const bounded = Math.min(minRight, raw)\n minRight = bounded\n q[entry.i] = Math.min(1, bounded)\n }\n return { qValues: q, significant: q.map((v) => v <= fdr) }\n}\n\nfunction assertAlpha(fn: string, label: string, value: number): void {\n if (!Number.isFinite(value) || value <= 0 || value >= 1) {\n throw new ValidationError(`${fn}: ${label} must be in (0,1), got ${value}`)\n }\n}\n\nfunction assertPValues(fn: string, pValues: readonly number[]): void {\n for (const [index, pValue] of pValues.entries()) {\n if (!Number.isFinite(pValue) || pValue < 0 || pValue > 1) {\n throw new ValidationError(`${fn}: pValues[${index}] must be in [0,1], got ${pValue}`)\n }\n }\n}\n","import { normalCdf } from '../math/normal'\nimport { zQuantile } from './internal'\n\n// ── Power analysis + minimum detectable effect ───────────────────────\n\n/**\n * Required N per arm for a two-sample comparison at target effect size,\n * alpha, and power. Normal-approximation formula:\n * n = 2 * ( (z_{1-α/2} + z_{1-β}) / d )^2\n * where d is Cohen's d. Returns Infinity for effect ≤ 0.\n */\nexport function requiredSampleSize(opts: {\n effect: number\n alpha?: number\n power?: number\n twoSided?: boolean\n}): number {\n const effect = opts.effect\n if (!Number.isFinite(effect) || effect <= 0) return Infinity\n const alpha = opts.alpha ?? 0.05\n const power = opts.power ?? 0.8\n const twoSided = opts.twoSided ?? true\n const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha)\n const zBeta = zQuantile(power)\n const n = 2 * ((zAlpha + zBeta) / effect) ** 2\n return Math.ceil(n)\n}\n\n/**\n * Required number of paired observations for a target Cohen's dz.\n * Unlike the independent-groups formula, this has no two-arm factor of two.\n *\n * Normal quantiles with no t correction, so treat the result as a LOWER bound:\n * it returns 32 where the exact t-based answer is 34 at dz = 0.5, and 13 where\n * it is 15 at dz = 0.8 — a 6–13 % shortfall precisely in the range a caller\n * consults to decide whether 3–10 repetitions suffice.\n */\nexport function requiredPairedSampleSize(opts: {\n effect: number\n alpha?: number\n power?: number\n twoSided?: boolean\n}): number {\n const effect = opts.effect\n if (!Number.isFinite(effect) || effect <= 0) return Infinity\n const alpha = opts.alpha ?? 0.05\n const power = opts.power ?? 0.8\n const twoSided = opts.twoSided ?? true\n const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha)\n const zBeta = zQuantile(power)\n return Math.ceil(((zAlpha + zBeta) / effect) ** 2)\n}\n\n/**\n * Minimum detectable paired effect (standardised units) for a target paired\n * sample size: d_min = (z_{1-α/2} + z_β) / sqrt(n_paired). Multiply by\n * sd(deltas) for score units; treat as a lower bound — Wilcoxon and bootstrap\n * have asymptotic relative efficiency below 1 vs the t-test on heavy tails.\n */\nexport function pairedMde(opts: {\n nPaired: number\n alpha?: number\n power?: number\n twoSided?: boolean\n}): number {\n if (!Number.isFinite(opts.nPaired) || opts.nPaired <= 0) return Infinity\n const alpha = opts.alpha ?? 0.05\n const power = opts.power ?? 0.8\n const twoSided = opts.twoSided ?? true\n const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha)\n const zBeta = zQuantile(power)\n return (zAlpha + zBeta) / Math.sqrt(opts.nPaired)\n}\n\n/**\n * Number of paired observations needed for a McNemar test to reach a target\n * power — the pre-registration companion to {@link mcnemar}. Parametrised by the\n * expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and\n * `p01` (P[control wins]); concordant pairs carry no information, so the count\n * is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal\n * approximation: with discordant rate `pDisc = p10 + p01` and marginal effect\n * `δ = p10 − p01`,\n * n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².\n * Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the\n * tiny discordant counts where the exact {@link mcnemar} differs from the normal\n * approximation, treat the result as a lower bound and prefer the discordant-pair\n * floor.\n */\nexport function mcnemarRequiredN(opts: {\n p10: number\n p01: number\n alpha?: number\n power?: number\n twoSided?: boolean\n}): number {\n const { p10, p01 } = opts\n if (p10 < 0 || p01 < 0 || p10 + p01 > 1) {\n throw new Error(`mcnemarRequiredN: require p10,p01 ≥ 0 and p10+p01 ≤ 1 (got ${p10}, ${p01})`)\n }\n const delta = p10 - p01\n if (delta === 0) return Infinity\n const alpha = opts.alpha ?? 0.05\n const power = opts.power ?? 0.8\n const twoSided = opts.twoSided ?? true\n const pDisc = p10 + p01\n const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha)\n const zBeta = zQuantile(power)\n const n =\n (zAlpha * Math.sqrt(pDisc) + zBeta * Math.sqrt(Math.max(0, pDisc - delta * delta))) ** 2 /\n (delta * delta)\n return Math.ceil(n)\n}\n\n/**\n * Power of a McNemar test at a given number of paired observations, the inverse\n * of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).\n * Returns a value in [0, 1]; equals `alpha` when there is no effect.\n */\nexport function mcnemarPower(opts: {\n p10: number\n p01: number\n nPairs: number\n alpha?: number\n twoSided?: boolean\n}): number {\n const { p10, p01, nPairs } = opts\n if (p10 < 0 || p01 < 0 || p10 + p01 > 1) {\n throw new Error(`mcnemarPower: require p10,p01 ≥ 0 and p10+p01 ≤ 1 (got ${p10}, ${p01})`)\n }\n const alpha = opts.alpha ?? 0.05\n const twoSided = opts.twoSided ?? true\n const delta = p10 - p01\n if (delta === 0 || nPairs <= 0) return alpha\n const pDisc = p10 + p01\n const zAlpha = zQuantile(twoSided ? 1 - alpha / 2 : 1 - alpha)\n const denom = Math.sqrt(Math.max(1e-12, pDisc - delta * delta))\n const zBeta = (Math.sqrt(nPairs) * Math.abs(delta) - zAlpha * Math.sqrt(pDisc)) / denom\n return Math.min(1, Math.max(0, normalCdf(zBeta)))\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,SAAgB,WACd,SACA,QAAQ,KACwC;CAChD,YAAY,cAAc,SAAS,KAAK;CACxC,cAAc,cAAc,OAAO;CACnC,MAAM,IAAI,QAAQ;CAClB,MAAM,WAAW,QAAQ,KAAK,MAAM,KAAK,IAAI,GAAG,IAAI,CAAC,CAAC;CACtD,OAAO;EAAE;EAAU,aAAa,SAAS,KAAK,MAAM,KAAK,KAAK;CAAE;AAClE;;;;;;;;;AAUA,SAAgB,KACd,SACA,QAAQ,KACwC;CAChD,YAAY,QAAQ,SAAS,KAAK;CAClC,cAAc,QAAQ,OAAO;CAC7B,MAAM,QAAQ,QAAQ;CACtB,IAAI,UAAU,GAAG,OAAO;EAAE,UAAU,CAAC;EAAG,aAAa,CAAC;CAAE;CAExD,MAAM,UAAU,QACb,KAAK,QAAQ,WAAW;EAAE;EAAQ;CAAM,EAAE,CAAC,CAC3C,MAAM,GAAG,MAAM,EAAE,SAAS,EAAE,UAAU,EAAE,QAAQ,EAAE,KAAK;CAC1D,MAAM,WAAW,IAAI,MAAc,KAAK;CACxC,IAAI,WAAW;CACf,KAAK,IAAI,OAAO,GAAG,OAAO,OAAO,QAAQ;EACvC,MAAM,QAAQ,QAAQ;EACtB,MAAM,eAAe,KAAK,IAAI,GAAG,MAAM,UAAU,QAAQ,KAAK;EAC9D,WAAW,KAAK,IAAI,UAAU,YAAY;EAC1C,SAAS,MAAM,SAAS;CAC1B;CAEA,OAAO;EAAE;EAAU,aAAa,SAAS,KAAK,WAAW,UAAU,KAAK;CAAE;AAC5E;;;;;;;;AASA,SAAgB,kBACd,SACA,MAAM,KACyC;CAC/C,YAAY,qBAAqB,OAAO,GAAG;CAC3C,cAAc,qBAAqB,OAAO;CAC1C,MAAM,IAAI,QAAQ;CAClB,IAAI,MAAM,GAAG,OAAO;EAAE,SAAS,CAAC;EAAG,aAAa,CAAC;CAAE;CACnD,MAAM,UAAU,QAAQ,KAAK,GAAG,OAAO;EAAE;EAAG;CAAE,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,IAAI,EAAE,CAAC;CAC1E,MAAM,IAAI,IAAI,MAAc,CAAC;CAC7B,IAAI,WAAW;CACf,KAAK,IAAI,IAAI,IAAI,GAAG,KAAK,GAAG,KAAK;EAC/B,MAAM,OAAO,IAAI;EACjB,MAAM,QAAQ,QAAQ;EAKtB,MAAM,MAAO,IAAI,OAAQ,MAAM;EAC/B,MAAM,UAAU,KAAK,IAAI,UAAU,GAAG;EACtC,WAAW;EACX,EAAE,MAAM,KAAK,KAAK,IAAI,GAAG,OAAO;CAClC;CACA,OAAO;EAAE,SAAS;EAAG,aAAa,EAAE,KAAK,MAAM,KAAK,GAAG;CAAE;AAC3D;AAEA,SAAS,YAAY,IAAY,OAAe,OAAqB;CACnE,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,SAAS,KAAK,SAAS,GACpD,MAAM,IAAI,gBAAgB,GAAG,GAAG,IAAI,MAAM,yBAAyB,OAAO;AAE9E;AAEA,SAAS,cAAc,IAAY,SAAkC;CACnE,KAAK,MAAM,CAAC,OAAO,WAAW,QAAQ,QAAQ,GAC5C,IAAI,CAAC,OAAO,SAAS,MAAM,KAAK,SAAS,KAAK,SAAS,GACrD,MAAM,IAAI,gBAAgB,GAAG,GAAG,YAAY,MAAM,0BAA0B,QAAQ;AAG1F;;;;;;;;;AC5FA,SAAgB,mBAAmB,MAKxB;CACT,MAAM,SAAS,KAAK;CACpB,IAAI,CAAC,OAAO,SAAS,MAAM,KAAK,UAAU,GAAG,OAAO;CACpD,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,QAAQ,KAAK,SAAS;CAI5B,MAAM,IAAI,MAFK,UADE,KAAK,YAAY,OACE,IAAI,QAAQ,IAAI,IAAI,KAEnC,IADP,UAAU,KACK,KAAK,WAAW;CAC7C,OAAO,KAAK,KAAK,CAAC;AACpB;;;;;;;;;;AAWA,SAAgB,yBAAyB,MAK9B;CACT,MAAM,SAAS,KAAK;CACpB,IAAI,CAAC,OAAO,SAAS,MAAM,KAAK,UAAU,GAAG,OAAO;CACpD,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,QAAQ,KAAK,SAAS;CAE5B,MAAM,SAAS,UADE,KAAK,YAAY,OACE,IAAI,QAAQ,IAAI,IAAI,KAAK;CAC7D,MAAM,QAAQ,UAAU,KAAK;CAC7B,OAAO,KAAK,OAAO,SAAS,SAAS,WAAW,CAAC;AACnD;;;;;;;AAQA,SAAgB,UAAU,MAKf;CACT,IAAI,CAAC,OAAO,SAAS,KAAK,OAAO,KAAK,KAAK,WAAW,GAAG,OAAO;CAChE,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,QAAQ,KAAK,SAAS;CAI5B,QAFe,UADE,KAAK,YAAY,OACE,IAAI,QAAQ,IAAI,IAAI,KAE3C,IADC,UAAU,KACH,KAAK,KAAK,KAAK,KAAK,OAAO;AAClD;;;;;;;;;;;;;;;AAgBA,SAAgB,iBAAiB,MAMtB;CACT,MAAM,EAAE,KAAK,QAAQ;CACrB,IAAI,MAAM,KAAK,MAAM,KAAK,MAAM,MAAM,GACpC,MAAM,IAAI,MAAM,8DAA8D,IAAI,IAAI,IAAI,EAAE;CAE9F,MAAM,QAAQ,MAAM;CACpB,IAAI,UAAU,GAAG,OAAO;CACxB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,WAAW,KAAK,YAAY;CAClC,MAAM,QAAQ,MAAM;CACpB,MAAM,SAAS,UAAU,WAAW,IAAI,QAAQ,IAAI,IAAI,KAAK;CAC7D,MAAM,QAAQ,UAAU,KAAK;CAC7B,MAAM,KACH,SAAS,KAAK,KAAK,KAAK,IAAI,QAAQ,KAAK,KAAK,KAAK,IAAI,GAAG,QAAQ,QAAQ,KAAK,CAAC,MAAM,KACtF,QAAQ;CACX,OAAO,KAAK,KAAK,CAAC;AACpB;;;;;;AAOA,SAAgB,aAAa,MAMlB;CACT,MAAM,EAAE,KAAK,KAAK,WAAW;CAC7B,IAAI,MAAM,KAAK,MAAM,KAAK,MAAM,MAAM,GACpC,MAAM,IAAI,MAAM,0DAA0D,IAAI,IAAI,IAAI,EAAE;CAE1F,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,WAAW,KAAK,YAAY;CAClC,MAAM,QAAQ,MAAM;CACpB,IAAI,UAAU,KAAK,UAAU,GAAG,OAAO;CACvC,MAAM,QAAQ,MAAM;CACpB,MAAM,SAAS,UAAU,WAAW,IAAI,QAAQ,IAAI,IAAI,KAAK;CAC7D,MAAM,QAAQ,KAAK,KAAK,KAAK,IAAI,OAAO,QAAQ,QAAQ,KAAK,CAAC;CAC9D,MAAM,SAAS,KAAK,KAAK,MAAM,IAAI,KAAK,IAAI,KAAK,IAAI,SAAS,KAAK,KAAK,KAAK,KAAK;CAClF,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,UAAU,KAAK,CAAC,CAAC;AAClD"}
@@ -1,5 +1,5 @@
1
- import { g as pairedRiskDifferenceScore, h as pairedRiskDifferenceExact, p as pairedBinaryScale } from "./paired-arms-D-XRF_fy.js";
2
- import { a as pairedSignTest, i as pairedDeltaTieFraction, r as pairedBootstrap } from "./paired-tests-BHIhYVdu.js";
1
+ import { g as pairedRiskDifferenceScore, h as pairedRiskDifferenceExact, p as pairedBinaryScale } from "./paired-arms-D4aeIHUy.js";
2
+ import { a as pairedSignTest, i as pairedDeltaTieFraction, r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
3
3
  //#region src/paired-delta-test.ts
4
4
  /** Smallest all-positive sample that can clear a one-sided exact sign test. */
5
5
  function minimumPairsForPairedDeltaTest(confidence = .95) {
@@ -499,4 +499,4 @@ function powerPreflight(opts) {
499
499
  //#endregion
500
500
  export { heldoutSignificance as a, pairedDecisionShape as c, dimensionRegressions as i, minimumPairsForPairedDeltaTest as l, TIE_WARN_FRACTION as n, pairHoldout as o, detectScale as r, decidePairedPromotion as s, powerPreflight as t, pairedDeltaTest as u };
501
501
 
502
- //# sourceMappingURL=power-preflight-DEw-uC7q.js.map
502
+ //# sourceMappingURL=power-preflight-CFXm0Vjo.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"power-preflight-DEw-uC7q.js","names":[],"sources":["../src/paired-delta-test.ts","../src/paired-promotion-decision.ts","../src/campaign/gates/statistical-heldout.ts","../src/campaign/gates/power-preflight.ts"],"sourcesContent":["import {\n type PairedBootstrapOptions,\n type PairedBootstrapResult,\n pairedBootstrap,\n pairedSignTest,\n} from './statistics'\n\nexport interface PairedDeltaTestOptions extends PairedBootstrapOptions {\n /** Smallest candidate-minus-baseline delta that counts as improvement. Default 0. */\n threshold?: number\n /** Caller-required paired observations. The exact test may impose a higher minimum. */\n minPairs?: number\n}\n\nexport interface PairedDeltaTestResult {\n bootstrap: PairedBootstrapResult\n method: 'bootstrap-ci' | 'exact-sign'\n /** Exact one-sided p-value below the bootstrap minimum; otherwise null. */\n pValue: number | null\n /** Effective observation minimum after accounting for confidence. */\n minimumPairs: number\n sufficient: boolean\n /**\n * The bootstrap interval has zero width (or is non-finite), so it carries no\n * information about how far the estimate could be wrong and cannot support a\n * decision in either direction. See {@link pairedDeltaTest}.\n */\n indeterminate: boolean\n significant: boolean\n}\n\n/** Smallest all-positive sample that can clear a one-sided exact sign test. */\nexport function minimumPairsForPairedDeltaTest(confidence = 0.95): number {\n if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) {\n throw new Error(\n `minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`,\n )\n }\n const oneSidedAlpha = (1 - confidence) / 2\n return Math.ceil(Math.log2(1 / oneSidedAlpha))\n}\n\n/**\n * Tests whether a paired candidate-minus-baseline delta clears a threshold.\n *\n * At 20 or more pairs, the percentile bootstrap lower bound carries the\n * decision. Below that point the interval is descriptive only, so the function\n * switches to a pre-registered one-sided exact sign test. The exact path is\n * deliberately conservative: it requires both a point estimate above the\n * threshold and enough consistently positive paired differences.\n *\n * ## A zero-width interval is never significant\n *\n * When every paired delta is identical the resample distribution is a point\n * mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n\n * identical deltas of g. Neither says the effect is certain — both say the\n * sample carries no information about how far the estimate could be wrong, and\n * `low > threshold` then answers on the point estimate alone. It fails in both\n * directions: `[0, 0]` clears every NEGATIVE threshold, which is how a\n * tie-dominated pass/fail comparison laundered a regression into a\n * noninferiority pass, and `[g, g]` clears every threshold below g with no\n * spread behind it. Under a bounded asymmetric null whose true mean paired\n * delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —\n * every sample that misses the drop is exactly that shape, and deciding on\n * `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.\n *\n * So `indeterminate` is reported and `significant` is false whenever the\n * interval has zero width, on BOTH paths: at small n the exact sign test is a\n * test of the MEDIAN and a zero-spread sample is precisely where it stops\n * saying anything about the mean the caller is thresholding.\n *\n * `threshold` may be negative — that is a noninferiority margin, and it is the\n * regime the zero-width hole is worst in. For a two-point (pass/fail) outcome\n * the percentile bootstrap is not a valid interval at a nonzero margin at all;\n * use {@link decidePairedPromotion}, which routes those to Tango's score\n * interval, rather than thresholding this function's bootstrap directly.\n */\nexport function pairedDeltaTest(\n before: number[],\n after: number[],\n options: PairedDeltaTestOptions = {},\n): PairedDeltaTestResult {\n const threshold = options.threshold ?? 0\n if (!Number.isFinite(threshold)) {\n throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`)\n }\n const confidence = options.confidence ?? 0.95\n const exactMinimum = minimumPairsForPairedDeltaTest(confidence)\n const requestedMinimum = options.minPairs ?? exactMinimum\n if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) {\n throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`)\n }\n const minimumPairs = Math.max(requestedMinimum, exactMinimum)\n const bootstrap = pairedBootstrap(before, after, options)\n const sufficient = bootstrap.n >= minimumPairs\n // A point-mass resample distribution is an absence of evidence, not a\n // certainty, and it is the shape that clears every negative threshold. Both\n // paths refuse it: see the \"zero-width\" section of this function's docs.\n const indeterminate =\n bootstrap.n > 0 &&\n (!Number.isFinite(bootstrap.low) ||\n !Number.isFinite(bootstrap.high) ||\n bootstrap.low === bootstrap.high)\n\n if (bootstrap.gateEligible) {\n return {\n bootstrap,\n method: 'bootstrap-ci',\n pValue: null,\n minimumPairs,\n sufficient,\n indeterminate,\n significant: sufficient && !indeterminate && bootstrap.low > threshold,\n }\n }\n\n const differences = before.map((value, index) => after[index]! - value - threshold)\n const exact = pairedSignTest(differences, 'greater')\n const estimate = options.statistic === 'mean' ? bootstrap.mean : bootstrap.median\n return {\n bootstrap,\n method: 'exact-sign',\n pValue: exact.pValue,\n minimumPairs,\n sufficient,\n indeterminate,\n significant:\n sufficient &&\n !indeterminate &&\n estimate > threshold &&\n exact.pValue <= (1 - bootstrap.confidence) / 2,\n }\n}\n","/**\n * @module\n * ONE rule for \"does this paired interval clear a promotion threshold\".\n *\n * The rule below was derived on `HeldOutGate` (#479) after the same estimator\n * bug shipped twice. It then turned out that a SECOND gate — the composable\n * `heldOutGate`, plus everything else routed through `heldoutSignificance` —\n * still carried the original defect, because the rule had been written into one\n * gate's method body rather than into a shared function. Two copies of a\n * statistical rule is how a defect survives in one of them, so there is now\n * exactly one copy and both gates call it.\n *\n * Three things the rule does that a bare `pairedBootstrap(...).low > threshold`\n * does not:\n *\n * 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a\n * pass/fail eval the paired delta vector is dominated by ties, so the\n * bootstrap of the mean is a resample of a lattice with three atoms and its\n * percentile interval is not valid at a nonzero margin. The score interval\n * (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under\n * each hypothesised margin instead of fixing it at the observed value, which\n * is the only construction that stays a confidence interval as the margin\n * moves off zero — the regime every noninferiority threshold lives in.\n * Measured on the composable gate before this change, at a true risk\n * difference sitting exactly on the production caller's -0.05 margin and a\n * nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.\n * 2. **McNemar's exact test holds a VETO at every non-negative threshold.**\n * Redundant with the interval by construction and kept anyway, so that\n * swapping the estimator for one without that duality cannot silently\n * reintroduce \"promotes what the exact test refuses\". Witness: n = 6, b = 5,\n * c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs\n * (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE\n * threshold is a noninferiority question, which McNemar's test of \"no\n * difference\" is not the right test for, so the veto does not apply there.\n * 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it\n * cannot tell a gain from a regression and clears every negative threshold.\n * Away from zero it fails the opposite way: n identical positive deltas give\n * [g, g], which clears threshold 0 on no spread at all. Both are an absence\n * of evidence. Measured on the composable gate before this change, under a\n * bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %\n * false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.\n *\n * Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that\n * picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered\n * exact sign test below it), this picks the ESTIMATOR from the outcome's shape.\n * Both are needed — an exact sign test applied to a tie-pinned median is still\n * blind, and a mean bootstrap CI at n = 6 is still not a valid test.\n */\n\nimport { minimumPairsForPairedDeltaTest, pairedDeltaTest } from './paired-delta-test'\nimport {\n type PairedBootstrapResult,\n pairedBinaryScale,\n pairedDeltaTieFraction,\n pairedRiskDifferenceExact,\n pairedRiskDifferenceScore,\n} from './statistics'\n\n/** Which paired estimator produced the deciding interval. */\nexport type PairedDecisionStatistic =\n | 'paired_risk_difference'\n | 'mean_bootstrap'\n | 'median_bootstrap'\n\n/** Which test carried the decision, given the estimator and the sample size. */\nexport type PairedDecisionMethod = 'score-interval' | 'bootstrap-ci' | 'exact-sign'\n\n/** McNemar's exact paired-binary evidence, on the two-point path only. */\nexport interface PairedMcNemarEvidence {\n /** Discordant pairs the treatment won. */\n b: number\n /** Discordant pairs the control won. */\n c: number\n /** b + c — the only pairs carrying information. */\n nDiscordant: number\n /** Two-sided exact p-value. */\n pValue: number\n}\n\nexport interface PairedPromotionDecisionOptions {\n /** Smallest candidate-minus-baseline delta that counts as improvement, in the\n * caller's native units. May be negative (a noninferiority margin). Default 0. */\n threshold?: number\n /** Confidence level. Default 0.95. */\n confidence?: number\n /** Bootstrap resamples, on the paths where a bootstrap decides. Default 2000. */\n resamples?: number\n /** Deterministic bootstrap seed. Omitted ⇒ derived from the deltas. */\n seed?: number\n /** Caller-required paired observations. The exact test may impose a higher\n * minimum; the effective one is reported as `minimumPairs`. */\n minPairs?: number\n /**\n * `'mean'` (default) routes by SHAPE: a two-point (pass/fail) outcome on any\n * encoding decides on the score interval, everything else on the mean\n * bootstrap. `'median'` forces the median bootstrap on every input, including\n * shapes where it is structurally blind — kept for callers who want outlier\n * robustness on genuinely continuous outcomes and accept that cost.\n */\n statistic?: 'mean' | 'median'\n}\n\nexport interface PairedPromotionDecision {\n /** Paired observations supplied. */\n n: number\n /** Threshold the interval was judged against, native units. */\n threshold: number\n confidence: number\n statistic: PairedDecisionStatistic\n method: PairedDecisionMethod\n /** Common positive level of a two-point outcome ({0,1} ⇒ 1, {0,100} ⇒ 100),\n * or null when the outcome is not two-point. Non-null is exactly the\n * condition for the `paired_risk_difference` path, and it is the factor\n * `delta` / `low` / `high` were rescaled by. */\n binaryScale: number | null\n /** Exact-tie fraction over the paired deltas; null when there are no pairs. */\n tieFraction: number | null\n /** Point estimate of the DECIDING statistic, in the caller's native units. */\n delta: number\n /** Lower bound of the DECIDING interval, native units. */\n low: number\n /** Upper bound of the DECIDING interval, native units. */\n high: number\n /** The bootstrap that decided, or null when the score interval did. Callers\n * that need a bootstrap as a diagnostic on the two-point path compute their\n * own — it is not computed here, so the binary path costs no resamples. */\n bootstrap: PairedBootstrapResult | null\n /** McNemar's exact evidence, or null off the two-point path. */\n mcnemar: PairedMcNemarEvidence | null\n /** Exact one-sided sign-test p-value on the small-sample bootstrap path;\n * null otherwise. */\n pValue: number | null\n /** Effective observation minimum after accounting for confidence. */\n minimumPairs: number\n /** n >= minimumPairs. */\n sufficient: boolean\n /** The deciding interval is zero-width or non-finite — no evidence in either\n * direction, so it cannot clear any threshold on evidence. */\n indeterminate: boolean\n /** McNemar's exact test refuses at a non-negative threshold. */\n exactTestVetoes: boolean\n /** The deciding interval clears the threshold, ignoring the other two guards. */\n clearsThreshold: boolean\n /** `sufficient && !indeterminate && clearsThreshold && !exactTestVetoes` —\n * the whole rule. */\n promote: boolean\n /** What `delta` measures, for a reason string. */\n label: 'success-rate' | 'mean' | 'median'\n /** Why a zero-width interval is zero-width; empty when it is not. */\n indeterminateCause: string\n /** Sentence naming the test when the exact sign test decided; else empty. */\n methodDetail: string\n}\n\n/** The shape facts that pick the estimator, without computing an interval. */\nexport interface PairedDecisionShape {\n statistic: PairedDecisionStatistic\n /** Common positive level of a two-point outcome; null when not two-point. */\n binaryScale: number | null\n /** Exact-tie fraction over the paired deltas; null when there are no pairs. */\n tieFraction: number | null\n}\n\n/**\n * Which estimator {@link decidePairedPromotion} would use on this data, and the\n * shape facts behind it — for callers that must report the shape on a path\n * where no interval is computed at all (an early rejection, or zero pairs).\n * Cheap: no bootstrap, no interval.\n */\nexport function pairedDecisionShape(\n before: number[],\n after: number[],\n statistic: 'mean' | 'median' = 'mean',\n): PairedDecisionShape {\n const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after)\n if (statistic === 'median') {\n return { statistic: 'median_bootstrap', binaryScale: null, tieFraction }\n }\n const binaryScale = pairedBinaryScale(before, after)\n if (binaryScale !== null) {\n return { statistic: 'paired_risk_difference', binaryScale, tieFraction }\n }\n return { statistic: 'mean_bootstrap', binaryScale: null, tieFraction }\n}\n\n/**\n * Decide whether a paired candidate-minus-baseline delta clears a promotion\n * threshold. `before` is the baseline arm, `after` the candidate arm, paired by\n * position. Throws on unequal lengths.\n */\nexport function decidePairedPromotion(\n before: number[],\n after: number[],\n options: PairedPromotionDecisionOptions = {},\n): PairedPromotionDecision {\n if (before.length !== after.length) {\n throw new Error(\n `decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`,\n )\n }\n const threshold = options.threshold ?? 0\n if (!Number.isFinite(threshold)) {\n throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`)\n }\n const confidence = options.confidence ?? 0.95\n const exactMinimum = minimumPairsForPairedDeltaTest(confidence)\n const requestedMinimum = options.minPairs ?? exactMinimum\n if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) {\n throw new Error(\n `decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`,\n )\n }\n const minimumPairs = Math.max(requestedMinimum, exactMinimum)\n const n = before.length\n const sufficient = n >= minimumPairs\n const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic)\n\n let core: {\n statistic: PairedDecisionStatistic\n method: PairedDecisionMethod\n delta: number\n low: number\n high: number\n bootstrap: PairedBootstrapResult | null\n mcnemar: PairedMcNemarEvidence | null\n pValue: number | null\n clearsThreshold: boolean\n label: PairedPromotionDecision['label']\n methodDetail: string\n }\n\n if (binaryScale !== null) {\n // Normalise the two-point encoding to {0,1} so the estimators see the\n // pass/fail structure, then rescale the answer back into the caller's\n // native units — the threshold is read in the units of the scores, so a\n // 0-100 pass/fail dimension must be gated in points, not in a rate.\n const unitControl = before.map((v) => v / binaryScale)\n const unitTreatment = after.map((v) => v / binaryScale)\n // TWO estimators with different jobs, because no single one does both.\n // `exact` is the authority on RD = 0 and supplies the veto; its\n // Clopper-Pearson interval conditions on the discordant count and is\n // therefore NOT a confidence interval at a nonzero margin. `score` is\n // Tango's, which is.\n const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence)\n const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence)\n const low = score.lower * binaryScale\n core = {\n statistic: 'paired_risk_difference',\n method: 'score-interval',\n delta: score.riskDifference * binaryScale,\n low,\n high: score.upper * binaryScale,\n bootstrap: null,\n mcnemar: {\n b: exact.b,\n c: exact.c,\n nDiscordant: exact.nDiscordant,\n pValue: exact.pValue,\n },\n pValue: null,\n clearsThreshold: low > threshold,\n label: 'success-rate',\n methodDetail: '',\n }\n } else {\n const bootstrapStatistic = options.statistic === 'median' ? 'median' : 'mean'\n const test = pairedDeltaTest(before, after, {\n confidence,\n resamples: options.resamples,\n statistic: bootstrapStatistic,\n seed: options.seed,\n threshold,\n minPairs: options.minPairs,\n })\n const ci = test.bootstrap\n core = {\n statistic: bootstrapStatistic === 'mean' ? 'mean_bootstrap' : 'median_bootstrap',\n method: test.method,\n delta: bootstrapStatistic === 'mean' ? ci.mean : ci.median,\n low: ci.low,\n high: ci.high,\n bootstrap: ci,\n mcnemar: null,\n pValue: test.pValue,\n clearsThreshold: test.significant,\n label: bootstrapStatistic,\n methodDetail:\n test.method === 'exact-sign'\n ? ` Below ${test.minimumPairs} pairs the interval is descriptive only;` +\n ` the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.`\n : '',\n }\n }\n\n const indeterminate =\n !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.low === core.high\n const indeterminateCause = !indeterminate\n ? ''\n : tieFraction === 1\n ? 'every paired delta is an exact tie'\n : core.mcnemar !== null && core.mcnemar.nDiscordant === 0\n ? 'every pair is concordant (0 discordant pairs)'\n : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`\n // Only at a non-negative threshold: a negative threshold asks a\n // noninferiority question, which McNemar's test of \"no difference\" does not\n // answer.\n const exactTestVetoes =\n core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence)\n\n return {\n n,\n threshold,\n confidence,\n binaryScale,\n tieFraction,\n minimumPairs,\n sufficient,\n indeterminate,\n indeterminateCause,\n exactTestVetoes,\n promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,\n ...core,\n }\n}\n\nfunction fmt(x: number): string {\n return x.toFixed(4)\n}\n","/**\n * Statistical held-out promotion machinery — the trustworthy core the\n * point-estimate `heldout-delta` gate lacked.\n *\n * The shipped false positive it prevents: a winner re-scored against the\n * baseline on the holdout read run-to-run model NOISE (e.g. 91 vs 95) as a\n * \"+4 lift\" and shipped, because the gate compared point estimates with no\n * confidence interval. Here we pair candidate vs baseline holdout observations\n * and bootstrap a CI on the paired delta — a candidate ships only when the CI\n * lower bound clears the effect-size threshold (the gain is real at the\n * confidence level, not noise), and is blocked when a critical dimension\n * (e.g. `hallucination_free` for a legal agent) significantly regresses even if\n * the net composite rose (anti-Goodhart).\n *\n * Two traps this module is built around (both produce a NEW false positive if\n * gotten wrong):\n * 1. PAIRING GRANULARITY — pairs by FULL `cellId` (`scenario:rep`), never by\n * `scenarioId` (which averages reps away and destroys the within-pair\n * variance reduction that makes a paired bootstrap tighter than unpaired).\n * One paired observation per cell ⇒ reps multiply n.\n * 2. SCALE — a judge may emit composites/dimensions on [0,1] or 0-100. The\n * threshold + tolerance are interpreted in the judge's NATIVE scale; the\n * per-dimension tolerance auto-scales off the observed baseline magnitudes\n * so `-0.10` on [0,1] doesn't silently become a no-op on a 0-100 dimension.\n */\n\nimport {\n decidePairedPromotion,\n type PairedDecisionMethod,\n type PairedDecisionStatistic,\n type PairedMcNemarEvidence,\n type PairedPromotionDecision,\n} from '../../paired-promotion-decision'\nimport {\n DECISION_PAIRED_DELTA_STATISTIC,\n type PairedBootstrapResult,\n pairedBootstrap,\n} from '../../statistics'\nimport type { JudgeScore } from '../types'\n\n/** Tie fraction at/above which a gate annotates its verdict with the tie share.\n * Tie-domination of the median bites structurally at >= 0.5 (the median is then\n * 0 by construction); 0.4 is a softer warn threshold that flags a run APPROACHING\n * that regime, so an operator sees it before the median goes fully blind. */\nexport const TIE_WARN_FRACTION = 0.4\n\nexport interface PairedHoldout {\n /** Baseline scalar per paired cell (same order as `after`/`cellIds`). */\n before: number[]\n /** Candidate scalar per paired cell. */\n after: number[]\n /** The full cellIds (`scenario:rep`) that paired, in order. */\n cellIds: string[]\n}\n\n/**\n * Pair candidate vs baseline holdout observations by FULL cellId. `select`\n * pulls the scalar from a cell's judge reports (composite, or a named\n * dimension); a cell contributes the mean of `select` across its judges. Cells\n * whose scenario is not in `scenarioIds`, or where `select` is undefined for\n * every judge on either side, are skipped on BOTH sides so the arrays stay\n * paired. Throws when the two maps disagree on which holdout cells exist — a\n * load-bearing invariant: the baseline + winner holdout campaigns run the same\n * scenarios with the same seed base, so their cellIds MUST align; a mismatch\n * means a silent pairing bug, not a soft fallback.\n */\nexport function pairHoldout(\n candidate: Map<string, Record<string, JudgeScore>>,\n baseline: Map<string, Record<string, JudgeScore>>,\n scenarioIds: Set<string>,\n select: (s: JudgeScore) => number | undefined,\n): PairedHoldout {\n const cellValue = (\n byCell: Map<string, Record<string, JudgeScore>>,\n cellId: string,\n ): number | undefined => {\n const scores = byCell.get(cellId)\n if (!scores) return undefined\n const vals: number[] = []\n for (const s of Object.values(scores)) {\n if (s.failed === true) {\n throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`)\n }\n const v = select(s)\n if (typeof v === 'number' && !Number.isFinite(v)) {\n throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`)\n }\n if (typeof v === 'number') vals.push(v)\n }\n if (vals.length === 0) return undefined\n return vals.reduce((a, b) => a + b, 0) / vals.length\n }\n\n const inScope = (cellId: string) => scenarioIds.has(cellId.split(':')[0] ?? '')\n const candCells = [...candidate.keys()].filter(inScope).sort()\n const baseCells = [...baseline.keys()].filter(inScope).sort()\n // Alignment invariant — the holdout campaigns share scenarios + seed, so the\n // cell sets must be identical. Differ ⇒ a real pairing bug; fail loud.\n if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) {\n throw new Error(\n `pairHoldout: candidate/baseline holdout cells do not align — ` +\n `candidate=[${candCells.join(',')}] baseline=[${baseCells.join(',')}]. ` +\n `Both holdout campaigns must run the same scenarios with the same seed base.`,\n )\n }\n\n const before: number[] = []\n const after: number[] = []\n const cellIds: string[] = []\n for (const cellId of candCells) {\n const b = cellValue(baseline, cellId)\n const a = cellValue(candidate, cellId)\n // A scalar absent on both sides means that dimension was not scored. A\n // one-sided absence is asymmetric evidence loss, never a row to discard.\n if (b === undefined && a === undefined) continue\n if (b === undefined || a === undefined) {\n throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`)\n }\n before.push(b)\n after.push(a)\n cellIds.push(cellId)\n }\n return { before, after, cellIds }\n}\n\nexport interface HeldoutSignificance {\n paired: PairedHoldout\n /**\n * The paired bootstrap on the requested statistic (MEAN by default — see the\n * tie note on `heldoutSignificance`).\n *\n * DIAGNOSTIC, not necessarily the interval the verdict keyed on. On a\n * two-point (pass/fail) outcome the decision routes to Tango's score interval\n * instead, because a percentile bootstrap of the mean over a three-atom\n * lattice is not a valid interval at a nonzero margin. Read\n * `decision.low`/`decision.high` for the interval that actually decided, and\n * `decisionStatistic` for which one it is.\n */\n bootstrap: PairedBootstrapResult\n /** The MEDIAN paired-delta bootstrap, reported as a diagnostic. When many\n * scenarios are tied (both sides solve them), the median is pinned near 0\n * regardless of the mean lift — comparing the two exposes tie-domination. */\n medianBootstrap: PairedBootstrapResult\n /**\n * The full promotion decision: which estimator the outcome's shape admits,\n * the interval it produced, McNemar's exact veto on the two-point path, and\n * whether the interval was zero-width (no evidence in either direction). The\n * single source of `significant`.\n */\n decision: PairedPromotionDecision\n /** Which paired estimator the verdict was decided on. */\n decisionStatistic: PairedDecisionStatistic\n /** McNemar's exact evidence on the two-point path; null otherwise. */\n mcnemar: PairedMcNemarEvidence | null\n /** Fraction of paired observations that are exact ties (|delta| < 1e-9). A\n * high tie fraction is WHY a median-based gate would have missed a real lift;\n * it is the observability the tie fix adds. */\n tieFraction: number\n /** n paired observations. */\n n: number\n /** Effective minimum after applying the bootstrap's hard statistical floor. */\n minimumRequired: number\n /** Statistical method that carried the decision. */\n decisionMethod: PairedDecisionMethod\n /** Exact one-sided p-value on the small-sample path; otherwise null. */\n pValue: number | null\n /** True iff n >= minimumRequired, the DECIDING interval has nonzero width,\n * its lower bound clears the threshold, and McNemar's exact test does not\n * veto at a non-negative threshold. */\n significant: boolean\n /** Set when n < minimumRequired — too little evidence to claim significance. */\n fewRuns: boolean\n}\n\nexport interface HeldoutSignificanceOptions {\n deltaThreshold?: number\n minProductiveRuns?: number\n confidence?: number\n resamples?: number\n /** Fixed by default for a deterministic, reproducible gate verdict. */\n seed?: number\n statistic?: 'mean' | 'median'\n}\n\n/**\n * Significance of the held-out composite lift: ship only when the lower bound\n * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default\n * 0 ⇒ \"confidently positive\"). Interpret `deltaThreshold` in the judge's native\n * scale.\n *\n * The decision is delegated whole to {@link decidePairedPromotion}, the one\n * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`\n * also calls. That module's header carries the measurements; the short version\n * is three guards a bare `bootstrap.low > threshold` does not have:\n *\n * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the\n * only paired-binary construction that stays valid at a nonzero margin;\n * - McNemar's exact test VETOES at any non-negative threshold;\n * - a ZERO-WIDTH interval is refused rather than promoted, in either\n * direction — [0,0] clears every negative threshold and [g,g] clears every\n * threshold below g, and both are an absence of evidence, not a result.\n *\n * Measured on this function before those guards landed, at a nominal 5 %:\n * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,\n * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired\n * delta is exactly 0.\n *\n * At small n, where the percentile bootstrap is descriptive only, a\n * pre-registered exact sign test still carries the bootstrap path.\n */\nexport function heldoutSignificance(\n paired: PairedHoldout,\n opts: HeldoutSignificanceOptions = {},\n): HeldoutSignificance {\n const deltaThreshold = opts.deltaThreshold ?? 0\n const confidence = opts.confidence ?? 0.95\n const resamples = opts.resamples ?? 2000\n const seed = opts.seed ?? 1337\n // DEFAULT to the MEAN paired delta, not the median. The median is destroyed by\n // TIES: whenever both baseline and candidate solve a holdout scenario (a common\n // case once the agent is decent — and INCREASINGLY common as you add holdout\n // scenarios for statistical power), that scenario contributes delta 0. When\n // >=50% of paired cells are ties the median is pinned at 0 regardless of a large,\n // consistent lift on the rest, and the gate holds a genuinely better candidate.\n // (Measured live, supervisor-lab run 6: 40 cells, 20 ties, MEAN +0.177, MEDIAN 0\n // → false hold. Doubling the holdout for \"power\" made it WORSE by adding ties.)\n // The mean equals the reported aggregate lift, ties correctly contribute 0\n // without dominating, and it is the textbook paired-comparison estimator; the\n // median is kept as a reported diagnostic. Callers wanting outlier-robustness at\n // the cost of tie-blindness can still pass `statistic: 'median'`.\n const statistic = opts.statistic ?? 'mean'\n const decision = decidePairedPromotion(paired.before, paired.after, {\n confidence,\n resamples,\n statistic,\n seed,\n threshold: deltaThreshold,\n minPairs: opts.minProductiveRuns,\n })\n // `decision.bootstrap` is null exactly when the score interval decided, so\n // the requested-statistic bootstrap is computed here for the diagnostic\n // field. Same two bootstraps as before on every path.\n const bootstrap =\n decision.bootstrap ??\n pairedBootstrap(paired.before, paired.after, { confidence, resamples, statistic, seed })\n const medianBootstrap =\n statistic === 'median'\n ? bootstrap\n : pairedBootstrap(paired.before, paired.after, {\n confidence,\n resamples,\n statistic: 'median',\n seed,\n })\n const n = paired.before.length\n let ties = 0\n for (let i = 0; i < n; i += 1) {\n const after = paired.after[i] ?? 0\n const before = paired.before[i] ?? 0\n if (Math.abs(after - before) < 1e-9) ties += 1\n }\n const tieFraction = n === 0 ? 0 : ties / n\n return {\n paired,\n bootstrap,\n medianBootstrap,\n decision,\n decisionStatistic: decision.statistic,\n mcnemar: decision.mcnemar,\n tieFraction,\n n,\n minimumRequired: decision.minimumPairs,\n decisionMethod: decision.method,\n pValue: decision.pValue,\n significant: decision.promote,\n fewRuns: !decision.sufficient,\n }\n}\n\nexport interface DimensionRegression {\n dimension: string\n /** Paired bootstrap on (candidate − baseline). DIAGNOSTIC on a pass/fail\n * dimension, where `ci` carries the interval that decided instead. */\n bootstrap: PairedBootstrapResult\n /** Which paired statistic `bootstrap.low` is the lower bound of. `'mean'`\n * unless the caller asked for the median. `bootstrap.median` still carries\n * the median point estimate either way. */\n bootstrapStatistic: 'median' | 'mean'\n /** The interval `regressed` was decided on, in the dimension's native units. */\n ci: { low: number; high: number }\n /** Which estimator produced `ci`. */\n decisionStatistic: PairedDecisionStatistic\n /** McNemar's exact evidence on a pass/fail dimension; null otherwise. */\n mcnemar: PairedMcNemarEvidence | null\n /** `ci` has zero width — no evidence in either direction. */\n indeterminate: boolean\n /** True iff the candidate may have regressed this dimension by more than\n * tolerance: the lower bound of the DECIDING interval on (candidate −\n * baseline) is below −tolerance, OR the exact small-sample test proves a drop\n * past tolerance. */\n regressed: boolean\n tolerance: number\n n: number\n}\n\n/** Detect the native scale of a set of scores: 0-100 when any magnitude clears\n * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default\n * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */\nexport function detectScale(values: number[]): 1 | 100 {\n return values.some((v) => Math.abs(v) > 1.5) ? 100 : 1\n}\n\n/** Per-critical-dimension regression guard. For each dimension, pair the\n * candidate vs baseline values by full cellId and bootstrap the paired delta;\n * a dimension is \"regressed\" when the CI lower bound < −tolerance (conservative\n * — blocks if the credible worst case exceeds tolerance, which is the right\n * posture for safety dimensions like `hallucination_free`). When `tolerance`\n * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.\n *\n * The interval comes from {@link decidePairedPromotion}, so a pass/fail\n * dimension is judged on Tango's score interval rather than a percentile\n * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap\n * is not a valid interval at one. That matters most here because this guard\n * fails OPEN by construction: `tolerance` is positive, so an interval pinned at\n * [0,0] never satisfies `low < −tolerance` and a real regression on a safety\n * dimension would be reported as `regressed: false`. On the median it fails the\n * same way for the same reason — when most pairs tie, which is automatic for a\n * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists\n * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to\n * restore the pre-0.134 behaviour. */\nexport function dimensionRegressions(\n candidate: Map<string, Record<string, JudgeScore>>,\n baseline: Map<string, Record<string, JudgeScore>>,\n scenarioIds: Set<string>,\n criticalDimensions: string[],\n opts: {\n tolerance?: number\n confidence?: number\n resamples?: number\n seed?: number\n /** Paired statistic the CI is computed on. Default `'mean'` — see\n * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */\n statistic?: 'mean' | 'median'\n } = {},\n): DimensionRegression[] {\n const out: DimensionRegression[] = []\n for (const dim of criticalDimensions) {\n const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim])\n if (paired.before.length === 0) continue // dimension not scored on this judge\n const tolerance = opts.tolerance ?? 0.05 * detectScale([...paired.before, ...paired.after])\n const bootstrapStatistic = opts.statistic ?? DECISION_PAIRED_DELTA_STATISTIC\n const shared = {\n confidence: opts.confidence ?? 0.95,\n resamples: opts.resamples ?? 2000,\n statistic: bootstrapStatistic,\n seed: opts.seed ?? 1337,\n }\n const guard = decidePairedPromotion(paired.before, paired.after, shared)\n const regression = decidePairedPromotion(paired.after, paired.before, {\n ...shared,\n threshold: tolerance,\n })\n const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared)\n out.push({\n dimension: dim,\n bootstrap,\n bootstrapStatistic,\n ci: { low: guard.low, high: guard.high },\n decisionStatistic: guard.statistic,\n mcnemar: guard.mcnemar,\n indeterminate: guard.indeterminate,\n // Fires on EITHER burden of proof — see `floorBreached` in\n // `promotion-policy.ts` for the same rule and the reasoning. The CI arm\n // is the contract this interface and `default-production-gate`'s reason\n // string both state (\"CI.low < -tolerance\"), read off the DECIDING\n // interval; the exact-test arm adds the small-sample path where the\n // bootstrap interval is descriptive only.\n // The credible-worst-case arm stays on the BOOTSTRAP — see the same\n // decision in `floorBreached` (`promotion-policy.ts`). On a pass/fail\n // dimension the score interval is ±z²/(n+z²) even when every pair is\n // concordant, so reading the floor off it would flag an unchanged safety\n // dimension as regressed at any realistic n. The PROVEN-drop arm does use\n // the shared rule, which is what closes the fail-open hole that mattered:\n // a genuine pass/fail regression is now judged on an interval valid at\n // the nonzero `tolerance`, instead of a bootstrap that is pinned wherever\n // ties dominate.\n regressed: bootstrap.low < -tolerance || regression.promote,\n tolerance,\n n: paired.before.length,\n })\n }\n return out\n}\n","/**\n * Power preflight — \"can this budget detect the effect you are hunting?\"\n *\n * The failure it prevents (measured, twice): a live prompt-improvement campaign ran\n * 333 sandbox cells over 5.6 hours and produced a +0.08 holdout lift the ship gate\n * (paired bootstrap, CI.low > 0.05) could not distinguish from zero — because at\n * that holdout size and worker variance the MINIMUM DETECTABLE lift was larger than\n * any effect a prompt change plausibly produces. The budget was spent learning what\n * a 30-second calculation on the baseline cells already knew. No eval framework we\n * know of surfaces this; every underpowered improvement run everywhere ends in an\n * uninformative \"hold\".\n *\n * Model: the ship rule is `CI.low(paired Δ) > deltaThreshold`. Approximating the\n * bootstrap CI as normal, `CI.low ≈ effect − z·sd_Δ/√n`, so the smallest shippable\n * true effect is `MDE = deltaThreshold + z·sd_Δ/√n`. The paired-delta SD is unknown\n * before the candidate exists; we bound it by the zero-correlation case\n * `sd_Δ ≤ √2·sd_baseline` — a CONSERVATIVE (upper) MDE, which is the correct\n * direction for a warning. Pairing is per cell (`scenario:rep`), so reps multiply n.\n *\n * Standalone by design: feed it any baseline composites (a `gate:'none'` run, a\n * live-proof table) BEFORE budgeting the real search; `selfImprove` also attaches\n * it to every result and warns when the run was structurally unable to ship.\n */\n\nexport interface PowerPreflightOptions {\n /** Per-cell baseline composites on the HOLDOUT scenarios (one per scenario:rep cell). */\n baselineComposites: number[]\n /** Paired observations the budgeted comparison will produce\n * (holdout scenarios × reps). Defaults to `baselineComposites.length`. */\n pairedN?: number\n /** The ship gate's effect-size threshold. Default 0.05 (defaultProductionGate). */\n deltaThreshold?: number\n /** CI confidence the gate uses. Default 0.95. */\n confidence?: number\n /** True when the holdout is scored by the SAME judge/scorer family as the gate\n * (selfImprove's default composition — one judge scores everything). Under a\n * shared channel, raising paired n reduces only the IDIOSYNCRATIC noise share;\n * systematic judge bias is untouched, so the MDE here is a lower bound and the\n * only full debiaser is an independent second scoring channel\n * (recursive-self-improvement S1c, closed form in EXP-023 P0). Default false. */\n sharedScorerChannel?: boolean\n}\n\nexport interface PowerPreflight {\n /** Paired observations the comparison will have. */\n n: number\n /** Baseline per-cell composite standard deviation (the variance the effect must beat). */\n sd: number\n /** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */\n mde: number\n /** Baseline holdout composite mean. */\n baselineMean: number\n /** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */\n headroom: number\n /** True when even the largest achievable effect (headroom) is below the MDE —\n * the run is structurally unable to ship regardless of proposal quality.\n * Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */\n underpowered: boolean\n /** True when composites look [0,1]-scaled; headroom/underpowered are only\n * meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */\n scaleAssumed: boolean\n deltaThreshold: number\n confidence: number\n /** Set when the holdout shares the gate's scoring channel: more cells cannot\n * buy back systematic judge bias — treat the MDE as a lower bound. */\n sharedChannelCaveat?: string\n /** One actionable sentence for humans and logs. */\n recommendation: string\n}\n\n/** Two-sided z for the common confidence levels; interpolation is overkill here. */\nfunction zFor(confidence: number): number {\n if (confidence >= 0.99) return 2.576\n if (confidence >= 0.95) return 1.96\n if (confidence >= 0.9) return 1.645\n return 1.282\n}\n\n/** Estimate the minimum detectable lift a paired-holdout improvement run can\n * ship at a given budget, from the baseline holdout composites — call it BEFORE\n * spending a search to learn whether the effect you are hunting is even\n * observable at this holdout size and worker variance. */\nexport function powerPreflight(opts: PowerPreflightOptions): PowerPreflight {\n const composites = opts.baselineComposites.filter((v) => Number.isFinite(v))\n if (composites.length < 3) {\n throw new Error(\n `powerPreflight: need >= 3 finite baseline composites to estimate variance, got ${composites.length}`,\n )\n }\n const deltaThreshold = opts.deltaThreshold ?? 0.05\n const confidence = opts.confidence ?? 0.95\n const n = opts.pairedN ?? composites.length\n if (n < 2) throw new Error(`powerPreflight: pairedN must be >= 2, got ${n}`)\n\n const mean = composites.reduce((a, b) => a + b, 0) / composites.length\n const variance =\n composites.reduce((a, b) => a + (b - mean) * (b - mean), 0) / (composites.length - 1)\n const sd = Math.sqrt(variance)\n const z = zFor(confidence)\n const mde = deltaThreshold + (z * Math.SQRT2 * sd) / Math.sqrt(n)\n\n const scaleAssumed = composites.every((v) => v >= -0.001 && v <= 1.5)\n const headroom = Math.max(0, 1 - mean)\n const underpowered = scaleAssumed && mde > headroom\n\n const sharedChannelCaveat = opts.sharedScorerChannel\n ? 'Holdout and gate share one scoring channel: raising n/reps reduces only idiosyncratic noise — systematic judge bias remains and this MDE is a lower bound. Full debiasing needs an independent second scoring channel (different judge/benchmark family).'\n : undefined\n\n const recommendation = underpowered\n ? `UNDERPOWERED: minimum detectable lift ${mde.toFixed(3)} exceeds the ${headroom.toFixed(3)} headroom above the baseline (${mean.toFixed(3)}) — no achievable effect can ship at this budget. Raise paired n (scenarios x reps) to ~${Math.ceil(((z * Math.SQRT2 * sd) / Math.max(headroom - deltaThreshold, 0.01)) ** 2)} or reduce worker variance before searching.`\n : `Minimum detectable lift at n=${n}: ${mde.toFixed(3)} (baseline sd ${sd.toFixed(3)}). Effects smaller than this cannot clear the gate; budget the search for effects you believe exceed it.`\n\n return {\n n,\n sd,\n mde,\n baselineMean: mean,\n headroom,\n underpowered,\n scaleAssumed,\n deltaThreshold,\n confidence,\n ...(sharedChannelCaveat ? { sharedChannelCaveat } : {}),\n recommendation: sharedChannelCaveat\n ? `${recommendation} ${sharedChannelCaveat}`\n : recommendation,\n }\n}\n"],"mappings":";;;;AAgCA,SAAgB,+BAA+B,aAAa,KAAc;CACxE,IAAI,CAAC,OAAO,SAAS,UAAU,KAAK,cAAc,KAAK,cAAc,GACnE,MAAM,IAAI,MACR,oEAAoE,YACtE;CAEF,MAAM,iBAAiB,IAAI,cAAc;CACzC,OAAO,KAAK,KAAK,KAAK,KAAK,IAAI,aAAa,CAAC;AAC/C;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCA,SAAgB,gBACd,QACA,OACA,UAAkC,CAAC,GACZ;CACvB,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,SAAS,SAAS,GAC5B,MAAM,IAAI,MAAM,kDAAkD,WAAW;CAG/E,MAAM,eAAe,+BADF,QAAQ,cAAc,GACqB;CAC9D,MAAM,mBAAmB,QAAQ,YAAY;CAC7C,IAAI,CAAC,OAAO,UAAU,gBAAgB,KAAK,mBAAmB,GAC5D,MAAM,IAAI,MAAM,6DAA6D,kBAAkB;CAEjG,MAAM,eAAe,KAAK,IAAI,kBAAkB,YAAY;CAC5D,MAAM,YAAY,gBAAgB,QAAQ,OAAO,OAAO;CACxD,MAAM,aAAa,UAAU,KAAK;CAIlC,MAAM,gBACJ,UAAU,IAAI,MACb,CAAC,OAAO,SAAS,UAAU,GAAG,KAC7B,CAAC,OAAO,SAAS,UAAU,IAAI,KAC/B,UAAU,QAAQ,UAAU;CAEhC,IAAI,UAAU,cACZ,OAAO;EACL;EACA,QAAQ;EACR,QAAQ;EACR;EACA;EACA;EACA,aAAa,cAAc,CAAC,iBAAiB,UAAU,MAAM;CAC/D;CAIF,MAAM,QAAQ,eADM,OAAO,KAAK,OAAO,UAAU,MAAM,SAAU,QAAQ,SAClC,GAAG,SAAS;CACnD,MAAM,WAAW,QAAQ,cAAc,SAAS,UAAU,OAAO,UAAU;CAC3E,OAAO;EACL;EACA,QAAQ;EACR,QAAQ,MAAM;EACd;EACA;EACA;EACA,aACE,cACA,CAAC,iBACD,WAAW,aACX,MAAM,WAAW,IAAI,UAAU,cAAc;CACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACqCA,SAAgB,oBACd,QACA,OACA,YAA+B,QACV;CACrB,MAAM,cAAc,OAAO,WAAW,IAAI,OAAO,uBAAuB,QAAQ,KAAK;CACrF,IAAI,cAAc,UAChB,OAAO;EAAE,WAAW;EAAoB,aAAa;EAAM;CAAY;CAEzE,MAAM,cAAc,kBAAkB,QAAQ,KAAK;CACnD,IAAI,gBAAgB,MAClB,OAAO;EAAE,WAAW;EAA0B;EAAa;CAAY;CAEzE,OAAO;EAAE,WAAW;EAAkB,aAAa;EAAM;CAAY;AACvE;;;;;;AAOA,SAAgB,sBACd,QACA,OACA,UAA0C,CAAC,GAClB;CACzB,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,MACR,gDAAgD,OAAO,OAAO,MAAM,MAAM,OAAO,EACnF;CAEF,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,SAAS,SAAS,GAC5B,MAAM,IAAI,MAAM,wDAAwD,WAAW;CAErF,MAAM,aAAa,QAAQ,cAAc;CACzC,MAAM,eAAe,+BAA+B,UAAU;CAC9D,MAAM,mBAAmB,QAAQ,YAAY;CAC7C,IAAI,CAAC,OAAO,UAAU,gBAAgB,KAAK,mBAAmB,GAC5D,MAAM,IAAI,MACR,mEAAmE,kBACrE;CAEF,MAAM,eAAe,KAAK,IAAI,kBAAkB,YAAY;CAC5D,MAAM,IAAI,OAAO;CACjB,MAAM,aAAa,KAAK;CACxB,MAAM,EAAE,aAAa,gBAAgB,oBAAoB,QAAQ,OAAO,QAAQ,SAAS;CAEzF,IAAI;CAcJ,IAAI,gBAAgB,MAAM;EAKxB,MAAM,cAAc,OAAO,KAAK,MAAM,IAAI,WAAW;EACrD,MAAM,gBAAgB,MAAM,KAAK,MAAM,IAAI,WAAW;EAMtD,MAAM,QAAQ,0BAA0B,aAAa,eAAe,UAAU;EAC9E,MAAM,QAAQ,0BAA0B,aAAa,eAAe,UAAU;EAC9E,MAAM,MAAM,MAAM,QAAQ;EAC1B,OAAO;GACL,WAAW;GACX,QAAQ;GACR,OAAO,MAAM,iBAAiB;GAC9B;GACA,MAAM,MAAM,QAAQ;GACpB,WAAW;GACX,SAAS;IACP,GAAG,MAAM;IACT,GAAG,MAAM;IACT,aAAa,MAAM;IACnB,QAAQ,MAAM;GAChB;GACA,QAAQ;GACR,iBAAiB,MAAM;GACvB,OAAO;GACP,cAAc;EAChB;CACF,OAAO;EACL,MAAM,qBAAqB,QAAQ,cAAc,WAAW,WAAW;EACvE,MAAM,OAAO,gBAAgB,QAAQ,OAAO;GAC1C;GACA,WAAW,QAAQ;GACnB,WAAW;GACX,MAAM,QAAQ;GACd;GACA,UAAU,QAAQ;EACpB,CAAC;EACD,MAAM,KAAK,KAAK;EAChB,OAAO;GACL,WAAW,uBAAuB,SAAS,mBAAmB;GAC9D,QAAQ,KAAK;GACb,OAAO,uBAAuB,SAAS,GAAG,OAAO,GAAG;GACpD,KAAK,GAAG;GACR,MAAM,GAAG;GACT,WAAW;GACX,SAAS;GACT,QAAQ,KAAK;GACb,iBAAiB,KAAK;GACtB,OAAO;GACP,cACE,KAAK,WAAW,eACZ,UAAU,KAAK,aAAa,4FACyB,IAAI,KAAK,UAAU,CAAC,EAAE,KAC3E;EACR;CACF;CAEA,MAAM,gBACJ,CAAC,OAAO,SAAS,KAAK,GAAG,KAAK,CAAC,OAAO,SAAS,KAAK,IAAI,KAAK,KAAK,QAAQ,KAAK;CACjF,MAAM,qBAAqB,CAAC,gBACxB,KACA,gBAAgB,IACd,uCACA,KAAK,YAAY,QAAQ,KAAK,QAAQ,gBAAgB,IACpD,kDACA,OAAO,KAAK,MAAM,8BAA8B,IAAI,KAAK,GAAG;CAIpE,MAAM,kBACJ,KAAK,YAAY,QAAQ,aAAa,KAAK,EAAE,KAAK,QAAQ,SAAS,IAAI;CAEzE,OAAO;EACL;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,SAAS,cAAc,CAAC,iBAAiB,KAAK,mBAAmB,CAAC;EAClE,GAAG;CACL;AACF;AAEA,SAAS,IAAI,GAAmB;CAC9B,OAAO,EAAE,QAAQ,CAAC;AACpB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AC3RA,MAAa,oBAAoB;;;;;;;;;;;;AAsBjC,SAAgB,YACd,WACA,UACA,aACA,QACe;CACf,MAAM,aACJ,QACA,WACuB;EACvB,MAAM,SAAS,OAAO,IAAI,MAAM;EAChC,IAAI,CAAC,QAAQ,OAAO,KAAA;EACpB,MAAM,OAAiB,CAAC;EACxB,KAAK,MAAM,KAAK,OAAO,OAAO,MAAM,GAAG;GACrC,IAAI,EAAE,WAAW,MACf,MAAM,IAAI,MAAM,sBAAsB,OAAO,gCAAgC;GAE/E,MAAM,IAAI,OAAO,CAAC;GAClB,IAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAC7C,MAAM,IAAI,MAAM,sBAAsB,OAAO,uCAAuC;GAEtF,IAAI,OAAO,MAAM,UAAU,KAAK,KAAK,CAAC;EACxC;EACA,IAAI,KAAK,WAAW,GAAG,OAAO,KAAA;EAC9B,OAAO,KAAK,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,KAAK;CAChD;CAEA,MAAM,WAAW,WAAmB,YAAY,IAAI,OAAO,MAAM,GAAG,CAAC,CAAC,MAAM,EAAE;CAC9E,MAAM,YAAY,CAAC,GAAG,UAAU,KAAK,CAAC,CAAC,CAAC,OAAO,OAAO,CAAC,CAAC,KAAK;CAC7D,MAAM,YAAY,CAAC,GAAG,SAAS,KAAK,CAAC,CAAC,CAAC,OAAO,OAAO,CAAC,CAAC,KAAK;CAG5D,IAAI,UAAU,WAAW,UAAU,UAAU,UAAU,MAAM,GAAG,MAAM,MAAM,UAAU,EAAE,GACtF,MAAM,IAAI,MACR,2EACgB,UAAU,KAAK,GAAG,EAAE,cAAc,UAAU,KAAK,GAAG,EAAE,+EAExE;CAGF,MAAM,SAAmB,CAAC;CAC1B,MAAM,QAAkB,CAAC;CACzB,MAAM,UAAoB,CAAC;CAC3B,KAAK,MAAM,UAAU,WAAW;EAC9B,MAAM,IAAI,UAAU,UAAU,MAAM;EACpC,MAAM,IAAI,UAAU,WAAW,MAAM;EAGrC,IAAI,MAAM,KAAA,KAAa,MAAM,KAAA,GAAW;EACxC,IAAI,MAAM,KAAA,KAAa,MAAM,KAAA,GAC3B,MAAM,IAAI,MAAM,sBAAsB,OAAO,uCAAuC;EAEtF,OAAO,KAAK,CAAC;EACb,MAAM,KAAK,CAAC;EACZ,QAAQ,KAAK,MAAM;CACrB;CACA,OAAO;EAAE;EAAQ;EAAO;CAAQ;AAClC;;;;;;;;;;;;;;;;;;;;;;;;;;;AAuFA,SAAgB,oBACd,QACA,OAAmC,CAAC,GACf;CACrB,MAAM,iBAAiB,KAAK,kBAAkB;CAC9C,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,OAAO,KAAK,QAAQ;CAa1B,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,WAAW,sBAAsB,OAAO,QAAQ,OAAO,OAAO;EAClE;EACA;EACA;EACA;EACA,WAAW;EACX,UAAU,KAAK;CACjB,CAAC;CAID,MAAM,YACJ,SAAS,aACT,gBAAgB,OAAO,QAAQ,OAAO,OAAO;EAAE;EAAY;EAAW;EAAW;CAAK,CAAC;CACzF,MAAM,kBACJ,cAAc,WACV,YACA,gBAAgB,OAAO,QAAQ,OAAO,OAAO;EAC3C;EACA;EACA,WAAW;EACX;CACF,CAAC;CACP,MAAM,IAAI,OAAO,OAAO;CACxB,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,GAAG;EAC7B,MAAM,QAAQ,OAAO,MAAM,MAAM;EACjC,MAAM,SAAS,OAAO,OAAO,MAAM;EACnC,IAAI,KAAK,IAAI,QAAQ,MAAM,IAAI,MAAM,QAAQ;CAC/C;CACA,MAAM,cAAc,MAAM,IAAI,IAAI,OAAO;CACzC,OAAO;EACL;EACA;EACA;EACA;EACA,mBAAmB,SAAS;EAC5B,SAAS,SAAS;EAClB;EACA;EACA,iBAAiB,SAAS;EAC1B,gBAAgB,SAAS;EACzB,QAAQ,SAAS;EACjB,aAAa,SAAS;EACtB,SAAS,CAAC,SAAS;CACrB;AACF;;;;AA+BA,SAAgB,YAAY,QAA2B;CACrD,OAAO,OAAO,MAAM,MAAM,KAAK,IAAI,CAAC,IAAI,GAAG,IAAI,MAAM;AACvD;;;;;;;;;;;;;;;;;;;AAoBA,SAAgB,qBACd,WACA,UACA,aACA,oBACA,OAQI,CAAC,GACkB;CACvB,MAAM,MAA6B,CAAC;CACpC,KAAK,MAAM,OAAO,oBAAoB;EACpC,MAAM,SAAS,YAAY,WAAW,UAAU,cAAc,MAAM,EAAE,WAAW,IAAI;EACrF,IAAI,OAAO,OAAO,WAAW,GAAG;EAChC,MAAM,YAAY,KAAK,aAAa,MAAO,YAAY,CAAC,GAAG,OAAO,QAAQ,GAAG,OAAO,KAAK,CAAC;EAC1F,MAAM,qBAAqB,KAAK,aAAA;EAChC,MAAM,SAAS;GACb,YAAY,KAAK,cAAc;GAC/B,WAAW,KAAK,aAAa;GAC7B,WAAW;GACX,MAAM,KAAK,QAAQ;EACrB;EACA,MAAM,QAAQ,sBAAsB,OAAO,QAAQ,OAAO,OAAO,MAAM;EACvE,MAAM,aAAa,sBAAsB,OAAO,OAAO,OAAO,QAAQ;GACpE,GAAG;GACH,WAAW;EACb,CAAC;EACD,MAAM,YAAY,MAAM,aAAa,gBAAgB,OAAO,QAAQ,OAAO,OAAO,MAAM;EACxF,IAAI,KAAK;GACP,WAAW;GACX;GACA;GACA,IAAI;IAAE,KAAK,MAAM;IAAK,MAAM,MAAM;GAAK;GACvC,mBAAmB,MAAM;GACzB,SAAS,MAAM;GACf,eAAe,MAAM;GAgBrB,WAAW,UAAU,MAAM,CAAC,aAAa,WAAW;GACpD;GACA,GAAG,OAAO,OAAO;EACnB,CAAC;CACH;CACA,OAAO;AACT;;;;ACjUA,SAAS,KAAK,YAA4B;CACxC,IAAI,cAAc,KAAM,OAAO;CAC/B,IAAI,cAAc,KAAM,OAAO;CAC/B,IAAI,cAAc,IAAK,OAAO;CAC9B,OAAO;AACT;;;;;AAMA,SAAgB,eAAe,MAA6C;CAC1E,MAAM,aAAa,KAAK,mBAAmB,QAAQ,MAAM,OAAO,SAAS,CAAC,CAAC;CAC3E,IAAI,WAAW,SAAS,GACtB,MAAM,IAAI,MACR,kFAAkF,WAAW,QAC/F;CAEF,MAAM,iBAAiB,KAAK,kBAAkB;CAC9C,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,IAAI,KAAK,WAAW,WAAW;CACrC,IAAI,IAAI,GAAG,MAAM,IAAI,MAAM,6CAA6C,GAAG;CAE3E,MAAM,OAAO,WAAW,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,WAAW;CAChE,MAAM,WACJ,WAAW,QAAQ,GAAG,MAAM,KAAK,IAAI,SAAS,IAAI,OAAO,CAAC,KAAK,WAAW,SAAS;CACrF,MAAM,KAAK,KAAK,KAAK,QAAQ;CAC7B,MAAM,IAAI,KAAK,UAAU;CACzB,MAAM,MAAM,iBAAkB,IAAI,KAAK,QAAQ,KAAM,KAAK,KAAK,CAAC;CAEhE,MAAM,eAAe,WAAW,OAAO,MAAM,KAAK,SAAU,KAAK,GAAG;CACpE,MAAM,WAAW,KAAK,IAAI,GAAG,IAAI,IAAI;CACrC,MAAM,eAAe,gBAAgB,MAAM;CAE3C,MAAM,sBAAsB,KAAK,sBAC7B,8PACA,KAAA;CAEJ,MAAM,iBAAiB,eACnB,yCAAyC,IAAI,QAAQ,CAAC,EAAE,eAAe,SAAS,QAAQ,CAAC,EAAE,gCAAgC,KAAK,QAAQ,CAAC,EAAE,0FAA0F,KAAK,MAAO,IAAI,KAAK,QAAQ,KAAM,KAAK,IAAI,WAAW,gBAAgB,GAAI,MAAM,CAAC,EAAE,gDACzT,gCAAgC,EAAE,IAAI,IAAI,QAAQ,CAAC,EAAE,gBAAgB,GAAG,QAAQ,CAAC,EAAE;CAEvF,OAAO;EACL;EACA;EACA;EACA,cAAc;EACd;EACA;EACA;EACA;EACA;EACA,GAAI,sBAAsB,EAAE,oBAAoB,IAAI,CAAC;EACrD,gBAAgB,sBACZ,GAAG,eAAe,GAAG,wBACrB;CACN;AACF"}
1
+ {"version":3,"file":"power-preflight-CFXm0Vjo.js","names":[],"sources":["../src/paired-delta-test.ts","../src/paired-promotion-decision.ts","../src/campaign/gates/statistical-heldout.ts","../src/campaign/gates/power-preflight.ts"],"sourcesContent":["import {\n type PairedBootstrapOptions,\n type PairedBootstrapResult,\n pairedBootstrap,\n pairedSignTest,\n} from './statistics'\n\nexport interface PairedDeltaTestOptions extends PairedBootstrapOptions {\n /** Smallest candidate-minus-baseline delta that counts as improvement. Default 0. */\n threshold?: number\n /** Caller-required paired observations. The exact test may impose a higher minimum. */\n minPairs?: number\n}\n\nexport interface PairedDeltaTestResult {\n bootstrap: PairedBootstrapResult\n method: 'bootstrap-ci' | 'exact-sign'\n /** Exact one-sided p-value below the bootstrap minimum; otherwise null. */\n pValue: number | null\n /** Effective observation minimum after accounting for confidence. */\n minimumPairs: number\n sufficient: boolean\n /**\n * The bootstrap interval has zero width (or is non-finite), so it carries no\n * information about how far the estimate could be wrong and cannot support a\n * decision in either direction. See {@link pairedDeltaTest}.\n */\n indeterminate: boolean\n significant: boolean\n}\n\n/** Smallest all-positive sample that can clear a one-sided exact sign test. */\nexport function minimumPairsForPairedDeltaTest(confidence = 0.95): number {\n if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) {\n throw new Error(\n `minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`,\n )\n }\n const oneSidedAlpha = (1 - confidence) / 2\n return Math.ceil(Math.log2(1 / oneSidedAlpha))\n}\n\n/**\n * Tests whether a paired candidate-minus-baseline delta clears a threshold.\n *\n * At 20 or more pairs, the percentile bootstrap lower bound carries the\n * decision. Below that point the interval is descriptive only, so the function\n * switches to a pre-registered one-sided exact sign test. The exact path is\n * deliberately conservative: it requires both a point estimate above the\n * threshold and enough consistently positive paired differences.\n *\n * ## A zero-width interval is never significant\n *\n * When every paired delta is identical the resample distribution is a point\n * mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n\n * identical deltas of g. Neither says the effect is certain — both say the\n * sample carries no information about how far the estimate could be wrong, and\n * `low > threshold` then answers on the point estimate alone. It fails in both\n * directions: `[0, 0]` clears every NEGATIVE threshold, which is how a\n * tie-dominated pass/fail comparison laundered a regression into a\n * noninferiority pass, and `[g, g]` clears every threshold below g with no\n * spread behind it. Under a bounded asymmetric null whose true mean paired\n * delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —\n * every sample that misses the drop is exactly that shape, and deciding on\n * `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.\n *\n * So `indeterminate` is reported and `significant` is false whenever the\n * interval has zero width, on BOTH paths: at small n the exact sign test is a\n * test of the MEDIAN and a zero-spread sample is precisely where it stops\n * saying anything about the mean the caller is thresholding.\n *\n * `threshold` may be negative — that is a noninferiority margin, and it is the\n * regime the zero-width hole is worst in. For a two-point (pass/fail) outcome\n * the percentile bootstrap is not a valid interval at a nonzero margin at all;\n * use {@link decidePairedPromotion}, which routes those to Tango's score\n * interval, rather than thresholding this function's bootstrap directly.\n */\nexport function pairedDeltaTest(\n before: number[],\n after: number[],\n options: PairedDeltaTestOptions = {},\n): PairedDeltaTestResult {\n const threshold = options.threshold ?? 0\n if (!Number.isFinite(threshold)) {\n throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`)\n }\n const confidence = options.confidence ?? 0.95\n const exactMinimum = minimumPairsForPairedDeltaTest(confidence)\n const requestedMinimum = options.minPairs ?? exactMinimum\n if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) {\n throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`)\n }\n const minimumPairs = Math.max(requestedMinimum, exactMinimum)\n const bootstrap = pairedBootstrap(before, after, options)\n const sufficient = bootstrap.n >= minimumPairs\n // A point-mass resample distribution is an absence of evidence, not a\n // certainty, and it is the shape that clears every negative threshold. Both\n // paths refuse it: see the \"zero-width\" section of this function's docs.\n const indeterminate =\n bootstrap.n > 0 &&\n (!Number.isFinite(bootstrap.low) ||\n !Number.isFinite(bootstrap.high) ||\n bootstrap.low === bootstrap.high)\n\n if (bootstrap.gateEligible) {\n return {\n bootstrap,\n method: 'bootstrap-ci',\n pValue: null,\n minimumPairs,\n sufficient,\n indeterminate,\n significant: sufficient && !indeterminate && bootstrap.low > threshold,\n }\n }\n\n const differences = before.map((value, index) => after[index]! - value - threshold)\n const exact = pairedSignTest(differences, 'greater')\n const estimate = options.statistic === 'mean' ? bootstrap.mean : bootstrap.median\n return {\n bootstrap,\n method: 'exact-sign',\n pValue: exact.pValue,\n minimumPairs,\n sufficient,\n indeterminate,\n significant:\n sufficient &&\n !indeterminate &&\n estimate > threshold &&\n exact.pValue <= (1 - bootstrap.confidence) / 2,\n }\n}\n","/**\n * @module\n * ONE rule for \"does this paired interval clear a promotion threshold\".\n *\n * The rule below was derived on `HeldOutGate` (#479) after the same estimator\n * bug shipped twice. It then turned out that a SECOND gate — the composable\n * `heldOutGate`, plus everything else routed through `heldoutSignificance` —\n * still carried the original defect, because the rule had been written into one\n * gate's method body rather than into a shared function. Two copies of a\n * statistical rule is how a defect survives in one of them, so there is now\n * exactly one copy and both gates call it.\n *\n * Three things the rule does that a bare `pairedBootstrap(...).low > threshold`\n * does not:\n *\n * 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a\n * pass/fail eval the paired delta vector is dominated by ties, so the\n * bootstrap of the mean is a resample of a lattice with three atoms and its\n * percentile interval is not valid at a nonzero margin. The score interval\n * (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under\n * each hypothesised margin instead of fixing it at the observed value, which\n * is the only construction that stays a confidence interval as the margin\n * moves off zero — the regime every noninferiority threshold lives in.\n * Measured on the composable gate before this change, at a true risk\n * difference sitting exactly on the production caller's -0.05 margin and a\n * nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.\n * 2. **McNemar's exact test holds a VETO at every non-negative threshold.**\n * Redundant with the interval by construction and kept anyway, so that\n * swapping the estimator for one without that duality cannot silently\n * reintroduce \"promotes what the exact test refuses\". Witness: n = 6, b = 5,\n * c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs\n * (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE\n * threshold is a noninferiority question, which McNemar's test of \"no\n * difference\" is not the right test for, so the veto does not apply there.\n * 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it\n * cannot tell a gain from a regression and clears every negative threshold.\n * Away from zero it fails the opposite way: n identical positive deltas give\n * [g, g], which clears threshold 0 on no spread at all. Both are an absence\n * of evidence. Measured on the composable gate before this change, under a\n * bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %\n * false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.\n *\n * Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that\n * picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered\n * exact sign test below it), this picks the ESTIMATOR from the outcome's shape.\n * Both are needed — an exact sign test applied to a tie-pinned median is still\n * blind, and a mean bootstrap CI at n = 6 is still not a valid test.\n */\n\nimport { minimumPairsForPairedDeltaTest, pairedDeltaTest } from './paired-delta-test'\nimport {\n type PairedBootstrapResult,\n pairedBinaryScale,\n pairedDeltaTieFraction,\n pairedRiskDifferenceExact,\n pairedRiskDifferenceScore,\n} from './statistics'\n\n/** Which paired estimator produced the deciding interval. */\nexport type PairedDecisionStatistic =\n | 'paired_risk_difference'\n | 'mean_bootstrap'\n | 'median_bootstrap'\n\n/** Which test carried the decision, given the estimator and the sample size. */\nexport type PairedDecisionMethod = 'score-interval' | 'bootstrap-ci' | 'exact-sign'\n\n/** McNemar's exact paired-binary evidence, on the two-point path only. */\nexport interface PairedMcNemarEvidence {\n /** Discordant pairs the treatment won. */\n b: number\n /** Discordant pairs the control won. */\n c: number\n /** b + c — the only pairs carrying information. */\n nDiscordant: number\n /** Two-sided exact p-value. */\n pValue: number\n}\n\nexport interface PairedPromotionDecisionOptions {\n /** Smallest candidate-minus-baseline delta that counts as improvement, in the\n * caller's native units. May be negative (a noninferiority margin). Default 0. */\n threshold?: number\n /** Confidence level. Default 0.95. */\n confidence?: number\n /** Bootstrap resamples, on the paths where a bootstrap decides. Default 2000. */\n resamples?: number\n /** Deterministic bootstrap seed. Omitted ⇒ derived from the deltas. */\n seed?: number\n /** Caller-required paired observations. The exact test may impose a higher\n * minimum; the effective one is reported as `minimumPairs`. */\n minPairs?: number\n /**\n * `'mean'` (default) routes by SHAPE: a two-point (pass/fail) outcome on any\n * encoding decides on the score interval, everything else on the mean\n * bootstrap. `'median'` forces the median bootstrap on every input, including\n * shapes where it is structurally blind — kept for callers who want outlier\n * robustness on genuinely continuous outcomes and accept that cost.\n */\n statistic?: 'mean' | 'median'\n}\n\nexport interface PairedPromotionDecision {\n /** Paired observations supplied. */\n n: number\n /** Threshold the interval was judged against, native units. */\n threshold: number\n confidence: number\n statistic: PairedDecisionStatistic\n method: PairedDecisionMethod\n /** Common positive level of a two-point outcome ({0,1} ⇒ 1, {0,100} ⇒ 100),\n * or null when the outcome is not two-point. Non-null is exactly the\n * condition for the `paired_risk_difference` path, and it is the factor\n * `delta` / `low` / `high` were rescaled by. */\n binaryScale: number | null\n /** Exact-tie fraction over the paired deltas; null when there are no pairs. */\n tieFraction: number | null\n /** Point estimate of the DECIDING statistic, in the caller's native units. */\n delta: number\n /** Lower bound of the DECIDING interval, native units. */\n low: number\n /** Upper bound of the DECIDING interval, native units. */\n high: number\n /** The bootstrap that decided, or null when the score interval did. Callers\n * that need a bootstrap as a diagnostic on the two-point path compute their\n * own — it is not computed here, so the binary path costs no resamples. */\n bootstrap: PairedBootstrapResult | null\n /** McNemar's exact evidence, or null off the two-point path. */\n mcnemar: PairedMcNemarEvidence | null\n /** Exact one-sided sign-test p-value on the small-sample bootstrap path;\n * null otherwise. */\n pValue: number | null\n /** Effective observation minimum after accounting for confidence. */\n minimumPairs: number\n /** n >= minimumPairs. */\n sufficient: boolean\n /** The deciding interval is zero-width or non-finite — no evidence in either\n * direction, so it cannot clear any threshold on evidence. */\n indeterminate: boolean\n /** McNemar's exact test refuses at a non-negative threshold. */\n exactTestVetoes: boolean\n /** The deciding interval clears the threshold, ignoring the other two guards. */\n clearsThreshold: boolean\n /** `sufficient && !indeterminate && clearsThreshold && !exactTestVetoes` —\n * the whole rule. */\n promote: boolean\n /** What `delta` measures, for a reason string. */\n label: 'success-rate' | 'mean' | 'median'\n /** Why a zero-width interval is zero-width; empty when it is not. */\n indeterminateCause: string\n /** Sentence naming the test when the exact sign test decided; else empty. */\n methodDetail: string\n}\n\n/** The shape facts that pick the estimator, without computing an interval. */\nexport interface PairedDecisionShape {\n statistic: PairedDecisionStatistic\n /** Common positive level of a two-point outcome; null when not two-point. */\n binaryScale: number | null\n /** Exact-tie fraction over the paired deltas; null when there are no pairs. */\n tieFraction: number | null\n}\n\n/**\n * Which estimator {@link decidePairedPromotion} would use on this data, and the\n * shape facts behind it — for callers that must report the shape on a path\n * where no interval is computed at all (an early rejection, or zero pairs).\n * Cheap: no bootstrap, no interval.\n */\nexport function pairedDecisionShape(\n before: number[],\n after: number[],\n statistic: 'mean' | 'median' = 'mean',\n): PairedDecisionShape {\n const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after)\n if (statistic === 'median') {\n return { statistic: 'median_bootstrap', binaryScale: null, tieFraction }\n }\n const binaryScale = pairedBinaryScale(before, after)\n if (binaryScale !== null) {\n return { statistic: 'paired_risk_difference', binaryScale, tieFraction }\n }\n return { statistic: 'mean_bootstrap', binaryScale: null, tieFraction }\n}\n\n/**\n * Decide whether a paired candidate-minus-baseline delta clears a promotion\n * threshold. `before` is the baseline arm, `after` the candidate arm, paired by\n * position. Throws on unequal lengths.\n */\nexport function decidePairedPromotion(\n before: number[],\n after: number[],\n options: PairedPromotionDecisionOptions = {},\n): PairedPromotionDecision {\n if (before.length !== after.length) {\n throw new Error(\n `decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`,\n )\n }\n const threshold = options.threshold ?? 0\n if (!Number.isFinite(threshold)) {\n throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`)\n }\n const confidence = options.confidence ?? 0.95\n const exactMinimum = minimumPairsForPairedDeltaTest(confidence)\n const requestedMinimum = options.minPairs ?? exactMinimum\n if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) {\n throw new Error(\n `decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`,\n )\n }\n const minimumPairs = Math.max(requestedMinimum, exactMinimum)\n const n = before.length\n const sufficient = n >= minimumPairs\n const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic)\n\n let core: {\n statistic: PairedDecisionStatistic\n method: PairedDecisionMethod\n delta: number\n low: number\n high: number\n bootstrap: PairedBootstrapResult | null\n mcnemar: PairedMcNemarEvidence | null\n pValue: number | null\n clearsThreshold: boolean\n label: PairedPromotionDecision['label']\n methodDetail: string\n }\n\n if (binaryScale !== null) {\n // Normalise the two-point encoding to {0,1} so the estimators see the\n // pass/fail structure, then rescale the answer back into the caller's\n // native units — the threshold is read in the units of the scores, so a\n // 0-100 pass/fail dimension must be gated in points, not in a rate.\n const unitControl = before.map((v) => v / binaryScale)\n const unitTreatment = after.map((v) => v / binaryScale)\n // TWO estimators with different jobs, because no single one does both.\n // `exact` is the authority on RD = 0 and supplies the veto; its\n // Clopper-Pearson interval conditions on the discordant count and is\n // therefore NOT a confidence interval at a nonzero margin. `score` is\n // Tango's, which is.\n const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence)\n const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence)\n const low = score.lower * binaryScale\n core = {\n statistic: 'paired_risk_difference',\n method: 'score-interval',\n delta: score.riskDifference * binaryScale,\n low,\n high: score.upper * binaryScale,\n bootstrap: null,\n mcnemar: {\n b: exact.b,\n c: exact.c,\n nDiscordant: exact.nDiscordant,\n pValue: exact.pValue,\n },\n pValue: null,\n clearsThreshold: low > threshold,\n label: 'success-rate',\n methodDetail: '',\n }\n } else {\n const bootstrapStatistic = options.statistic === 'median' ? 'median' : 'mean'\n const test = pairedDeltaTest(before, after, {\n confidence,\n resamples: options.resamples,\n statistic: bootstrapStatistic,\n seed: options.seed,\n threshold,\n minPairs: options.minPairs,\n })\n const ci = test.bootstrap\n core = {\n statistic: bootstrapStatistic === 'mean' ? 'mean_bootstrap' : 'median_bootstrap',\n method: test.method,\n delta: bootstrapStatistic === 'mean' ? ci.mean : ci.median,\n low: ci.low,\n high: ci.high,\n bootstrap: ci,\n mcnemar: null,\n pValue: test.pValue,\n clearsThreshold: test.significant,\n label: bootstrapStatistic,\n methodDetail:\n test.method === 'exact-sign'\n ? ` Below ${test.minimumPairs} pairs the interval is descriptive only;` +\n ` the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.`\n : '',\n }\n }\n\n const indeterminate =\n !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.low === core.high\n const indeterminateCause = !indeterminate\n ? ''\n : tieFraction === 1\n ? 'every paired delta is an exact tie'\n : core.mcnemar !== null && core.mcnemar.nDiscordant === 0\n ? 'every pair is concordant (0 discordant pairs)'\n : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`\n // Only at a non-negative threshold: a negative threshold asks a\n // noninferiority question, which McNemar's test of \"no difference\" does not\n // answer.\n const exactTestVetoes =\n core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence)\n\n return {\n n,\n threshold,\n confidence,\n binaryScale,\n tieFraction,\n minimumPairs,\n sufficient,\n indeterminate,\n indeterminateCause,\n exactTestVetoes,\n promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,\n ...core,\n }\n}\n\nfunction fmt(x: number): string {\n return x.toFixed(4)\n}\n","/**\n * Statistical held-out promotion machinery — the trustworthy core the\n * point-estimate `heldout-delta` gate lacked.\n *\n * The shipped false positive it prevents: a winner re-scored against the\n * baseline on the holdout read run-to-run model NOISE (e.g. 91 vs 95) as a\n * \"+4 lift\" and shipped, because the gate compared point estimates with no\n * confidence interval. Here we pair candidate vs baseline holdout observations\n * and bootstrap a CI on the paired delta — a candidate ships only when the CI\n * lower bound clears the effect-size threshold (the gain is real at the\n * confidence level, not noise), and is blocked when a critical dimension\n * (e.g. `hallucination_free` for a legal agent) significantly regresses even if\n * the net composite rose (anti-Goodhart).\n *\n * Two traps this module is built around (both produce a NEW false positive if\n * gotten wrong):\n * 1. PAIRING GRANULARITY — pairs by FULL `cellId` (`scenario:rep`), never by\n * `scenarioId` (which averages reps away and destroys the within-pair\n * variance reduction that makes a paired bootstrap tighter than unpaired).\n * One paired observation per cell ⇒ reps multiply n.\n * 2. SCALE — a judge may emit composites/dimensions on [0,1] or 0-100. The\n * threshold + tolerance are interpreted in the judge's NATIVE scale; the\n * per-dimension tolerance auto-scales off the observed baseline magnitudes\n * so `-0.10` on [0,1] doesn't silently become a no-op on a 0-100 dimension.\n */\n\nimport {\n decidePairedPromotion,\n type PairedDecisionMethod,\n type PairedDecisionStatistic,\n type PairedMcNemarEvidence,\n type PairedPromotionDecision,\n} from '../../paired-promotion-decision'\nimport {\n DECISION_PAIRED_DELTA_STATISTIC,\n type PairedBootstrapResult,\n pairedBootstrap,\n} from '../../statistics'\nimport type { JudgeScore } from '../types'\n\n/** Tie fraction at/above which a gate annotates its verdict with the tie share.\n * Tie-domination of the median bites structurally at >= 0.5 (the median is then\n * 0 by construction); 0.4 is a softer warn threshold that flags a run APPROACHING\n * that regime, so an operator sees it before the median goes fully blind. */\nexport const TIE_WARN_FRACTION = 0.4\n\nexport interface PairedHoldout {\n /** Baseline scalar per paired cell (same order as `after`/`cellIds`). */\n before: number[]\n /** Candidate scalar per paired cell. */\n after: number[]\n /** The full cellIds (`scenario:rep`) that paired, in order. */\n cellIds: string[]\n}\n\n/**\n * Pair candidate vs baseline holdout observations by FULL cellId. `select`\n * pulls the scalar from a cell's judge reports (composite, or a named\n * dimension); a cell contributes the mean of `select` across its judges. Cells\n * whose scenario is not in `scenarioIds`, or where `select` is undefined for\n * every judge on either side, are skipped on BOTH sides so the arrays stay\n * paired. Throws when the two maps disagree on which holdout cells exist — a\n * load-bearing invariant: the baseline + winner holdout campaigns run the same\n * scenarios with the same seed base, so their cellIds MUST align; a mismatch\n * means a silent pairing bug, not a soft fallback.\n */\nexport function pairHoldout(\n candidate: Map<string, Record<string, JudgeScore>>,\n baseline: Map<string, Record<string, JudgeScore>>,\n scenarioIds: Set<string>,\n select: (s: JudgeScore) => number | undefined,\n): PairedHoldout {\n const cellValue = (\n byCell: Map<string, Record<string, JudgeScore>>,\n cellId: string,\n ): number | undefined => {\n const scores = byCell.get(cellId)\n if (!scores) return undefined\n const vals: number[] = []\n for (const s of Object.values(scores)) {\n if (s.failed === true) {\n throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`)\n }\n const v = select(s)\n if (typeof v === 'number' && !Number.isFinite(v)) {\n throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`)\n }\n if (typeof v === 'number') vals.push(v)\n }\n if (vals.length === 0) return undefined\n return vals.reduce((a, b) => a + b, 0) / vals.length\n }\n\n const inScope = (cellId: string) => scenarioIds.has(cellId.split(':')[0] ?? '')\n const candCells = [...candidate.keys()].filter(inScope).sort()\n const baseCells = [...baseline.keys()].filter(inScope).sort()\n // Alignment invariant — the holdout campaigns share scenarios + seed, so the\n // cell sets must be identical. Differ ⇒ a real pairing bug; fail loud.\n if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) {\n throw new Error(\n `pairHoldout: candidate/baseline holdout cells do not align — ` +\n `candidate=[${candCells.join(',')}] baseline=[${baseCells.join(',')}]. ` +\n `Both holdout campaigns must run the same scenarios with the same seed base.`,\n )\n }\n\n const before: number[] = []\n const after: number[] = []\n const cellIds: string[] = []\n for (const cellId of candCells) {\n const b = cellValue(baseline, cellId)\n const a = cellValue(candidate, cellId)\n // A scalar absent on both sides means that dimension was not scored. A\n // one-sided absence is asymmetric evidence loss, never a row to discard.\n if (b === undefined && a === undefined) continue\n if (b === undefined || a === undefined) {\n throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`)\n }\n before.push(b)\n after.push(a)\n cellIds.push(cellId)\n }\n return { before, after, cellIds }\n}\n\nexport interface HeldoutSignificance {\n paired: PairedHoldout\n /**\n * The paired bootstrap on the requested statistic (MEAN by default — see the\n * tie note on `heldoutSignificance`).\n *\n * DIAGNOSTIC, not necessarily the interval the verdict keyed on. On a\n * two-point (pass/fail) outcome the decision routes to Tango's score interval\n * instead, because a percentile bootstrap of the mean over a three-atom\n * lattice is not a valid interval at a nonzero margin. Read\n * `decision.low`/`decision.high` for the interval that actually decided, and\n * `decisionStatistic` for which one it is.\n */\n bootstrap: PairedBootstrapResult\n /** The MEDIAN paired-delta bootstrap, reported as a diagnostic. When many\n * scenarios are tied (both sides solve them), the median is pinned near 0\n * regardless of the mean lift — comparing the two exposes tie-domination. */\n medianBootstrap: PairedBootstrapResult\n /**\n * The full promotion decision: which estimator the outcome's shape admits,\n * the interval it produced, McNemar's exact veto on the two-point path, and\n * whether the interval was zero-width (no evidence in either direction). The\n * single source of `significant`.\n */\n decision: PairedPromotionDecision\n /** Which paired estimator the verdict was decided on. */\n decisionStatistic: PairedDecisionStatistic\n /** McNemar's exact evidence on the two-point path; null otherwise. */\n mcnemar: PairedMcNemarEvidence | null\n /** Fraction of paired observations that are exact ties (|delta| < 1e-9). A\n * high tie fraction is WHY a median-based gate would have missed a real lift;\n * it is the observability the tie fix adds. */\n tieFraction: number\n /** n paired observations. */\n n: number\n /** Effective minimum after applying the bootstrap's hard statistical floor. */\n minimumRequired: number\n /** Statistical method that carried the decision. */\n decisionMethod: PairedDecisionMethod\n /** Exact one-sided p-value on the small-sample path; otherwise null. */\n pValue: number | null\n /** True iff n >= minimumRequired, the DECIDING interval has nonzero width,\n * its lower bound clears the threshold, and McNemar's exact test does not\n * veto at a non-negative threshold. */\n significant: boolean\n /** Set when n < minimumRequired — too little evidence to claim significance. */\n fewRuns: boolean\n}\n\nexport interface HeldoutSignificanceOptions {\n deltaThreshold?: number\n minProductiveRuns?: number\n confidence?: number\n resamples?: number\n /** Fixed by default for a deterministic, reproducible gate verdict. */\n seed?: number\n statistic?: 'mean' | 'median'\n}\n\n/**\n * Significance of the held-out composite lift: ship only when the lower bound\n * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default\n * 0 ⇒ \"confidently positive\"). Interpret `deltaThreshold` in the judge's native\n * scale.\n *\n * The decision is delegated whole to {@link decidePairedPromotion}, the one\n * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`\n * also calls. That module's header carries the measurements; the short version\n * is three guards a bare `bootstrap.low > threshold` does not have:\n *\n * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the\n * only paired-binary construction that stays valid at a nonzero margin;\n * - McNemar's exact test VETOES at any non-negative threshold;\n * - a ZERO-WIDTH interval is refused rather than promoted, in either\n * direction — [0,0] clears every negative threshold and [g,g] clears every\n * threshold below g, and both are an absence of evidence, not a result.\n *\n * Measured on this function before those guards landed, at a nominal 5 %:\n * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,\n * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired\n * delta is exactly 0.\n *\n * At small n, where the percentile bootstrap is descriptive only, a\n * pre-registered exact sign test still carries the bootstrap path.\n */\nexport function heldoutSignificance(\n paired: PairedHoldout,\n opts: HeldoutSignificanceOptions = {},\n): HeldoutSignificance {\n const deltaThreshold = opts.deltaThreshold ?? 0\n const confidence = opts.confidence ?? 0.95\n const resamples = opts.resamples ?? 2000\n const seed = opts.seed ?? 1337\n // DEFAULT to the MEAN paired delta, not the median. The median is destroyed by\n // TIES: whenever both baseline and candidate solve a holdout scenario (a common\n // case once the agent is decent — and INCREASINGLY common as you add holdout\n // scenarios for statistical power), that scenario contributes delta 0. When\n // >=50% of paired cells are ties the median is pinned at 0 regardless of a large,\n // consistent lift on the rest, and the gate holds a genuinely better candidate.\n // (Measured live, supervisor-lab run 6: 40 cells, 20 ties, MEAN +0.177, MEDIAN 0\n // → false hold. Doubling the holdout for \"power\" made it WORSE by adding ties.)\n // The mean equals the reported aggregate lift, ties correctly contribute 0\n // without dominating, and it is the textbook paired-comparison estimator; the\n // median is kept as a reported diagnostic. Callers wanting outlier-robustness at\n // the cost of tie-blindness can still pass `statistic: 'median'`.\n const statistic = opts.statistic ?? 'mean'\n const decision = decidePairedPromotion(paired.before, paired.after, {\n confidence,\n resamples,\n statistic,\n seed,\n threshold: deltaThreshold,\n minPairs: opts.minProductiveRuns,\n })\n // `decision.bootstrap` is null exactly when the score interval decided, so\n // the requested-statistic bootstrap is computed here for the diagnostic\n // field. Same two bootstraps as before on every path.\n const bootstrap =\n decision.bootstrap ??\n pairedBootstrap(paired.before, paired.after, { confidence, resamples, statistic, seed })\n const medianBootstrap =\n statistic === 'median'\n ? bootstrap\n : pairedBootstrap(paired.before, paired.after, {\n confidence,\n resamples,\n statistic: 'median',\n seed,\n })\n const n = paired.before.length\n let ties = 0\n for (let i = 0; i < n; i += 1) {\n const after = paired.after[i] ?? 0\n const before = paired.before[i] ?? 0\n if (Math.abs(after - before) < 1e-9) ties += 1\n }\n const tieFraction = n === 0 ? 0 : ties / n\n return {\n paired,\n bootstrap,\n medianBootstrap,\n decision,\n decisionStatistic: decision.statistic,\n mcnemar: decision.mcnemar,\n tieFraction,\n n,\n minimumRequired: decision.minimumPairs,\n decisionMethod: decision.method,\n pValue: decision.pValue,\n significant: decision.promote,\n fewRuns: !decision.sufficient,\n }\n}\n\nexport interface DimensionRegression {\n dimension: string\n /** Paired bootstrap on (candidate − baseline). DIAGNOSTIC on a pass/fail\n * dimension, where `ci` carries the interval that decided instead. */\n bootstrap: PairedBootstrapResult\n /** Which paired statistic `bootstrap.low` is the lower bound of. `'mean'`\n * unless the caller asked for the median. `bootstrap.median` still carries\n * the median point estimate either way. */\n bootstrapStatistic: 'median' | 'mean'\n /** The interval `regressed` was decided on, in the dimension's native units. */\n ci: { low: number; high: number }\n /** Which estimator produced `ci`. */\n decisionStatistic: PairedDecisionStatistic\n /** McNemar's exact evidence on a pass/fail dimension; null otherwise. */\n mcnemar: PairedMcNemarEvidence | null\n /** `ci` has zero width — no evidence in either direction. */\n indeterminate: boolean\n /** True iff the candidate may have regressed this dimension by more than\n * tolerance: the lower bound of the DECIDING interval on (candidate −\n * baseline) is below −tolerance, OR the exact small-sample test proves a drop\n * past tolerance. */\n regressed: boolean\n tolerance: number\n n: number\n}\n\n/** Detect the native scale of a set of scores: 0-100 when any magnitude clears\n * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default\n * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */\nexport function detectScale(values: number[]): 1 | 100 {\n return values.some((v) => Math.abs(v) > 1.5) ? 100 : 1\n}\n\n/** Per-critical-dimension regression guard. For each dimension, pair the\n * candidate vs baseline values by full cellId and bootstrap the paired delta;\n * a dimension is \"regressed\" when the CI lower bound < −tolerance (conservative\n * — blocks if the credible worst case exceeds tolerance, which is the right\n * posture for safety dimensions like `hallucination_free`). When `tolerance`\n * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.\n *\n * The interval comes from {@link decidePairedPromotion}, so a pass/fail\n * dimension is judged on Tango's score interval rather than a percentile\n * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap\n * is not a valid interval at one. That matters most here because this guard\n * fails OPEN by construction: `tolerance` is positive, so an interval pinned at\n * [0,0] never satisfies `low < −tolerance` and a real regression on a safety\n * dimension would be reported as `regressed: false`. On the median it fails the\n * same way for the same reason — when most pairs tie, which is automatic for a\n * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists\n * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to\n * restore the pre-0.134 behaviour. */\nexport function dimensionRegressions(\n candidate: Map<string, Record<string, JudgeScore>>,\n baseline: Map<string, Record<string, JudgeScore>>,\n scenarioIds: Set<string>,\n criticalDimensions: string[],\n opts: {\n tolerance?: number\n confidence?: number\n resamples?: number\n seed?: number\n /** Paired statistic the CI is computed on. Default `'mean'` — see\n * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */\n statistic?: 'mean' | 'median'\n } = {},\n): DimensionRegression[] {\n const out: DimensionRegression[] = []\n for (const dim of criticalDimensions) {\n const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim])\n if (paired.before.length === 0) continue // dimension not scored on this judge\n const tolerance = opts.tolerance ?? 0.05 * detectScale([...paired.before, ...paired.after])\n const bootstrapStatistic = opts.statistic ?? DECISION_PAIRED_DELTA_STATISTIC\n const shared = {\n confidence: opts.confidence ?? 0.95,\n resamples: opts.resamples ?? 2000,\n statistic: bootstrapStatistic,\n seed: opts.seed ?? 1337,\n }\n const guard = decidePairedPromotion(paired.before, paired.after, shared)\n const regression = decidePairedPromotion(paired.after, paired.before, {\n ...shared,\n threshold: tolerance,\n })\n const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared)\n out.push({\n dimension: dim,\n bootstrap,\n bootstrapStatistic,\n ci: { low: guard.low, high: guard.high },\n decisionStatistic: guard.statistic,\n mcnemar: guard.mcnemar,\n indeterminate: guard.indeterminate,\n // Fires on EITHER burden of proof — see `floorBreached` in\n // `promotion-policy.ts` for the same rule and the reasoning. The CI arm\n // is the contract this interface and `default-production-gate`'s reason\n // string both state (\"CI.low < -tolerance\"), read off the DECIDING\n // interval; the exact-test arm adds the small-sample path where the\n // bootstrap interval is descriptive only.\n // The credible-worst-case arm stays on the BOOTSTRAP — see the same\n // decision in `floorBreached` (`promotion-policy.ts`). On a pass/fail\n // dimension the score interval is ±z²/(n+z²) even when every pair is\n // concordant, so reading the floor off it would flag an unchanged safety\n // dimension as regressed at any realistic n. The PROVEN-drop arm does use\n // the shared rule, which is what closes the fail-open hole that mattered:\n // a genuine pass/fail regression is now judged on an interval valid at\n // the nonzero `tolerance`, instead of a bootstrap that is pinned wherever\n // ties dominate.\n regressed: bootstrap.low < -tolerance || regression.promote,\n tolerance,\n n: paired.before.length,\n })\n }\n return out\n}\n","/**\n * Power preflight — \"can this budget detect the effect you are hunting?\"\n *\n * The failure it prevents (measured, twice): a live prompt-improvement campaign ran\n * 333 sandbox cells over 5.6 hours and produced a +0.08 holdout lift the ship gate\n * (paired bootstrap, CI.low > 0.05) could not distinguish from zero — because at\n * that holdout size and worker variance the MINIMUM DETECTABLE lift was larger than\n * any effect a prompt change plausibly produces. The budget was spent learning what\n * a 30-second calculation on the baseline cells already knew. No eval framework we\n * know of surfaces this; every underpowered improvement run everywhere ends in an\n * uninformative \"hold\".\n *\n * Model: the ship rule is `CI.low(paired Δ) > deltaThreshold`. Approximating the\n * bootstrap CI as normal, `CI.low ≈ effect − z·sd_Δ/√n`, so the smallest shippable\n * true effect is `MDE = deltaThreshold + z·sd_Δ/√n`. The paired-delta SD is unknown\n * before the candidate exists; we bound it by the zero-correlation case\n * `sd_Δ ≤ √2·sd_baseline` — a CONSERVATIVE (upper) MDE, which is the correct\n * direction for a warning. Pairing is per cell (`scenario:rep`), so reps multiply n.\n *\n * Standalone by design: feed it any baseline composites (a `gate:'none'` run, a\n * live-proof table) BEFORE budgeting the real search; `selfImprove` also attaches\n * it to every result and warns when the run was structurally unable to ship.\n */\n\nexport interface PowerPreflightOptions {\n /** Per-cell baseline composites on the HOLDOUT scenarios (one per scenario:rep cell). */\n baselineComposites: number[]\n /** Paired observations the budgeted comparison will produce\n * (holdout scenarios × reps). Defaults to `baselineComposites.length`. */\n pairedN?: number\n /** The ship gate's effect-size threshold. Default 0.05 (defaultProductionGate). */\n deltaThreshold?: number\n /** CI confidence the gate uses. Default 0.95. */\n confidence?: number\n /** True when the holdout is scored by the SAME judge/scorer family as the gate\n * (selfImprove's default composition — one judge scores everything). Under a\n * shared channel, raising paired n reduces only the IDIOSYNCRATIC noise share;\n * systematic judge bias is untouched, so the MDE here is a lower bound and the\n * only full debiaser is an independent second scoring channel\n * (recursive-self-improvement S1c, closed form in EXP-023 P0). Default false. */\n sharedScorerChannel?: boolean\n}\n\nexport interface PowerPreflight {\n /** Paired observations the comparison will have. */\n n: number\n /** Baseline per-cell composite standard deviation (the variance the effect must beat). */\n sd: number\n /** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */\n mde: number\n /** Baseline holdout composite mean. */\n baselineMean: number\n /** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */\n headroom: number\n /** True when even the largest achievable effect (headroom) is below the MDE —\n * the run is structurally unable to ship regardless of proposal quality.\n * Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */\n underpowered: boolean\n /** True when composites look [0,1]-scaled; headroom/underpowered are only\n * meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */\n scaleAssumed: boolean\n deltaThreshold: number\n confidence: number\n /** Set when the holdout shares the gate's scoring channel: more cells cannot\n * buy back systematic judge bias — treat the MDE as a lower bound. */\n sharedChannelCaveat?: string\n /** One actionable sentence for humans and logs. */\n recommendation: string\n}\n\n/** Two-sided z for the common confidence levels; interpolation is overkill here. */\nfunction zFor(confidence: number): number {\n if (confidence >= 0.99) return 2.576\n if (confidence >= 0.95) return 1.96\n if (confidence >= 0.9) return 1.645\n return 1.282\n}\n\n/** Estimate the minimum detectable lift a paired-holdout improvement run can\n * ship at a given budget, from the baseline holdout composites — call it BEFORE\n * spending a search to learn whether the effect you are hunting is even\n * observable at this holdout size and worker variance. */\nexport function powerPreflight(opts: PowerPreflightOptions): PowerPreflight {\n const composites = opts.baselineComposites.filter((v) => Number.isFinite(v))\n if (composites.length < 3) {\n throw new Error(\n `powerPreflight: need >= 3 finite baseline composites to estimate variance, got ${composites.length}`,\n )\n }\n const deltaThreshold = opts.deltaThreshold ?? 0.05\n const confidence = opts.confidence ?? 0.95\n const n = opts.pairedN ?? composites.length\n if (n < 2) throw new Error(`powerPreflight: pairedN must be >= 2, got ${n}`)\n\n const mean = composites.reduce((a, b) => a + b, 0) / composites.length\n const variance =\n composites.reduce((a, b) => a + (b - mean) * (b - mean), 0) / (composites.length - 1)\n const sd = Math.sqrt(variance)\n const z = zFor(confidence)\n const mde = deltaThreshold + (z * Math.SQRT2 * sd) / Math.sqrt(n)\n\n const scaleAssumed = composites.every((v) => v >= -0.001 && v <= 1.5)\n const headroom = Math.max(0, 1 - mean)\n const underpowered = scaleAssumed && mde > headroom\n\n const sharedChannelCaveat = opts.sharedScorerChannel\n ? 'Holdout and gate share one scoring channel: raising n/reps reduces only idiosyncratic noise — systematic judge bias remains and this MDE is a lower bound. Full debiasing needs an independent second scoring channel (different judge/benchmark family).'\n : undefined\n\n const recommendation = underpowered\n ? `UNDERPOWERED: minimum detectable lift ${mde.toFixed(3)} exceeds the ${headroom.toFixed(3)} headroom above the baseline (${mean.toFixed(3)}) — no achievable effect can ship at this budget. Raise paired n (scenarios x reps) to ~${Math.ceil(((z * Math.SQRT2 * sd) / Math.max(headroom - deltaThreshold, 0.01)) ** 2)} or reduce worker variance before searching.`\n : `Minimum detectable lift at n=${n}: ${mde.toFixed(3)} (baseline sd ${sd.toFixed(3)}). Effects smaller than this cannot clear the gate; budget the search for effects you believe exceed it.`\n\n return {\n n,\n sd,\n mde,\n baselineMean: mean,\n headroom,\n underpowered,\n scaleAssumed,\n deltaThreshold,\n confidence,\n ...(sharedChannelCaveat ? { sharedChannelCaveat } : {}),\n recommendation: sharedChannelCaveat\n ? `${recommendation} ${sharedChannelCaveat}`\n : recommendation,\n }\n}\n"],"mappings":";;;;AAgCA,SAAgB,+BAA+B,aAAa,KAAc;CACxE,IAAI,CAAC,OAAO,SAAS,UAAU,KAAK,cAAc,KAAK,cAAc,GACnE,MAAM,IAAI,MACR,oEAAoE,YACtE;CAEF,MAAM,iBAAiB,IAAI,cAAc;CACzC,OAAO,KAAK,KAAK,KAAK,KAAK,IAAI,aAAa,CAAC;AAC/C;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCA,SAAgB,gBACd,QACA,OACA,UAAkC,CAAC,GACZ;CACvB,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,SAAS,SAAS,GAC5B,MAAM,IAAI,MAAM,kDAAkD,WAAW;CAG/E,MAAM,eAAe,+BADF,QAAQ,cAAc,GACqB;CAC9D,MAAM,mBAAmB,QAAQ,YAAY;CAC7C,IAAI,CAAC,OAAO,UAAU,gBAAgB,KAAK,mBAAmB,GAC5D,MAAM,IAAI,MAAM,6DAA6D,kBAAkB;CAEjG,MAAM,eAAe,KAAK,IAAI,kBAAkB,YAAY;CAC5D,MAAM,YAAY,gBAAgB,QAAQ,OAAO,OAAO;CACxD,MAAM,aAAa,UAAU,KAAK;CAIlC,MAAM,gBACJ,UAAU,IAAI,MACb,CAAC,OAAO,SAAS,UAAU,GAAG,KAC7B,CAAC,OAAO,SAAS,UAAU,IAAI,KAC/B,UAAU,QAAQ,UAAU;CAEhC,IAAI,UAAU,cACZ,OAAO;EACL;EACA,QAAQ;EACR,QAAQ;EACR;EACA;EACA;EACA,aAAa,cAAc,CAAC,iBAAiB,UAAU,MAAM;CAC/D;CAIF,MAAM,QAAQ,eADM,OAAO,KAAK,OAAO,UAAU,MAAM,SAAU,QAAQ,SAClC,GAAG,SAAS;CACnD,MAAM,WAAW,QAAQ,cAAc,SAAS,UAAU,OAAO,UAAU;CAC3E,OAAO;EACL;EACA,QAAQ;EACR,QAAQ,MAAM;EACd;EACA;EACA;EACA,aACE,cACA,CAAC,iBACD,WAAW,aACX,MAAM,WAAW,IAAI,UAAU,cAAc;CACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACqCA,SAAgB,oBACd,QACA,OACA,YAA+B,QACV;CACrB,MAAM,cAAc,OAAO,WAAW,IAAI,OAAO,uBAAuB,QAAQ,KAAK;CACrF,IAAI,cAAc,UAChB,OAAO;EAAE,WAAW;EAAoB,aAAa;EAAM;CAAY;CAEzE,MAAM,cAAc,kBAAkB,QAAQ,KAAK;CACnD,IAAI,gBAAgB,MAClB,OAAO;EAAE,WAAW;EAA0B;EAAa;CAAY;CAEzE,OAAO;EAAE,WAAW;EAAkB,aAAa;EAAM;CAAY;AACvE;;;;;;AAOA,SAAgB,sBACd,QACA,OACA,UAA0C,CAAC,GAClB;CACzB,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,MACR,gDAAgD,OAAO,OAAO,MAAM,MAAM,OAAO,EACnF;CAEF,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,SAAS,SAAS,GAC5B,MAAM,IAAI,MAAM,wDAAwD,WAAW;CAErF,MAAM,aAAa,QAAQ,cAAc;CACzC,MAAM,eAAe,+BAA+B,UAAU;CAC9D,MAAM,mBAAmB,QAAQ,YAAY;CAC7C,IAAI,CAAC,OAAO,UAAU,gBAAgB,KAAK,mBAAmB,GAC5D,MAAM,IAAI,MACR,mEAAmE,kBACrE;CAEF,MAAM,eAAe,KAAK,IAAI,kBAAkB,YAAY;CAC5D,MAAM,IAAI,OAAO;CACjB,MAAM,aAAa,KAAK;CACxB,MAAM,EAAE,aAAa,gBAAgB,oBAAoB,QAAQ,OAAO,QAAQ,SAAS;CAEzF,IAAI;CAcJ,IAAI,gBAAgB,MAAM;EAKxB,MAAM,cAAc,OAAO,KAAK,MAAM,IAAI,WAAW;EACrD,MAAM,gBAAgB,MAAM,KAAK,MAAM,IAAI,WAAW;EAMtD,MAAM,QAAQ,0BAA0B,aAAa,eAAe,UAAU;EAC9E,MAAM,QAAQ,0BAA0B,aAAa,eAAe,UAAU;EAC9E,MAAM,MAAM,MAAM,QAAQ;EAC1B,OAAO;GACL,WAAW;GACX,QAAQ;GACR,OAAO,MAAM,iBAAiB;GAC9B;GACA,MAAM,MAAM,QAAQ;GACpB,WAAW;GACX,SAAS;IACP,GAAG,MAAM;IACT,GAAG,MAAM;IACT,aAAa,MAAM;IACnB,QAAQ,MAAM;GAChB;GACA,QAAQ;GACR,iBAAiB,MAAM;GACvB,OAAO;GACP,cAAc;EAChB;CACF,OAAO;EACL,MAAM,qBAAqB,QAAQ,cAAc,WAAW,WAAW;EACvE,MAAM,OAAO,gBAAgB,QAAQ,OAAO;GAC1C;GACA,WAAW,QAAQ;GACnB,WAAW;GACX,MAAM,QAAQ;GACd;GACA,UAAU,QAAQ;EACpB,CAAC;EACD,MAAM,KAAK,KAAK;EAChB,OAAO;GACL,WAAW,uBAAuB,SAAS,mBAAmB;GAC9D,QAAQ,KAAK;GACb,OAAO,uBAAuB,SAAS,GAAG,OAAO,GAAG;GACpD,KAAK,GAAG;GACR,MAAM,GAAG;GACT,WAAW;GACX,SAAS;GACT,QAAQ,KAAK;GACb,iBAAiB,KAAK;GACtB,OAAO;GACP,cACE,KAAK,WAAW,eACZ,UAAU,KAAK,aAAa,4FACyB,IAAI,KAAK,UAAU,CAAC,EAAE,KAC3E;EACR;CACF;CAEA,MAAM,gBACJ,CAAC,OAAO,SAAS,KAAK,GAAG,KAAK,CAAC,OAAO,SAAS,KAAK,IAAI,KAAK,KAAK,QAAQ,KAAK;CACjF,MAAM,qBAAqB,CAAC,gBACxB,KACA,gBAAgB,IACd,uCACA,KAAK,YAAY,QAAQ,KAAK,QAAQ,gBAAgB,IACpD,kDACA,OAAO,KAAK,MAAM,8BAA8B,IAAI,KAAK,GAAG;CAIpE,MAAM,kBACJ,KAAK,YAAY,QAAQ,aAAa,KAAK,EAAE,KAAK,QAAQ,SAAS,IAAI;CAEzE,OAAO;EACL;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,SAAS,cAAc,CAAC,iBAAiB,KAAK,mBAAmB,CAAC;EAClE,GAAG;CACL;AACF;AAEA,SAAS,IAAI,GAAmB;CAC9B,OAAO,EAAE,QAAQ,CAAC;AACpB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AC3RA,MAAa,oBAAoB;;;;;;;;;;;;AAsBjC,SAAgB,YACd,WACA,UACA,aACA,QACe;CACf,MAAM,aACJ,QACA,WACuB;EACvB,MAAM,SAAS,OAAO,IAAI,MAAM;EAChC,IAAI,CAAC,QAAQ,OAAO,KAAA;EACpB,MAAM,OAAiB,CAAC;EACxB,KAAK,MAAM,KAAK,OAAO,OAAO,MAAM,GAAG;GACrC,IAAI,EAAE,WAAW,MACf,MAAM,IAAI,MAAM,sBAAsB,OAAO,gCAAgC;GAE/E,MAAM,IAAI,OAAO,CAAC;GAClB,IAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAC7C,MAAM,IAAI,MAAM,sBAAsB,OAAO,uCAAuC;GAEtF,IAAI,OAAO,MAAM,UAAU,KAAK,KAAK,CAAC;EACxC;EACA,IAAI,KAAK,WAAW,GAAG,OAAO,KAAA;EAC9B,OAAO,KAAK,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,KAAK;CAChD;CAEA,MAAM,WAAW,WAAmB,YAAY,IAAI,OAAO,MAAM,GAAG,CAAC,CAAC,MAAM,EAAE;CAC9E,MAAM,YAAY,CAAC,GAAG,UAAU,KAAK,CAAC,CAAC,CAAC,OAAO,OAAO,CAAC,CAAC,KAAK;CAC7D,MAAM,YAAY,CAAC,GAAG,SAAS,KAAK,CAAC,CAAC,CAAC,OAAO,OAAO,CAAC,CAAC,KAAK;CAG5D,IAAI,UAAU,WAAW,UAAU,UAAU,UAAU,MAAM,GAAG,MAAM,MAAM,UAAU,EAAE,GACtF,MAAM,IAAI,MACR,2EACgB,UAAU,KAAK,GAAG,EAAE,cAAc,UAAU,KAAK,GAAG,EAAE,+EAExE;CAGF,MAAM,SAAmB,CAAC;CAC1B,MAAM,QAAkB,CAAC;CACzB,MAAM,UAAoB,CAAC;CAC3B,KAAK,MAAM,UAAU,WAAW;EAC9B,MAAM,IAAI,UAAU,UAAU,MAAM;EACpC,MAAM,IAAI,UAAU,WAAW,MAAM;EAGrC,IAAI,MAAM,KAAA,KAAa,MAAM,KAAA,GAAW;EACxC,IAAI,MAAM,KAAA,KAAa,MAAM,KAAA,GAC3B,MAAM,IAAI,MAAM,sBAAsB,OAAO,uCAAuC;EAEtF,OAAO,KAAK,CAAC;EACb,MAAM,KAAK,CAAC;EACZ,QAAQ,KAAK,MAAM;CACrB;CACA,OAAO;EAAE;EAAQ;EAAO;CAAQ;AAClC;;;;;;;;;;;;;;;;;;;;;;;;;;;AAuFA,SAAgB,oBACd,QACA,OAAmC,CAAC,GACf;CACrB,MAAM,iBAAiB,KAAK,kBAAkB;CAC9C,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,OAAO,KAAK,QAAQ;CAa1B,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,WAAW,sBAAsB,OAAO,QAAQ,OAAO,OAAO;EAClE;EACA;EACA;EACA;EACA,WAAW;EACX,UAAU,KAAK;CACjB,CAAC;CAID,MAAM,YACJ,SAAS,aACT,gBAAgB,OAAO,QAAQ,OAAO,OAAO;EAAE;EAAY;EAAW;EAAW;CAAK,CAAC;CACzF,MAAM,kBACJ,cAAc,WACV,YACA,gBAAgB,OAAO,QAAQ,OAAO,OAAO;EAC3C;EACA;EACA,WAAW;EACX;CACF,CAAC;CACP,MAAM,IAAI,OAAO,OAAO;CACxB,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,GAAG;EAC7B,MAAM,QAAQ,OAAO,MAAM,MAAM;EACjC,MAAM,SAAS,OAAO,OAAO,MAAM;EACnC,IAAI,KAAK,IAAI,QAAQ,MAAM,IAAI,MAAM,QAAQ;CAC/C;CACA,MAAM,cAAc,MAAM,IAAI,IAAI,OAAO;CACzC,OAAO;EACL;EACA;EACA;EACA;EACA,mBAAmB,SAAS;EAC5B,SAAS,SAAS;EAClB;EACA;EACA,iBAAiB,SAAS;EAC1B,gBAAgB,SAAS;EACzB,QAAQ,SAAS;EACjB,aAAa,SAAS;EACtB,SAAS,CAAC,SAAS;CACrB;AACF;;;;AA+BA,SAAgB,YAAY,QAA2B;CACrD,OAAO,OAAO,MAAM,MAAM,KAAK,IAAI,CAAC,IAAI,GAAG,IAAI,MAAM;AACvD;;;;;;;;;;;;;;;;;;;AAoBA,SAAgB,qBACd,WACA,UACA,aACA,oBACA,OAQI,CAAC,GACkB;CACvB,MAAM,MAA6B,CAAC;CACpC,KAAK,MAAM,OAAO,oBAAoB;EACpC,MAAM,SAAS,YAAY,WAAW,UAAU,cAAc,MAAM,EAAE,WAAW,IAAI;EACrF,IAAI,OAAO,OAAO,WAAW,GAAG;EAChC,MAAM,YAAY,KAAK,aAAa,MAAO,YAAY,CAAC,GAAG,OAAO,QAAQ,GAAG,OAAO,KAAK,CAAC;EAC1F,MAAM,qBAAqB,KAAK,aAAA;EAChC,MAAM,SAAS;GACb,YAAY,KAAK,cAAc;GAC/B,WAAW,KAAK,aAAa;GAC7B,WAAW;GACX,MAAM,KAAK,QAAQ;EACrB;EACA,MAAM,QAAQ,sBAAsB,OAAO,QAAQ,OAAO,OAAO,MAAM;EACvE,MAAM,aAAa,sBAAsB,OAAO,OAAO,OAAO,QAAQ;GACpE,GAAG;GACH,WAAW;EACb,CAAC;EACD,MAAM,YAAY,MAAM,aAAa,gBAAgB,OAAO,QAAQ,OAAO,OAAO,MAAM;EACxF,IAAI,KAAK;GACP,WAAW;GACX;GACA;GACA,IAAI;IAAE,KAAK,MAAM;IAAK,MAAM,MAAM;GAAK;GACvC,mBAAmB,MAAM;GACzB,SAAS,MAAM;GACf,eAAe,MAAM;GAgBrB,WAAW,UAAU,MAAM,CAAC,aAAa,WAAW;GACpD;GACA,GAAG,OAAO,OAAO;EACnB,CAAC;CACH;CACA,OAAO;AACT;;;;ACjUA,SAAS,KAAK,YAA4B;CACxC,IAAI,cAAc,KAAM,OAAO;CAC/B,IAAI,cAAc,KAAM,OAAO;CAC/B,IAAI,cAAc,IAAK,OAAO;CAC9B,OAAO;AACT;;;;;AAMA,SAAgB,eAAe,MAA6C;CAC1E,MAAM,aAAa,KAAK,mBAAmB,QAAQ,MAAM,OAAO,SAAS,CAAC,CAAC;CAC3E,IAAI,WAAW,SAAS,GACtB,MAAM,IAAI,MACR,kFAAkF,WAAW,QAC/F;CAEF,MAAM,iBAAiB,KAAK,kBAAkB;CAC9C,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,IAAI,KAAK,WAAW,WAAW;CACrC,IAAI,IAAI,GAAG,MAAM,IAAI,MAAM,6CAA6C,GAAG;CAE3E,MAAM,OAAO,WAAW,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,WAAW;CAChE,MAAM,WACJ,WAAW,QAAQ,GAAG,MAAM,KAAK,IAAI,SAAS,IAAI,OAAO,CAAC,KAAK,WAAW,SAAS;CACrF,MAAM,KAAK,KAAK,KAAK,QAAQ;CAC7B,MAAM,IAAI,KAAK,UAAU;CACzB,MAAM,MAAM,iBAAkB,IAAI,KAAK,QAAQ,KAAM,KAAK,KAAK,CAAC;CAEhE,MAAM,eAAe,WAAW,OAAO,MAAM,KAAK,SAAU,KAAK,GAAG;CACpE,MAAM,WAAW,KAAK,IAAI,GAAG,IAAI,IAAI;CACrC,MAAM,eAAe,gBAAgB,MAAM;CAE3C,MAAM,sBAAsB,KAAK,sBAC7B,8PACA,KAAA;CAEJ,MAAM,iBAAiB,eACnB,yCAAyC,IAAI,QAAQ,CAAC,EAAE,eAAe,SAAS,QAAQ,CAAC,EAAE,gCAAgC,KAAK,QAAQ,CAAC,EAAE,0FAA0F,KAAK,MAAO,IAAI,KAAK,QAAQ,KAAM,KAAK,IAAI,WAAW,gBAAgB,GAAI,MAAM,CAAC,EAAE,gDACzT,gCAAgC,EAAE,IAAI,IAAI,QAAQ,CAAC,EAAE,gBAAgB,GAAG,QAAQ,CAAC,EAAE;CAEvF,OAAO;EACL;EACA;EACA;EACA,cAAc;EACd;EACA;EACA;EACA;EACA;EACA,GAAI,sBAAsB,EAAE,oBAAoB,IAAI,CAAC;EACrD,gBAAgB,sBACZ,GAAG,eAAe,GAAG,wBACrB;CACN;AACF"}
@@ -1,9 +1,9 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
2
  import { r as canonicalString } from "./canonical-IL-Bu-14.js";
3
- import { K as recoverTruncatedJson, n as JudgeParseError } from "./llm-judge-BhasIPFT.js";
3
+ import { K as recoverTruncatedJson, n as JudgeParseError } from "./llm-judge-Du7WQPh7.js";
4
4
  import { i as CostLedger } from "./cost-ledger-B1qx30B4.js";
5
5
  import { t as certificationEvidenceDigest } from "./verdict-B0xltqu6.js";
6
- import { f as ModelSubstitutionError, h as assertServedModel, o as costReceiptFromLlm, s as costReceiptFromLlmError, u as maximumChargeForLlmRequest } from "./llm-client-hgDieDNN.js";
6
+ import { f as ModelSubstitutionError, h as assertServedModel, o as costReceiptFromLlm, s as costReceiptFromLlmError, u as maximumChargeForLlmRequest } from "./llm-client-BFMRpmqb.js";
7
7
  import { createHash, randomUUID } from "node:crypto";
8
8
  import { harnessSupportsModel } from "@tangle-network/agent-interface";
9
9
  //#region src/agent-profile.ts
@@ -583,4 +583,4 @@ function extractProducedState(events) {
583
583
  //#endregion
584
584
  export { CODING_HARNESSES as a, agentProfileId as c, harnessAxisOf as d, verifyCompletion as i, agentProfileModelId as l, completionVerdict as n, HARNESS_NATIVE_MODEL as o, createLlmCorrectnessChecker as r, agentProfileHash as s, extractProducedState as t, expandProfileAxes as u };
585
585
 
586
- //# sourceMappingURL=produced-state-DZ89riy5.js.map
586
+ //# sourceMappingURL=produced-state-Be0BK3RN.js.map