@tangle-network/agent-eval 0.133.2 → 0.133.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/CHANGELOG.md +156 -0
  2. package/dist/{analyze-runs-BmX-h_yn.d.ts → analyze-runs-BClW9OSe.d.ts} +3 -3
  3. package/dist/{analyze-runs-BmX-h_yn.d.ts.map → analyze-runs-BClW9OSe.d.ts.map} +1 -1
  4. package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
  5. package/dist/analyze-runs-qk8op0tN.js.map +1 -0
  6. package/dist/baseline-BaPxoROc.js +149 -0
  7. package/dist/baseline-BaPxoROc.js.map +1 -0
  8. package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
  9. package/dist/baseline-D_fT6277.d.ts.map +1 -0
  10. package/dist/benchmarks/index.d.ts +1 -1
  11. package/dist/benchmarks/index.js +1 -1
  12. package/dist/{benchmarks-CJr1H1_a.js → benchmarks-BP9sgMia.js} +3 -3
  13. package/dist/{benchmarks-CJr1H1_a.js.map → benchmarks-BP9sgMia.js.map} +1 -1
  14. package/dist/builder-eval/index.js +1 -1
  15. package/dist/campaign/index.d.ts +2 -2
  16. package/dist/campaign/index.js +2 -2
  17. package/dist/{campaign-BJjn1rhw.js → campaign--V4ffEKR.js} +12 -6
  18. package/dist/{campaign-BJjn1rhw.js.map → campaign--V4ffEKR.js.map} +1 -1
  19. package/dist/{client-COvaLoQG.d.ts → client-Du7B81wW.d.ts} +28 -14
  20. package/dist/client-Du7B81wW.d.ts.map +1 -0
  21. package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
  22. package/dist/client-LIuo-KPv.js.map +1 -0
  23. package/dist/contract/index.d.ts +3 -3
  24. package/dist/contract/index.d.ts.map +1 -1
  25. package/dist/contract/index.js +9 -8
  26. package/dist/contract/index.js.map +1 -1
  27. package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
  28. package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
  29. package/dist/hosted/index.d.ts +1 -1
  30. package/dist/hosted/index.d.ts.map +1 -1
  31. package/dist/hosted/index.js +1 -1
  32. package/dist/{index-BREtv3ZZ.d.ts → index-B5MNN1f1.d.ts} +3 -3
  33. package/dist/{index-BREtv3ZZ.d.ts.map → index-B5MNN1f1.d.ts.map} +1 -1
  34. package/dist/{index-C7Wue8R6.d.ts → index-DOqvIJ8I.d.ts} +27 -10
  35. package/dist/index-DOqvIJ8I.d.ts.map +1 -0
  36. package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
  37. package/dist/index-DSC51roc2.d.ts.map +1 -0
  38. package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
  39. package/dist/index-DuhJaaiH.d.ts.map +1 -0
  40. package/dist/index.d.ts +56 -10
  41. package/dist/index.d.ts.map +1 -1
  42. package/dist/index.js +29 -20
  43. package/dist/index.js.map +1 -1
  44. package/dist/ledger-core/index.d.ts +2 -2
  45. package/dist/ledger-core/index.js +2 -2
  46. package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
  47. package/dist/ledger-core-DAKFKRzi.js.map +1 -0
  48. package/dist/matrix/index.d.ts +1 -1
  49. package/dist/meta-eval/index.d.ts +1 -1
  50. package/dist/meta-eval/index.js +2 -2
  51. package/dist/multishot/index.d.ts +1 -1
  52. package/dist/openapi.json +1 -1
  53. package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
  54. package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
  55. package/dist/pipelines/index.d.ts +1 -1
  56. package/dist/pipelines/index.js +3 -2
  57. package/dist/pipelines/index.js.map +1 -1
  58. package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
  59. package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
  60. package/dist/{release-report-CjHWa8Ia.d.ts → release-report-DKBtegGt.d.ts} +2 -2
  61. package/dist/{release-report-CjHWa8Ia.d.ts.map → release-report-DKBtegGt.d.ts.map} +1 -1
  62. package/dist/reporting.d.ts +3 -3
  63. package/dist/reporting.js +4 -4
  64. package/dist/{researcher-CbSKhK8z.d.ts → researcher-BtD5U1Up.d.ts} +2 -2
  65. package/dist/{researcher-CbSKhK8z.d.ts.map → researcher-BtD5U1Up.d.ts.map} +1 -1
  66. package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
  67. package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
  68. package/dist/rl.d.ts +1 -1
  69. package/dist/rl.js +4 -4
  70. package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
  71. package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
  72. package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts → skillopt-optimization-method-Dxr8pdZd.d.ts} +12 -7
  73. package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts.map → skillopt-optimization-method-Dxr8pdZd.d.ts.map} +1 -1
  74. package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-vvJ4bMNI.js} +119 -24
  75. package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +1 -0
  76. package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
  77. package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
  78. package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
  79. package/dist/statistics-RwRNu2__.js.map +1 -0
  80. package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
  81. package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
  82. package/dist/{summary-report-CFnQgNfg.d.ts → summary-report-DyOhItws.d.ts} +3 -2
  83. package/dist/summary-report-DyOhItws.d.ts.map +1 -0
  84. package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
  85. package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
  86. package/docs/design/statistics-decisions.md +271 -0
  87. package/docs/design.md +1 -0
  88. package/docs/insight-report.md +1 -1
  89. package/docs/research-report-methodology.md +4 -1
  90. package/package.json +2 -1
  91. package/dist/analyze-runs-B-afTpCv.js.map +0 -1
  92. package/dist/baseline-DcX5hQDv.js.map +0 -1
  93. package/dist/baseline-hG3K85h4.d.ts.map +0 -1
  94. package/dist/client-COvaLoQG.d.ts.map +0 -1
  95. package/dist/client-CYzbdJOZ.js.map +0 -1
  96. package/dist/index-C7Wue8R6.d.ts.map +0 -1
  97. package/dist/index-DSC51roc.d.ts.map +0 -1
  98. package/dist/index-nhIYz9hn.d.ts.map +0 -1
  99. package/dist/ledger-core-CPZfcrC2.js.map +0 -1
  100. package/dist/skillopt-optimization-method-CF6a327Q.js.map +0 -1
  101. package/dist/statistics-DWM_AyLe.js.map +0 -1
  102. package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
  103. package/dist/summary-report-CFnQgNfg.d.ts.map +0 -1
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","names":[],"sources":["../../src/pipelines/budget-breach.ts","../../src/pipelines/failure-cluster.ts","../../src/pipelines/first-divergence.ts","../../src/pipelines/judge-agreement.ts","../../src/pipelines/regression.ts","../../src/pipelines/stuck-loop.ts","../../src/pipelines/tool-waste.ts"],"sourcesContent":["/**\n * BudgetBreachView — aggregates breach events across the corpus.\n *\n * Answers: which dimensions get hit most often? Which scenarios are\n * underbudgeted? Which variants trigger the most breaches?\n */\n\nimport type { BudgetSpec } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface BudgetBreachFinding {\n runId: string\n scenarioId: string\n variantId?: string\n dimension: keyof BudgetSpec\n limit: number\n consumed: number\n excessRatio: number\n timestamp: number\n}\n\nexport interface BudgetBreachReport {\n findings: BudgetBreachFinding[]\n byDimension: Record<string, number>\n byScenario: Record<string, number>\n byVariant: Record<string, number>\n totalRuns: number\n breachedRunRatio: number\n}\n\nexport async function budgetBreachView(\n store: TraceStore,\n options: { scenarioId?: string; variantId?: string } = {},\n): Promise<BudgetBreachReport> {\n const runs = await store.listRuns({\n scenarioId: options.scenarioId,\n variantId: options.variantId,\n })\n const findings: BudgetBreachFinding[] = []\n const byDimension: Record<string, number> = {}\n const byScenario: Record<string, number> = {}\n const byVariant: Record<string, number> = {}\n\n for (const run of runs) {\n const entries = await store.budget(run.runId)\n for (const e of entries) {\n if (!e.breached) continue\n const excessRatio = e.limit > 0 ? e.consumed / e.limit : Infinity\n findings.push({\n runId: run.runId,\n scenarioId: run.scenarioId,\n variantId: run.variantId,\n dimension: e.dimension,\n limit: e.limit,\n consumed: e.consumed,\n excessRatio,\n timestamp: e.timestamp,\n })\n byDimension[e.dimension] = (byDimension[e.dimension] ?? 0) + 1\n byScenario[run.scenarioId] = (byScenario[run.scenarioId] ?? 0) + 1\n if (run.variantId) byVariant[run.variantId] = (byVariant[run.variantId] ?? 0) + 1\n }\n }\n\n const breachedRuns = new Set(findings.map((f) => f.runId))\n return {\n findings,\n byDimension,\n byScenario,\n byVariant,\n totalRuns: runs.length,\n breachedRunRatio: runs.length > 0 ? breachedRuns.size / runs.length : 0,\n }\n}\n","/**\n * FailureClusterView — groups failed runs by (failureClass, triggerTool,\n * argHash-prefix) so weekly reviews can prioritize the top-N clusters.\n *\n * Each cluster includes: N runs, scenarios affected, representative\n * error message, a proposed mitigation hint (rule → action table).\n */\n\nimport { classifyFailure, DEFAULT_RULES, type FailureRule } from '../failure-taxonomy'\nimport { argHash, hasCapturedToolArgs, toolSpans } from '../trace/query'\nimport type { FailureClass, Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface FailureCluster {\n failureClass: FailureClass\n /** Tool name when the trigger was a tool span, else undefined. */\n toolName?: string\n /** First 16 chars of argHash — clusters similar args. */\n argPrefix?: string\n /**\n * Source dimension when the trigger was a judge span (e.g. `'format'`,\n * `'safety'`, `'correctness'`). Lets cross-template aggregators\n * group failures by the dimension that fired without overloading\n * `argPrefix`. Optional — clusters without this field deserialize cleanly.\n */\n dimension?: string\n runCount: number\n scenarioIds: string[]\n exampleError?: string\n exampleRunId: string\n}\n\nexport interface FailureClusterReport {\n clusters: FailureCluster[]\n totalFailures: number\n totalRuns: number\n}\n\nexport async function failureClusterView(\n store: TraceStore,\n options: { rules?: FailureRule[]; minClusterSize?: number } = {},\n): Promise<FailureClusterReport> {\n const rules = options.rules ?? DEFAULT_RULES\n const minSize = options.minClusterSize ?? 1\n const runs = await store.listRuns()\n\n type Key = string\n const clusters = new Map<Key, FailureCluster>()\n let totalFailures = 0\n\n for (const run of runs) {\n if (run.status === 'completed' && run.outcome?.pass !== false) continue\n totalFailures++\n const spans = await store.spans({ runId: run.runId })\n const events = await store.events({ runId: run.runId })\n const cls = classifyFailure({ run, spans, events }, rules)\n\n let toolName: string | undefined\n let argPrefix: string | undefined\n let dimension: string | undefined\n if (cls.triggerSpanId) {\n const trig = spans.find((s) => s.spanId === cls.triggerSpanId)\n if (trig?.kind === 'tool') {\n toolName = trig.toolName\n if (hasCapturedToolArgs(trig)) argPrefix = argHash(trig.args).slice(0, 16)\n } else if (trig?.kind === 'judge') {\n dimension = trig.dimension\n }\n }\n // Fallback: look at the last errored tool span\n if (!toolName) {\n const ts = await toolSpans(store, run.runId)\n const errored = ts.filter((t) => t.status === 'error').pop()\n if (errored) {\n toolName = errored.toolName\n if (hasCapturedToolArgs(errored)) argPrefix = argHash(errored.args).slice(0, 16)\n }\n }\n // Secondary signal: any judge span on the failed run carries a\n // dimension. Useful when the rule classified by judge score but\n // didn't surface the trigger span (or surfaced a non-judge span).\n if (!dimension) {\n const judge = spans.find((s) => s.kind === 'judge' && typeof s.dimension === 'string')\n if (judge?.kind === 'judge') dimension = judge.dimension\n }\n\n const key = `${cls.failureClass}|${toolName ?? ''}|${argPrefix ?? ''}|${dimension ?? ''}`\n let cluster = clusters.get(key)\n if (!cluster) {\n cluster = {\n failureClass: cls.failureClass,\n toolName,\n argPrefix,\n dimension,\n runCount: 0,\n scenarioIds: [],\n exampleRunId: run.runId,\n exampleError: firstErrorMessage(spans) ?? cls.reason,\n }\n clusters.set(key, cluster)\n }\n cluster.runCount++\n if (!cluster.scenarioIds.includes(run.scenarioId)) cluster.scenarioIds.push(run.scenarioId)\n }\n\n const arr = [...clusters.values()]\n .filter((c) => c.runCount >= minSize)\n .sort((a, b) => b.runCount - a.runCount)\n\n return { clusters: arr, totalFailures, totalRuns: runs.length }\n}\n\nfunction firstErrorMessage(spans: Span[]): string | undefined {\n const errored = spans.find((s) => s.status === 'error')\n return errored?.error\n}\n","/**\n * FirstDivergenceView — aligns two trajectories by step index, reports\n * the first step where they differ.\n *\n * \"Differ\" is configurable — default is (kind, toolName if tool, model\n * if llm). Use this view to attribute \"why is variant B better?\" to a\n * specific step rather than an aggregate mean delta.\n */\n\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\nexport interface DivergenceReport {\n runA: string\n runB: string\n firstDivergenceIndex: number | null\n aStep?: TrajectoryStep\n bStep?: TrajectoryStep\n reason?: string\n /** Common prefix length (steps that matched). */\n commonPrefixLen: number\n}\n\nexport interface DivergenceOptions {\n /** Returns true if two steps are considered equal. Default: kind + tool/model match. */\n stepEquals?: (a: TrajectoryStep, b: TrajectoryStep) => boolean\n}\n\nexport async function firstDivergenceView(\n store: TraceStore,\n runA: string,\n runB: string,\n options: DivergenceOptions = {},\n): Promise<DivergenceReport> {\n const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)])\n const eq = options.stepEquals ?? defaultStepEquals\n const minLen = Math.min(a.steps.length, b.steps.length)\n for (let i = 0; i < minLen; i++) {\n const aStep = a.steps[i]!\n const bStep = b.steps[i]!\n if (!eq(aStep, bStep)) {\n return {\n runA,\n runB,\n firstDivergenceIndex: i,\n aStep,\n bStep,\n reason: describeDifference(aStep, bStep),\n commonPrefixLen: i,\n }\n }\n }\n if (a.steps.length === b.steps.length) {\n return { runA, runB, firstDivergenceIndex: null, commonPrefixLen: minLen }\n }\n const longer: Trajectory = a.steps.length > b.steps.length ? a : b\n const extra = longer.steps.length - minLen\n // minLen === 0 means one trajectory is empty: divergence is the first step\n // itself (index 0), and the absent side has no step to surface.\n const reason =\n minLen === 0\n ? `one trajectory is empty; the other has ${extra} step(s) starting at index 0`\n : `one trajectory has ${extra} more step(s) after index ${minLen - 1}`\n return {\n runA,\n runB,\n firstDivergenceIndex: minLen,\n aStep: a.steps[minLen],\n bStep: b.steps[minLen],\n reason,\n commonPrefixLen: minLen,\n }\n}\n\nfunction defaultStepEquals(a: TrajectoryStep, b: TrajectoryStep): boolean {\n if (a.span.kind !== b.span.kind) return false\n if (a.span.kind === 'tool' && b.span.kind === 'tool') return a.span.toolName === b.span.toolName\n if (a.span.kind === 'llm' && b.span.kind === 'llm') return a.span.model === b.span.model\n if (a.span.kind === 'judge' && b.span.kind === 'judge')\n return a.span.dimension === b.span.dimension\n return a.span.name === b.span.name\n}\n\nfunction describeDifference(a: TrajectoryStep, b: TrajectoryStep): string {\n if (a.span.kind !== b.span.kind) return `kind ${a.span.kind} vs ${b.span.kind}`\n if (a.span.kind === 'tool' && b.span.kind === 'tool' && a.span.toolName !== b.span.toolName) {\n return `tool ${a.span.toolName} vs ${b.span.toolName}`\n }\n if (a.span.kind === 'llm' && b.span.kind === 'llm' && a.span.model !== b.span.model) {\n return `model ${a.span.model} vs ${b.span.model}`\n }\n return `name \"${a.span.name}\" vs \"${b.span.name}\"`\n}\n","/**\n * JudgeAgreementView — pairwise agreement between judges across the\n * corpus, grouped by dimension.\n *\n * Output drives two workflows:\n * - Judge robustness audit: \"does Claude agree with GPT at κ ≥ 0.6?\"\n * - Calibration tracking: κ vs golden human labels over time (by\n * providing a `humanGoldenJudgeId`).\n */\n\nimport { interRaterReliability, pearsonR } from '../statistics'\nimport type { JudgeSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface JudgePair {\n judgeA: string\n judgeB: string\n dimension: string\n /** Number of (targetSpanId, dimension) tuples both judges scored. */\n commonItems: number\n pearson: number\n krippendorff: number\n}\n\nexport interface JudgeAgreementReport {\n pairs: JudgePair[]\n dimensions: string[]\n judgeIds: string[]\n}\n\nexport async function judgeAgreementView(store: TraceStore): Promise<JudgeAgreementReport> {\n const all = (await store.spans({ kind: 'judge' })).filter(\n (s): s is JudgeSpan => s.kind === 'judge',\n )\n if (all.length === 0) return { pairs: [], dimensions: [], judgeIds: [] }\n\n const byDimension = new Map<string, JudgeSpan[]>()\n for (const s of all) {\n const arr = byDimension.get(s.dimension) ?? []\n arr.push(s)\n byDimension.set(s.dimension, arr)\n }\n\n const judgeIds = [...new Set(all.map((s) => s.judgeId))].sort()\n const pairs: JudgePair[] = []\n for (const [dim, spans] of byDimension) {\n const byJudge = new Map<string, Map<string, number>>()\n for (const s of spans) {\n const m = byJudge.get(s.judgeId) ?? new Map<string, number>()\n m.set(s.targetSpanId, s.score)\n byJudge.set(s.judgeId, m)\n }\n const judgesHere = [...byJudge.keys()]\n for (let i = 0; i < judgesHere.length; i++) {\n for (let j = i + 1; j < judgesHere.length; j++) {\n const judgeI = judgesHere[i]!\n const judgeJ = judgesHere[j]!\n const a = byJudge.get(judgeI)!\n const b = byJudge.get(judgeJ)!\n const common: Array<[number, number]> = []\n for (const [target, scoreA] of a) {\n const scoreB = b.get(target)\n if (scoreB !== undefined) common.push([scoreA, scoreB])\n }\n if (common.length < 2) continue\n const judgeScores = common.map(\n ([scoreA, scoreB]) =>\n [\n { judgeName: judgeI, dimension: dim, score: scoreA, reasoning: '' },\n { judgeName: judgeJ, dimension: dim, score: scoreB, reasoning: '' },\n ] as const,\n )\n const k = interRaterReliability(\n judgeScores[0]!.map((_, k2) => judgeScores.map((pair) => pair[k2]!)),\n )\n pairs.push({\n judgeA: judgeI,\n judgeB: judgeJ,\n dimension: dim,\n commonItems: common.length,\n pearson: pearsonR(\n common.map((c) => c[0]),\n common.map((c) => c[1]),\n ),\n krippendorff: k,\n })\n }\n }\n }\n\n return {\n pairs: pairs.sort((a, b) => b.commonItems - a.commonItems),\n dimensions: [...byDimension.keys()].sort(),\n judgeIds,\n }\n}\n","/**\n * RegressionView — compares a candidate slice to a baseline slice on a\n * named metric. Delegates the statistics (Welch's t-test, Cohen's d,\n * IQR stability) to `baseline.ts`.\n *\n * This is the entry point for CI regression gates: \"given runs tagged\n * release=A and release=B, did any metric regress?\"\n */\n\nimport { type BaselineOptions, type BaselineReport, compareToBaseline } from '../baseline'\nimport { aggregateLlm, llmSpans, runFailureClass } from '../trace/query'\nimport type { Run } from '../trace/schema'\nimport type { RunFilter, TraceStore } from '../trace/store'\n\nexport interface RegressionSpec {\n metric: string\n higherIsBetter: boolean\n /** Extract a scalar from a run. Default extractors handle common metrics. */\n extract?: (run: Run, store: TraceStore) => Promise<number | null>\n}\n\nexport interface RegressionOptions extends BaselineOptions {\n baseline: RunFilter\n candidate: RunFilter\n}\n\nexport async function regressionView(\n store: TraceStore,\n metrics: RegressionSpec[],\n options: RegressionOptions,\n): Promise<BaselineReport> {\n const baselineRuns = await store.listRuns(options.baseline)\n const candidateRuns = await store.listRuns(options.candidate)\n const samples = await Promise.all(\n metrics.map(async (m) => {\n const extract = m.extract ?? defaultExtract(m.metric)\n const baseline = await extractAll(baselineRuns, extract, store)\n const candidate = await extractAll(candidateRuns, extract, store)\n return { metric: m.metric, higherIsBetter: m.higherIsBetter, baseline, candidate }\n }),\n )\n return compareToBaseline(samples, options)\n}\n\nasync function extractAll(\n runs: Run[],\n extract: (r: Run, s: TraceStore) => Promise<number | null>,\n store: TraceStore,\n): Promise<number[]> {\n const out: number[] = []\n for (const r of runs) {\n const v = await extract(r, store)\n if (v !== null && Number.isFinite(v)) out.push(v)\n }\n return out\n}\n\nfunction defaultExtract(metric: string): (run: Run, store: TraceStore) => Promise<number | null> {\n return async (run, store) => {\n switch (metric) {\n case 'score':\n case 'overallScore':\n return run.outcome?.score ?? null\n case 'pass':\n return run.outcome?.pass === true ? 1 : 0\n case 'durationMs':\n return run.endedAt && run.startedAt ? run.endedAt - run.startedAt : null\n case 'costUsd': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).costUsd\n }\n case 'inputTokens': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).inputTokens\n }\n case 'outputTokens': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).outputTokens\n }\n case 'failureClass': {\n return runFailureClass(run) === 'success' ? 1 : 0\n }\n default:\n return null\n }\n }\n}\n","/**\n * StuckLoopView — detects when an agent calls the same tool with the\n * same (or structurally similar) arguments ≥ N times in a short window.\n *\n * Rationale: agents that loop are the number-one production failure\n * mode on long-horizon flows. The view returns (runId, toolName,\n * argHash, occurrences, windowMs) for each detected loop plus a\n * fraction of runs affected.\n */\n\nimport { executionTrackByLane } from '../trace/execution-tracks'\nimport { argHash, hasCapturedToolArgs } from '../trace/query'\nimport { isToolSpan, type Span, type ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nconst DEFAULT_MAX_WINDOW_MS = 60_000\nconst DEFAULT_MAX_INTERVENING_TOOL_CALLS = 0\n\ninterface ScopedToolCall {\n span: ToolSpan\n scopeSpanId: string | null\n laneSpanId: string | null\n}\n\ninterface IndexedToolCall extends ScopedToolCall {\n toolCallIndex: number\n trackId: string\n}\n\nexport interface StuckLoopFinding {\n runId: string\n toolName: string\n argHash: string\n /** Calls in this episode's densest qualifying interval, not the whole-run total. */\n occurrences: number\n spanIds: string[]\n /** Nearest agent ancestor, or the direct parent when ancestry is incomplete. */\n scopeSpanId?: string\n /** Milliseconds between first and last call in the loop. */\n windowMs: number\n}\n\nexport interface StuckLoopReport {\n findings: StuckLoopFinding[]\n affectedRunRatio: number\n totalRuns: number\n}\n\nexport interface StuckLoopOptions {\n /** Minimum call count to flag a loop (default 3). */\n minOccurrences?: number\n /** Maximum time between the first and last repeated call (default 60 seconds). */\n maxWindowMs?: number\n /**\n * Maximum other tool calls allowed between adjacent repeats (default 0).\n * Set to 1 to detect alternating patterns such as A,B,A,B,A.\n */\n maxInterveningToolCalls?: number\n /** Filter to a specific runId; omit to scan the entire corpus. */\n runId?: string\n}\n\nexport async function stuckLoopView(\n store: TraceStore,\n options: StuckLoopOptions = {},\n): Promise<StuckLoopReport> {\n const minOccurrences = options.minOccurrences ?? 3\n const maxWindowMs = options.maxWindowMs ?? DEFAULT_MAX_WINDOW_MS\n const maxInterveningToolCalls =\n options.maxInterveningToolCalls ?? DEFAULT_MAX_INTERVENING_TOOL_CALLS\n if (!Number.isInteger(minOccurrences) || minOccurrences < 1) {\n throw new RangeError('minOccurrences must be a positive integer')\n }\n if (!Number.isFinite(maxWindowMs) || maxWindowMs < 0) {\n throw new RangeError('maxWindowMs must be a finite non-negative number')\n }\n if (!Number.isInteger(maxInterveningToolCalls) || maxInterveningToolCalls < 0) {\n throw new RangeError('maxInterveningToolCalls must be a non-negative integer')\n }\n const runs = options.runId\n ? [{ runId: options.runId }]\n : (await store.listRuns()).map((r) => ({ runId: r.runId }))\n\n const findings: StuckLoopFinding[] = []\n for (const { runId } of runs) {\n const spans = await store.spans({ runId })\n const spansById = new Map(spans.map((span) => [span.spanId, span]))\n const scopedTools: ScopedToolCall[] = spans\n .filter(isToolSpan)\n .map((span, sourceIndex) => ({ span, sourceIndex }))\n .sort((a, b) => a.span.startedAt - b.span.startedAt || a.sourceIndex - b.sourceIndex)\n .map(({ span }) => ({ span, ...executionScope(span, spansById) }))\n const trackByLane = executionTrackByLane(\n scopedTools.map((call) => {\n const direct = call.laneSpanId === null || call.laneSpanId === call.scopeSpanId\n const timed = direct\n ? call.span\n : call.laneSpanId\n ? spansById.get(call.laneSpanId)\n : undefined\n return {\n key: executionKey(call),\n scopeKey: JSON.stringify(call.scopeSpanId),\n start: timed?.startedAt ?? null,\n end: timed?.endedAt ?? null,\n }\n }),\n )\n const nextToolIndexByTrack = new Map<string, number>()\n const orderedTools: IndexedToolCall[] = scopedTools.map((call) => {\n const trackId = trackByLane.get(executionKey(call))!\n const toolCallIndex = nextToolIndexByTrack.get(trackId) ?? 0\n nextToolIndexByTrack.set(trackId, toolCallIndex + 1)\n return { ...call, toolCallIndex, trackId }\n })\n const byKey = new Map<\n string,\n {\n calls: IndexedToolCall[]\n argHash: string\n toolName: string\n scopeSpanId: string | null\n }\n >()\n for (const call of orderedTools) {\n if (!hasCapturedToolArgs(call.span)) continue\n const h = argHash(call.span.args)\n const key = JSON.stringify([call.trackId, call.span.toolName, h])\n const bucket = byKey.get(key) ?? {\n calls: [],\n argHash: h,\n toolName: call.span.toolName,\n scopeSpanId: call.scopeSpanId,\n }\n bucket.calls.push(call)\n byKey.set(key, bucket)\n }\n for (const { calls, argHash: h, toolName, scopeSpanId } of byKey.values()) {\n if (calls.length < minOccurrences) continue\n let episodeStart = 0\n for (let episodeEnd = 1; episodeEnd <= calls.length; episodeEnd += 1) {\n const previous = calls[episodeEnd - 1]!\n const next = calls[episodeEnd]\n const episodeEnded =\n next === undefined ||\n next.span.startedAt - previous.span.startedAt > maxWindowMs ||\n next.toolCallIndex - previous.toolCallIndex - 1 > maxInterveningToolCalls ||\n !callsAreSerial(previous, next)\n if (!episodeEnded) continue\n\n const episode = calls.slice(episodeStart, episodeEnd)\n let left = 0\n let bestStart = 0\n let bestEnd = -1\n for (let right = 0; right < episode.length; right += 1) {\n while (episode[right]!.span.startedAt - episode[left]!.span.startedAt > maxWindowMs) {\n left += 1\n }\n if (right - left > bestEnd - bestStart) {\n bestStart = left\n bestEnd = right\n }\n }\n if (bestEnd - bestStart + 1 >= minOccurrences) {\n const loop = episode.slice(bestStart, bestEnd + 1)\n const first = loop[0]!.span.startedAt\n const last = loop[loop.length - 1]!.span.startedAt\n findings.push({\n runId,\n toolName,\n argHash: h,\n occurrences: loop.length,\n spanIds: loop.map((call) => call.span.spanId),\n ...(scopeSpanId ? { scopeSpanId } : {}),\n windowMs: last - first,\n })\n }\n episodeStart = episodeEnd\n }\n }\n }\n\n const affectedRuns = new Set(findings.map((f) => f.runId))\n return {\n findings,\n affectedRunRatio: runs.length > 0 ? affectedRuns.size / runs.length : 0,\n totalRuns: runs.length,\n }\n}\n\nfunction laneKey(call: Pick<ScopedToolCall, 'scopeSpanId' | 'laneSpanId'>): string {\n return JSON.stringify([call.scopeSpanId, call.laneSpanId])\n}\n\nfunction executionKey(call: ScopedToolCall): string {\n return call.laneSpanId === null || call.laneSpanId === call.scopeSpanId\n ? JSON.stringify([call.scopeSpanId, call.span.spanId])\n : laneKey(call)\n}\n\nfunction executionScope(\n span: ToolSpan,\n spansById: ReadonlyMap<string, Span>,\n): { scopeSpanId: string | null; laneSpanId: string | null } {\n const directParent = span.parentSpanId\n if (!directParent) return { scopeSpanId: null, laneSpanId: null }\n\n let currentId: string | undefined = directParent\n let laneSpanId: string | null = null\n const seen = new Set<string>()\n while (currentId && !seen.has(currentId)) {\n seen.add(currentId)\n const current = spansById.get(currentId)\n if (!current) return { scopeSpanId: directParent, laneSpanId: directParent }\n if (current.kind === 'agent') {\n return { scopeSpanId: current.spanId, laneSpanId: laneSpanId ?? current.spanId }\n }\n laneSpanId = current.spanId\n currentId = current.parentSpanId\n }\n return { scopeSpanId: directParent, laneSpanId: directParent }\n}\n\nfunction callsAreSerial(previous: IndexedToolCall, next: IndexedToolCall): boolean {\n return previous.span.endedAt !== undefined && previous.span.endedAt <= next.span.startedAt\n}\n","/**\n * ToolWasteView — fraction of tool calls whose results weren't used\n * downstream. Without a \"used\" signal we fall back to structural\n * proxies: error calls, duplicate calls, and tool calls followed by\n * zero subsequent LLM spans are all considered waste.\n *\n * Consumers can pass a `usageOracle` that inspects a tool span and\n * returns true iff the tool's result appears in a later LLM message,\n * artifact, or state mutation — that's the canonical definition; the\n * default heuristic is a reasonable fallback.\n */\n\nimport { computeToolUseMetrics } from '../tool-use-metrics'\nimport { llmSpans, toolSpans } from '../trace/query'\nimport type { LlmSpan, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface ToolWasteFinding {\n runId: string\n wastedCalls: number\n totalCalls: number\n wasteRate: number\n}\n\nexport interface ToolWasteReport {\n byRun: ToolWasteFinding[]\n overallWasteRate: number\n}\n\nexport interface ToolWasteOptions {\n runId?: string\n usageOracle?: (tool: ToolSpan, later: { llm: Awaited<ReturnType<typeof llmSpans>> }) => boolean\n}\n\nexport async function toolWasteView(\n store: TraceStore,\n options: ToolWasteOptions = {},\n): Promise<ToolWasteReport> {\n const runs = options.runId ? [options.runId] : (await store.listRuns()).map((r) => r.runId)\n\n const byRun: ToolWasteFinding[] = []\n let totalCalls = 0\n let totalWasted = 0\n for (const runId of runs) {\n const tools = await toolSpans(store, runId)\n if (tools.length === 0) {\n byRun.push({ runId, wastedCalls: 0, totalCalls: 0, wasteRate: 0 })\n continue\n }\n const llms = await llmSpans(store, runId)\n // Sort LLM spans once by start time, then build a suffix index of the\n // concatenated message text. `suffixText[i]` is the haystack of every\n // string message content in spans[i..]. Per tool we binary-search the\n // first span started strictly after the tool, then test that one suffix\n // — turning the per-tool O(llms × messages × content) scan into a single\n // O(log llms) lookup over precomputed text.\n const sortedLlm = [...llms].sort((a, b) => a.startedAt - b.startedAt)\n const startTimes = sortedLlm.map((l) => l.startedAt)\n const suffixText = buildSuffixText(sortedLlm)\n let wasted = 0\n for (const t of tools) {\n if (t.status === 'error') {\n wasted++\n continue\n }\n // First LLM span started strictly after this tool (upper-bound search).\n const cutoff = upperBound(startTimes, t.startedAt)\n if (options.usageOracle) {\n if (!options.usageOracle(t, { llm: sortedLlm.slice(cutoff) })) wasted++\n } else {\n // Default heuristic: a tool whose result is NOT mentioned in any\n // later LLM input message is likely wasted. An empty/null result has\n // no payload to propagate downstream — there is nothing to find in a\n // later message, so it is not evidence of waste; skip it.\n const resultStr = stringify(t.result)\n if (resultStr === '') continue\n const haystack = suffixText[cutoff] ?? ''\n const used = haystack.includes(resultStr.slice(0, 120))\n if (!used) wasted++\n }\n }\n const wasteRate = wasted / tools.length\n byRun.push({ runId, wastedCalls: wasted, totalCalls: tools.length, wasteRate })\n totalCalls += tools.length\n totalWasted += wasted\n }\n return { byRun, overallWasteRate: totalCalls > 0 ? totalWasted / totalCalls : 0 }\n}\n\n/**\n * Build per-position suffix haystacks: result[i] is the concatenation of every\n * string message content in spans[i..end]. Built back-to-front so each entry\n * reuses the next one — O(total message text) rather than O(spans²).\n */\nfunction buildSuffixText(spans: LlmSpan[]): string[] {\n const result = new Array<string>(spans.length + 1)\n result[spans.length] = ''\n for (let i = spans.length - 1; i >= 0; i--) {\n const own = spans[i]!.messages.map((m) =>\n typeof m.content === 'string' ? m.content : '',\n ).join('\\n')\n result[i] = `${own}\\n${result[i + 1]}`\n }\n return result\n}\n\n/** Index of the first element strictly greater than `target` in a sorted array. */\nfunction upperBound(sorted: number[], target: number): number {\n let lo = 0\n let hi = sorted.length\n while (lo < hi) {\n const mid = (lo + hi) >>> 1\n if (sorted[mid]! <= target) lo = mid + 1\n else hi = mid\n }\n return lo\n}\n\nfunction stringify(v: unknown): string {\n if (v === null || v === undefined) return ''\n if (typeof v === 'string') return v\n try {\n return JSON.stringify(v)\n } catch {\n return String(v)\n }\n}\n\n// Re-export for convenience in consumers that want both descriptive and usage metrics.\nexport { computeToolUseMetrics }\n"],"mappings":";;;;;;;AA8BA,eAAsB,iBACpB,OACA,UAAuD,CAAC,GAC3B;CAC7B,MAAM,OAAO,MAAM,MAAM,SAAS;EAChC,YAAY,QAAQ;EACpB,WAAW,QAAQ;CACrB,CAAC;CACD,MAAM,WAAkC,CAAC;CACzC,MAAM,cAAsC,CAAC;CAC7C,MAAM,aAAqC,CAAC;CAC5C,MAAM,YAAoC,CAAC;CAE3C,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,UAAU,MAAM,MAAM,OAAO,IAAI,KAAK;EAC5C,KAAK,MAAM,KAAK,SAAS;GACvB,IAAI,CAAC,EAAE,UAAU;GACjB,MAAM,cAAc,EAAE,QAAQ,IAAI,EAAE,WAAW,EAAE,QAAQ;GACzD,SAAS,KAAK;IACZ,OAAO,IAAI;IACX,YAAY,IAAI;IAChB,WAAW,IAAI;IACf,WAAW,EAAE;IACb,OAAO,EAAE;IACT,UAAU,EAAE;IACZ;IACA,WAAW,EAAE;GACf,CAAC;GACD,YAAY,EAAE,cAAc,YAAY,EAAE,cAAc,KAAK;GAC7D,WAAW,IAAI,eAAe,WAAW,IAAI,eAAe,KAAK;GACjE,IAAI,IAAI,WAAW,UAAU,IAAI,cAAc,UAAU,IAAI,cAAc,KAAK;EAClF;CACF;CAEA,MAAM,eAAe,IAAI,IAAI,SAAS,KAAK,MAAM,EAAE,KAAK,CAAC;CACzD,OAAO;EACL;EACA;EACA;EACA;EACA,WAAW,KAAK;EAChB,kBAAkB,KAAK,SAAS,IAAI,aAAa,OAAO,KAAK,SAAS;CACxE;AACF;;;;;;;;;;ACnCA,eAAsB,mBACpB,OACA,UAA8D,CAAC,GAChC;CAC/B,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,UAAU,QAAQ,kBAAkB;CAC1C,MAAM,OAAO,MAAM,MAAM,SAAS;CAGlC,MAAM,2BAAW,IAAI,IAAyB;CAC9C,IAAI,gBAAgB;CAEpB,KAAK,MAAM,OAAO,MAAM;EACtB,IAAI,IAAI,WAAW,eAAe,IAAI,SAAS,SAAS,OAAO;EAC/D;EACA,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,OAAO,IAAI,MAAM,CAAC;EAEpD,MAAM,MAAM,gBAAgB;GAAE;GAAK;GAAO,QAAA,MADrB,MAAM,OAAO,EAAE,OAAO,IAAI,MAAM,CAAC;EACL,GAAG,KAAK;EAEzD,IAAI;EACJ,IAAI;EACJ,IAAI;EACJ,IAAI,IAAI,eAAe;GACrB,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,WAAW,IAAI,aAAa;GAC7D,IAAI,MAAM,SAAS,QAAQ;IACzB,WAAW,KAAK;IAChB,IAAI,oBAAoB,IAAI,GAAG,YAAY,QAAQ,KAAK,IAAI,CAAC,CAAC,MAAM,GAAG,EAAE;GAC3E,OAAO,IAAI,MAAM,SAAS,SACxB,YAAY,KAAK;EAErB;EAEA,IAAI,CAAC,UAAU;GAEb,MAAM,WAAU,MADC,UAAU,OAAO,IAAI,KAAK,EAAA,CACxB,QAAQ,MAAM,EAAE,WAAW,OAAO,CAAC,CAAC,IAAI;GAC3D,IAAI,SAAS;IACX,WAAW,QAAQ;IACnB,IAAI,oBAAoB,OAAO,GAAG,YAAY,QAAQ,QAAQ,IAAI,CAAC,CAAC,MAAM,GAAG,EAAE;GACjF;EACF;EAIA,IAAI,CAAC,WAAW;GACd,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,SAAS,WAAW,OAAO,EAAE,cAAc,QAAQ;GACrF,IAAI,OAAO,SAAS,SAAS,YAAY,MAAM;EACjD;EAEA,MAAM,MAAM,GAAG,IAAI,aAAa,GAAG,YAAY,GAAG,GAAG,aAAa,GAAG,GAAG,aAAa;EACrF,IAAI,UAAU,SAAS,IAAI,GAAG;EAC9B,IAAI,CAAC,SAAS;GACZ,UAAU;IACR,cAAc,IAAI;IAClB;IACA;IACA;IACA,UAAU;IACV,aAAa,CAAC;IACd,cAAc,IAAI;IAClB,cAAc,kBAAkB,KAAK,KAAK,IAAI;GAChD;GACA,SAAS,IAAI,KAAK,OAAO;EAC3B;EACA,QAAQ;EACR,IAAI,CAAC,QAAQ,YAAY,SAAS,IAAI,UAAU,GAAG,QAAQ,YAAY,KAAK,IAAI,UAAU;CAC5F;CAMA,OAAO;EAAE,UAJG,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAC/B,QAAQ,MAAM,EAAE,YAAY,OAAO,CAAC,CACpC,MAAM,GAAG,MAAM,EAAE,WAAW,EAAE,QAEZ;EAAG;EAAe,WAAW,KAAK;CAAO;AAChE;AAEA,SAAS,kBAAkB,OAAmC;CAE5D,OADgB,MAAM,MAAM,MAAM,EAAE,WAAW,OAClC,CAAC,EAAE;AAClB;;;ACvFA,eAAsB,oBACpB,OACA,MACA,MACA,UAA6B,CAAC,GACH;CAC3B,MAAM,CAAC,GAAG,KAAK,MAAM,QAAQ,IAAI,CAAC,gBAAgB,OAAO,IAAI,GAAG,gBAAgB,OAAO,IAAI,CAAC,CAAC;CAC7F,MAAM,KAAK,QAAQ,cAAc;CACjC,MAAM,SAAS,KAAK,IAAI,EAAE,MAAM,QAAQ,EAAE,MAAM,MAAM;CACtD,KAAK,IAAI,IAAI,GAAG,IAAI,QAAQ,KAAK;EAC/B,MAAM,QAAQ,EAAE,MAAM;EACtB,MAAM,QAAQ,EAAE,MAAM;EACtB,IAAI,CAAC,GAAG,OAAO,KAAK,GAClB,OAAO;GACL;GACA;GACA,sBAAsB;GACtB;GACA;GACA,QAAQ,mBAAmB,OAAO,KAAK;GACvC,iBAAiB;EACnB;CAEJ;CACA,IAAI,EAAE,MAAM,WAAW,EAAE,MAAM,QAC7B,OAAO;EAAE;EAAM;EAAM,sBAAsB;EAAM,iBAAiB;CAAO;CAG3E,MAAM,SADqB,EAAE,MAAM,SAAS,EAAE,MAAM,SAAS,IAAI,EAAA,CAC5C,MAAM,SAAS;CAGpC,MAAM,SACJ,WAAW,IACP,0CAA0C,MAAM,gCAChD,sBAAsB,MAAM,4BAA4B,SAAS;CACvE,OAAO;EACL;EACA;EACA,sBAAsB;EACtB,OAAO,EAAE,MAAM;EACf,OAAO,EAAE,MAAM;EACf;EACA,iBAAiB;CACnB;AACF;AAEA,SAAS,kBAAkB,GAAmB,GAA4B;CACxE,IAAI,EAAE,KAAK,SAAS,EAAE,KAAK,MAAM,OAAO;CACxC,IAAI,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,SAAS,QAAQ,OAAO,EAAE,KAAK,aAAa,EAAE,KAAK;CACxF,IAAI,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,OAAO,OAAO,EAAE,KAAK,UAAU,EAAE,KAAK;CACnF,IAAI,EAAE,KAAK,SAAS,WAAW,EAAE,KAAK,SAAS,SAC7C,OAAO,EAAE,KAAK,cAAc,EAAE,KAAK;CACrC,OAAO,EAAE,KAAK,SAAS,EAAE,KAAK;AAChC;AAEA,SAAS,mBAAmB,GAAmB,GAA2B;CACxE,IAAI,EAAE,KAAK,SAAS,EAAE,KAAK,MAAM,OAAO,QAAQ,EAAE,KAAK,KAAK,MAAM,EAAE,KAAK;CACzE,IAAI,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,aAAa,EAAE,KAAK,UACjF,OAAO,QAAQ,EAAE,KAAK,SAAS,MAAM,EAAE,KAAK;CAE9C,IAAI,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,UAAU,EAAE,KAAK,OAC5E,OAAO,SAAS,EAAE,KAAK,MAAM,MAAM,EAAE,KAAK;CAE5C,OAAO,SAAS,EAAE,KAAK,KAAK,QAAQ,EAAE,KAAK,KAAK;AAClD;;;;;;;;;;;;AC9DA,eAAsB,mBAAmB,OAAkD;CACzF,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,MAAM,QAAQ,CAAC,EAAA,CAAG,QAChD,MAAsB,EAAE,SAAS,OACpC;CACA,IAAI,IAAI,WAAW,GAAG,OAAO;EAAE,OAAO,CAAC;EAAG,YAAY,CAAC;EAAG,UAAU,CAAC;CAAE;CAEvE,MAAM,8BAAc,IAAI,IAAyB;CACjD,KAAK,MAAM,KAAK,KAAK;EACnB,MAAM,MAAM,YAAY,IAAI,EAAE,SAAS,KAAK,CAAC;EAC7C,IAAI,KAAK,CAAC;EACV,YAAY,IAAI,EAAE,WAAW,GAAG;CAClC;CAEA,MAAM,WAAW,CAAC,GAAG,IAAI,IAAI,IAAI,KAAK,MAAM,EAAE,OAAO,CAAC,CAAC,CAAC,CAAC,KAAK;CAC9D,MAAM,QAAqB,CAAC;CAC5B,KAAK,MAAM,CAAC,KAAK,UAAU,aAAa;EACtC,MAAM,0BAAU,IAAI,IAAiC;EACrD,KAAK,MAAM,KAAK,OAAO;GACrB,MAAM,IAAI,QAAQ,IAAI,EAAE,OAAO,qBAAK,IAAI,IAAoB;GAC5D,EAAE,IAAI,EAAE,cAAc,EAAE,KAAK;GAC7B,QAAQ,IAAI,EAAE,SAAS,CAAC;EAC1B;EACA,MAAM,aAAa,CAAC,GAAG,QAAQ,KAAK,CAAC;EACrC,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,QAAQ,KACrC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,WAAW,QAAQ,KAAK;GAC9C,MAAM,SAAS,WAAW;GAC1B,MAAM,SAAS,WAAW;GAC1B,MAAM,IAAI,QAAQ,IAAI,MAAM;GAC5B,MAAM,IAAI,QAAQ,IAAI,MAAM;GAC5B,MAAM,SAAkC,CAAC;GACzC,KAAK,MAAM,CAAC,QAAQ,WAAW,GAAG;IAChC,MAAM,SAAS,EAAE,IAAI,MAAM;IAC3B,IAAI,WAAW,KAAA,GAAW,OAAO,KAAK,CAAC,QAAQ,MAAM,CAAC;GACxD;GACA,IAAI,OAAO,SAAS,GAAG;GACvB,MAAM,cAAc,OAAO,KACxB,CAAC,QAAQ,YACR,CACE;IAAE,WAAW;IAAQ,WAAW;IAAK,OAAO;IAAQ,WAAW;GAAG,GAClE;IAAE,WAAW;IAAQ,WAAW;IAAK,OAAO;IAAQ,WAAW;GAAG,CACpE,CACJ;GACA,MAAM,IAAI,sBACR,YAAY,EAAE,CAAE,KAAK,GAAG,OAAO,YAAY,KAAK,SAAS,KAAK,GAAI,CAAC,CACrE;GACA,MAAM,KAAK;IACT,QAAQ;IACR,QAAQ;IACR,WAAW;IACX,aAAa,OAAO;IACpB,SAAS,SACP,OAAO,KAAK,MAAM,EAAE,EAAE,GACtB,OAAO,KAAK,MAAM,EAAE,EAAE,CACxB;IACA,cAAc;GAChB,CAAC;EACH;CAEJ;CAEA,OAAO;EACL,OAAO,MAAM,MAAM,GAAG,MAAM,EAAE,cAAc,EAAE,WAAW;EACzD,YAAY,CAAC,GAAG,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK;EACzC;CACF;AACF;;;;;;;;;;;ACrEA,eAAsB,eACpB,OACA,SACA,SACyB;CACzB,MAAM,eAAe,MAAM,MAAM,SAAS,QAAQ,QAAQ;CAC1D,MAAM,gBAAgB,MAAM,MAAM,SAAS,QAAQ,SAAS;CAS5D,OAAO,kBAAkB,MARH,QAAQ,IAC5B,QAAQ,IAAI,OAAO,MAAM;EACvB,MAAM,UAAU,EAAE,WAAW,eAAe,EAAE,MAAM;EACpD,MAAM,WAAW,MAAM,WAAW,cAAc,SAAS,KAAK;EAC9D,MAAM,YAAY,MAAM,WAAW,eAAe,SAAS,KAAK;EAChE,OAAO;GAAE,QAAQ,EAAE;GAAQ,gBAAgB,EAAE;GAAgB;GAAU;EAAU;CACnF,CAAC,CACH,GACkC,OAAO;AAC3C;AAEA,eAAe,WACb,MACA,SACA,OACmB;CACnB,MAAM,MAAgB,CAAC;CACvB,KAAK,MAAM,KAAK,MAAM;EACpB,MAAM,IAAI,MAAM,QAAQ,GAAG,KAAK;EAChC,IAAI,MAAM,QAAQ,OAAO,SAAS,CAAC,GAAG,IAAI,KAAK,CAAC;CAClD;CACA,OAAO;AACT;AAEA,SAAS,eAAe,QAAyE;CAC/F,OAAO,OAAO,KAAK,UAAU;EAC3B,QAAQ,QAAR;GACE,KAAK;GACL,KAAK,gBACH,OAAO,IAAI,SAAS,SAAS;GAC/B,KAAK,QACH,OAAO,IAAI,SAAS,SAAS,OAAO,IAAI;GAC1C,KAAK,cACH,OAAO,IAAI,WAAW,IAAI,YAAY,IAAI,UAAU,IAAI,YAAY;GACtE,KAAK,WAEH,OAAO,aAAa,MADF,SAAS,OAAO,IAAI,KAAK,CACpB,CAAC,CAAC;GAE3B,KAAK,eAEH,OAAO,aAAa,MADF,SAAS,OAAO,IAAI,KAAK,CACpB,CAAC,CAAC;GAE3B,KAAK,gBAEH,OAAO,aAAa,MADF,SAAS,OAAO,IAAI,KAAK,CACpB,CAAC,CAAC;GAE3B,KAAK,gBACH,OAAO,gBAAgB,GAAG,MAAM,YAAY,IAAI;GAElD,SACE,OAAO;EACX;CACF;AACF;;;;;;;;;;;;ACvEA,MAAM,wBAAwB;AAC9B,MAAM,qCAAqC;AA8C3C,eAAsB,cACpB,OACA,UAA4B,CAAC,GACH;CAC1B,MAAM,iBAAiB,QAAQ,kBAAkB;CACjD,MAAM,cAAc,QAAQ,eAAe;CAC3C,MAAM,0BACJ,QAAQ,2BAA2B;CACrC,IAAI,CAAC,OAAO,UAAU,cAAc,KAAK,iBAAiB,GACxD,MAAM,IAAI,WAAW,2CAA2C;CAElE,IAAI,CAAC,OAAO,SAAS,WAAW,KAAK,cAAc,GACjD,MAAM,IAAI,WAAW,kDAAkD;CAEzE,IAAI,CAAC,OAAO,UAAU,uBAAuB,KAAK,0BAA0B,GAC1E,MAAM,IAAI,WAAW,wDAAwD;CAE/E,MAAM,OAAO,QAAQ,QACjB,CAAC,EAAE,OAAO,QAAQ,MAAM,CAAC,KACxB,MAAM,MAAM,SAAS,EAAA,CAAG,KAAK,OAAO,EAAE,OAAO,EAAE,MAAM,EAAE;CAE5D,MAAM,WAA+B,CAAC;CACtC,KAAK,MAAM,EAAE,WAAW,MAAM;EAC5B,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,MAAM,CAAC;EACzC,MAAM,YAAY,IAAI,IAAI,MAAM,KAAK,SAAS,CAAC,KAAK,QAAQ,IAAI,CAAC,CAAC;EAClE,MAAM,cAAgC,MACnC,OAAO,UAAU,CAAC,CAClB,KAAK,MAAM,iBAAiB;GAAE;GAAM;EAAY,EAAE,CAAC,CACnD,MAAM,GAAG,MAAM,EAAE,KAAK,YAAY,EAAE,KAAK,aAAa,EAAE,cAAc,EAAE,WAAW,CAAC,CACpF,KAAK,EAAE,YAAY;GAAE;GAAM,GAAG,eAAe,MAAM,SAAS;EAAE,EAAE;EACnE,MAAM,cAAc,qBAClB,YAAY,KAAK,SAAS;GAExB,MAAM,QADS,KAAK,eAAe,QAAQ,KAAK,eAAe,KAAK,cAEhE,KAAK,OACL,KAAK,aACH,UAAU,IAAI,KAAK,UAAU,IAC7B,KAAA;GACN,OAAO;IACL,KAAK,aAAa,IAAI;IACtB,UAAU,KAAK,UAAU,KAAK,WAAW;IACzC,OAAO,OAAO,aAAa;IAC3B,KAAK,OAAO,WAAW;GACzB;EACF,CAAC,CACH;EACA,MAAM,uCAAuB,IAAI,IAAoB;EACrD,MAAM,eAAkC,YAAY,KAAK,SAAS;GAChE,MAAM,UAAU,YAAY,IAAI,aAAa,IAAI,CAAC;GAClD,MAAM,gBAAgB,qBAAqB,IAAI,OAAO,KAAK;GAC3D,qBAAqB,IAAI,SAAS,gBAAgB,CAAC;GACnD,OAAO;IAAE,GAAG;IAAM;IAAe;GAAQ;EAC3C,CAAC;EACD,MAAM,wBAAQ,IAAI,IAQhB;EACF,KAAK,MAAM,QAAQ,cAAc;GAC/B,IAAI,CAAC,oBAAoB,KAAK,IAAI,GAAG;GACrC,MAAM,IAAI,QAAQ,KAAK,KAAK,IAAI;GAChC,MAAM,MAAM,KAAK,UAAU;IAAC,KAAK;IAAS,KAAK,KAAK;IAAU;GAAC,CAAC;GAChE,MAAM,SAAS,MAAM,IAAI,GAAG,KAAK;IAC/B,OAAO,CAAC;IACR,SAAS;IACT,UAAU,KAAK,KAAK;IACpB,aAAa,KAAK;GACpB;GACA,OAAO,MAAM,KAAK,IAAI;GACtB,MAAM,IAAI,KAAK,MAAM;EACvB;EACA,KAAK,MAAM,EAAE,OAAO,SAAS,GAAG,UAAU,iBAAiB,MAAM,OAAO,GAAG;GACzE,IAAI,MAAM,SAAS,gBAAgB;GACnC,IAAI,eAAe;GACnB,KAAK,IAAI,aAAa,GAAG,cAAc,MAAM,QAAQ,cAAc,GAAG;IACpE,MAAM,WAAW,MAAM,aAAa;IACpC,MAAM,OAAO,MAAM;IAMnB,IAAI,EAJF,SAAS,KAAA,KACT,KAAK,KAAK,YAAY,SAAS,KAAK,YAAY,eAChD,KAAK,gBAAgB,SAAS,gBAAgB,IAAI,2BAClD,CAAC,eAAe,UAAU,IAAI,IACb;IAEnB,MAAM,UAAU,MAAM,MAAM,cAAc,UAAU;IACpD,IAAI,OAAO;IACX,IAAI,YAAY;IAChB,IAAI,UAAU;IACd,KAAK,IAAI,QAAQ,GAAG,QAAQ,QAAQ,QAAQ,SAAS,GAAG;KACtD,OAAO,QAAQ,MAAM,CAAE,KAAK,YAAY,QAAQ,KAAK,CAAE,KAAK,YAAY,aACtE,QAAQ;KAEV,IAAI,QAAQ,OAAO,UAAU,WAAW;MACtC,YAAY;MACZ,UAAU;KACZ;IACF;IACA,IAAI,UAAU,YAAY,KAAK,gBAAgB;KAC7C,MAAM,OAAO,QAAQ,MAAM,WAAW,UAAU,CAAC;KACjD,MAAM,QAAQ,KAAK,EAAE,CAAE,KAAK;KAC5B,MAAM,OAAO,KAAK,KAAK,SAAS,EAAE,CAAE,KAAK;KACzC,SAAS,KAAK;MACZ;MACA;MACA,SAAS;MACT,aAAa,KAAK;MAClB,SAAS,KAAK,KAAK,SAAS,KAAK,KAAK,MAAM;MAC5C,GAAI,cAAc,EAAE,YAAY,IAAI,CAAC;MACrC,UAAU,OAAO;KACnB,CAAC;IACH;IACA,eAAe;GACjB;EACF;CACF;CAEA,MAAM,eAAe,IAAI,IAAI,SAAS,KAAK,MAAM,EAAE,KAAK,CAAC;CACzD,OAAO;EACL;EACA,kBAAkB,KAAK,SAAS,IAAI,aAAa,OAAO,KAAK,SAAS;EACtE,WAAW,KAAK;CAClB;AACF;AAEA,SAAS,QAAQ,MAAkE;CACjF,OAAO,KAAK,UAAU,CAAC,KAAK,aAAa,KAAK,UAAU,CAAC;AAC3D;AAEA,SAAS,aAAa,MAA8B;CAClD,OAAO,KAAK,eAAe,QAAQ,KAAK,eAAe,KAAK,cACxD,KAAK,UAAU,CAAC,KAAK,aAAa,KAAK,KAAK,MAAM,CAAC,IACnD,QAAQ,IAAI;AAClB;AAEA,SAAS,eACP,MACA,WAC2D;CAC3D,MAAM,eAAe,KAAK;CAC1B,IAAI,CAAC,cAAc,OAAO;EAAE,aAAa;EAAM,YAAY;CAAK;CAEhE,IAAI,YAAgC;CACpC,IAAI,aAA4B;CAChC,MAAM,uBAAO,IAAI,IAAY;CAC7B,OAAO,aAAa,CAAC,KAAK,IAAI,SAAS,GAAG;EACxC,KAAK,IAAI,SAAS;EAClB,MAAM,UAAU,UAAU,IAAI,SAAS;EACvC,IAAI,CAAC,SAAS,OAAO;GAAE,aAAa;GAAc,YAAY;EAAa;EAC3E,IAAI,QAAQ,SAAS,SACnB,OAAO;GAAE,aAAa,QAAQ;GAAQ,YAAY,cAAc,QAAQ;EAAO;EAEjF,aAAa,QAAQ;EACrB,YAAY,QAAQ;CACtB;CACA,OAAO;EAAE,aAAa;EAAc,YAAY;CAAa;AAC/D;AAEA,SAAS,eAAe,UAA2B,MAAgC;CACjF,OAAO,SAAS,KAAK,YAAY,KAAA,KAAa,SAAS,KAAK,WAAW,KAAK,KAAK;AACnF;;;;;;;;;;;;;;AC/LA,eAAsB,cACpB,OACA,UAA4B,CAAC,GACH;CAC1B,MAAM,OAAO,QAAQ,QAAQ,CAAC,QAAQ,KAAK,KAAK,MAAM,MAAM,SAAS,EAAA,CAAG,KAAK,MAAM,EAAE,KAAK;CAE1F,MAAM,QAA4B,CAAC;CACnC,IAAI,aAAa;CACjB,IAAI,cAAc;CAClB,KAAK,MAAM,SAAS,MAAM;EACxB,MAAM,QAAQ,MAAM,UAAU,OAAO,KAAK;EAC1C,IAAI,MAAM,WAAW,GAAG;GACtB,MAAM,KAAK;IAAE;IAAO,aAAa;IAAG,YAAY;IAAG,WAAW;GAAE,CAAC;GACjE;EACF;EAQA,MAAM,YAAY,CAAC,GAAG,MAPH,SAAS,OAAO,KAAK,CAOd,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;EACpE,MAAM,aAAa,UAAU,KAAK,MAAM,EAAE,SAAS;EACnD,MAAM,aAAa,gBAAgB,SAAS;EAC5C,IAAI,SAAS;EACb,KAAK,MAAM,KAAK,OAAO;GACrB,IAAI,EAAE,WAAW,SAAS;IACxB;IACA;GACF;GAEA,MAAM,SAAS,WAAW,YAAY,EAAE,SAAS;GACjD,IAAI,QAAQ,aACN;QAAA,CAAC,QAAQ,YAAY,GAAG,EAAE,KAAK,UAAU,MAAM,MAAM,EAAE,CAAC,GAAG;GAAA,OAC1D;IAKL,MAAM,YAAY,UAAU,EAAE,MAAM;IACpC,IAAI,cAAc,IAAI;IAGtB,IAAI,EAFa,WAAW,WAAW,GAAA,CACjB,SAAS,UAAU,MAAM,GAAG,GAAG,CAC7C,GAAG;GACb;EACF;EACA,MAAM,YAAY,SAAS,MAAM;EACjC,MAAM,KAAK;GAAE;GAAO,aAAa;GAAQ,YAAY,MAAM;GAAQ;EAAU,CAAC;EAC9E,cAAc,MAAM;EACpB,eAAe;CACjB;CACA,OAAO;EAAE;EAAO,kBAAkB,aAAa,IAAI,cAAc,aAAa;CAAE;AAClF;;;;;;AAOA,SAAS,gBAAgB,OAA4B;CACnD,MAAM,SAAS,IAAI,MAAc,MAAM,SAAS,CAAC;CACjD,OAAO,MAAM,UAAU;CACvB,KAAK,IAAI,IAAI,MAAM,SAAS,GAAG,KAAK,GAAG,KAIrC,OAAO,KAAK,GAHA,MAAM,EAAE,CAAE,SAAS,KAAK,MAClC,OAAO,EAAE,YAAY,WAAW,EAAE,UAAU,EAC9C,CAAC,CAAC,KAAK,IACU,EAAE,IAAI,OAAO,IAAI;CAEpC,OAAO;AACT;;AAGA,SAAS,WAAW,QAAkB,QAAwB;CAC5D,IAAI,KAAK;CACT,IAAI,KAAK,OAAO;CAChB,OAAO,KAAK,IAAI;EACd,MAAM,MAAO,KAAK,OAAQ;EAC1B,IAAI,OAAO,QAAS,QAAQ,KAAK,MAAM;OAClC,KAAK;CACZ;CACA,OAAO;AACT;AAEA,SAAS,UAAU,GAAoB;CACrC,IAAI,MAAM,QAAQ,MAAM,KAAA,GAAW,OAAO;CAC1C,IAAI,OAAO,MAAM,UAAU,OAAO;CAClC,IAAI;EACF,OAAO,KAAK,UAAU,CAAC;CACzB,QAAQ;EACN,OAAO,OAAO,CAAC;CACjB;AACF"}
1
+ {"version":3,"file":"index.js","names":[],"sources":["../../src/pipelines/budget-breach.ts","../../src/pipelines/failure-cluster.ts","../../src/pipelines/first-divergence.ts","../../src/pipelines/judge-agreement.ts","../../src/pipelines/regression.ts","../../src/pipelines/stuck-loop.ts","../../src/pipelines/tool-waste.ts"],"sourcesContent":["/**\n * BudgetBreachView — aggregates breach events across the corpus.\n *\n * Answers: which dimensions get hit most often? Which scenarios are\n * underbudgeted? Which variants trigger the most breaches?\n */\n\nimport type { BudgetSpec } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface BudgetBreachFinding {\n runId: string\n scenarioId: string\n variantId?: string\n dimension: keyof BudgetSpec\n limit: number\n consumed: number\n excessRatio: number\n timestamp: number\n}\n\nexport interface BudgetBreachReport {\n findings: BudgetBreachFinding[]\n byDimension: Record<string, number>\n byScenario: Record<string, number>\n byVariant: Record<string, number>\n totalRuns: number\n breachedRunRatio: number\n}\n\nexport async function budgetBreachView(\n store: TraceStore,\n options: { scenarioId?: string; variantId?: string } = {},\n): Promise<BudgetBreachReport> {\n const runs = await store.listRuns({\n scenarioId: options.scenarioId,\n variantId: options.variantId,\n })\n const findings: BudgetBreachFinding[] = []\n const byDimension: Record<string, number> = {}\n const byScenario: Record<string, number> = {}\n const byVariant: Record<string, number> = {}\n\n for (const run of runs) {\n const entries = await store.budget(run.runId)\n for (const e of entries) {\n if (!e.breached) continue\n const excessRatio = e.limit > 0 ? e.consumed / e.limit : Infinity\n findings.push({\n runId: run.runId,\n scenarioId: run.scenarioId,\n variantId: run.variantId,\n dimension: e.dimension,\n limit: e.limit,\n consumed: e.consumed,\n excessRatio,\n timestamp: e.timestamp,\n })\n byDimension[e.dimension] = (byDimension[e.dimension] ?? 0) + 1\n byScenario[run.scenarioId] = (byScenario[run.scenarioId] ?? 0) + 1\n if (run.variantId) byVariant[run.variantId] = (byVariant[run.variantId] ?? 0) + 1\n }\n }\n\n const breachedRuns = new Set(findings.map((f) => f.runId))\n return {\n findings,\n byDimension,\n byScenario,\n byVariant,\n totalRuns: runs.length,\n breachedRunRatio: runs.length > 0 ? breachedRuns.size / runs.length : 0,\n }\n}\n","/**\n * FailureClusterView — groups failed runs by (failureClass, triggerTool,\n * argHash-prefix) so weekly reviews can prioritize the top-N clusters.\n *\n * Each cluster includes: N runs, scenarios affected, representative\n * error message, a proposed mitigation hint (rule → action table).\n */\n\nimport { classifyFailure, DEFAULT_RULES, type FailureRule } from '../failure-taxonomy'\nimport { argHash, hasCapturedToolArgs, toolSpans } from '../trace/query'\nimport type { FailureClass, Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface FailureCluster {\n failureClass: FailureClass\n /** Tool name when the trigger was a tool span, else undefined. */\n toolName?: string\n /** First 16 chars of argHash — clusters similar args. */\n argPrefix?: string\n /**\n * Source dimension when the trigger was a judge span (e.g. `'format'`,\n * `'safety'`, `'correctness'`). Lets cross-template aggregators\n * group failures by the dimension that fired without overloading\n * `argPrefix`. Optional — clusters without this field deserialize cleanly.\n */\n dimension?: string\n runCount: number\n scenarioIds: string[]\n exampleError?: string\n exampleRunId: string\n}\n\nexport interface FailureClusterReport {\n clusters: FailureCluster[]\n totalFailures: number\n totalRuns: number\n}\n\nexport async function failureClusterView(\n store: TraceStore,\n options: { rules?: FailureRule[]; minClusterSize?: number } = {},\n): Promise<FailureClusterReport> {\n const rules = options.rules ?? DEFAULT_RULES\n const minSize = options.minClusterSize ?? 1\n const runs = await store.listRuns()\n\n type Key = string\n const clusters = new Map<Key, FailureCluster>()\n let totalFailures = 0\n\n for (const run of runs) {\n if (run.status === 'completed' && run.outcome?.pass !== false) continue\n totalFailures++\n const spans = await store.spans({ runId: run.runId })\n const events = await store.events({ runId: run.runId })\n const cls = classifyFailure({ run, spans, events }, rules)\n\n let toolName: string | undefined\n let argPrefix: string | undefined\n let dimension: string | undefined\n if (cls.triggerSpanId) {\n const trig = spans.find((s) => s.spanId === cls.triggerSpanId)\n if (trig?.kind === 'tool') {\n toolName = trig.toolName\n if (hasCapturedToolArgs(trig)) argPrefix = argHash(trig.args).slice(0, 16)\n } else if (trig?.kind === 'judge') {\n dimension = trig.dimension\n }\n }\n // Fallback: look at the last errored tool span\n if (!toolName) {\n const ts = await toolSpans(store, run.runId)\n const errored = ts.filter((t) => t.status === 'error').pop()\n if (errored) {\n toolName = errored.toolName\n if (hasCapturedToolArgs(errored)) argPrefix = argHash(errored.args).slice(0, 16)\n }\n }\n // Secondary signal: any judge span on the failed run carries a\n // dimension. Useful when the rule classified by judge score but\n // didn't surface the trigger span (or surfaced a non-judge span).\n if (!dimension) {\n const judge = spans.find((s) => s.kind === 'judge' && typeof s.dimension === 'string')\n if (judge?.kind === 'judge') dimension = judge.dimension\n }\n\n const key = `${cls.failureClass}|${toolName ?? ''}|${argPrefix ?? ''}|${dimension ?? ''}`\n let cluster = clusters.get(key)\n if (!cluster) {\n cluster = {\n failureClass: cls.failureClass,\n toolName,\n argPrefix,\n dimension,\n runCount: 0,\n scenarioIds: [],\n exampleRunId: run.runId,\n exampleError: firstErrorMessage(spans) ?? cls.reason,\n }\n clusters.set(key, cluster)\n }\n cluster.runCount++\n if (!cluster.scenarioIds.includes(run.scenarioId)) cluster.scenarioIds.push(run.scenarioId)\n }\n\n const arr = [...clusters.values()]\n .filter((c) => c.runCount >= minSize)\n .sort((a, b) => b.runCount - a.runCount)\n\n return { clusters: arr, totalFailures, totalRuns: runs.length }\n}\n\nfunction firstErrorMessage(spans: Span[]): string | undefined {\n const errored = spans.find((s) => s.status === 'error')\n return errored?.error\n}\n","/**\n * FirstDivergenceView — aligns two trajectories by step index, reports\n * the first step where they differ.\n *\n * \"Differ\" is configurable — default is (kind, toolName if tool, model\n * if llm). Use this view to attribute \"why is variant B better?\" to a\n * specific step rather than an aggregate mean delta.\n */\n\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\nexport interface DivergenceReport {\n runA: string\n runB: string\n firstDivergenceIndex: number | null\n aStep?: TrajectoryStep\n bStep?: TrajectoryStep\n reason?: string\n /** Common prefix length (steps that matched). */\n commonPrefixLen: number\n}\n\nexport interface DivergenceOptions {\n /** Returns true if two steps are considered equal. Default: kind + tool/model match. */\n stepEquals?: (a: TrajectoryStep, b: TrajectoryStep) => boolean\n}\n\nexport async function firstDivergenceView(\n store: TraceStore,\n runA: string,\n runB: string,\n options: DivergenceOptions = {},\n): Promise<DivergenceReport> {\n const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)])\n const eq = options.stepEquals ?? defaultStepEquals\n const minLen = Math.min(a.steps.length, b.steps.length)\n for (let i = 0; i < minLen; i++) {\n const aStep = a.steps[i]!\n const bStep = b.steps[i]!\n if (!eq(aStep, bStep)) {\n return {\n runA,\n runB,\n firstDivergenceIndex: i,\n aStep,\n bStep,\n reason: describeDifference(aStep, bStep),\n commonPrefixLen: i,\n }\n }\n }\n if (a.steps.length === b.steps.length) {\n return { runA, runB, firstDivergenceIndex: null, commonPrefixLen: minLen }\n }\n const longer: Trajectory = a.steps.length > b.steps.length ? a : b\n const extra = longer.steps.length - minLen\n // minLen === 0 means one trajectory is empty: divergence is the first step\n // itself (index 0), and the absent side has no step to surface.\n const reason =\n minLen === 0\n ? `one trajectory is empty; the other has ${extra} step(s) starting at index 0`\n : `one trajectory has ${extra} more step(s) after index ${minLen - 1}`\n return {\n runA,\n runB,\n firstDivergenceIndex: minLen,\n aStep: a.steps[minLen],\n bStep: b.steps[minLen],\n reason,\n commonPrefixLen: minLen,\n }\n}\n\nfunction defaultStepEquals(a: TrajectoryStep, b: TrajectoryStep): boolean {\n if (a.span.kind !== b.span.kind) return false\n if (a.span.kind === 'tool' && b.span.kind === 'tool') return a.span.toolName === b.span.toolName\n if (a.span.kind === 'llm' && b.span.kind === 'llm') return a.span.model === b.span.model\n if (a.span.kind === 'judge' && b.span.kind === 'judge')\n return a.span.dimension === b.span.dimension\n return a.span.name === b.span.name\n}\n\nfunction describeDifference(a: TrajectoryStep, b: TrajectoryStep): string {\n if (a.span.kind !== b.span.kind) return `kind ${a.span.kind} vs ${b.span.kind}`\n if (a.span.kind === 'tool' && b.span.kind === 'tool' && a.span.toolName !== b.span.toolName) {\n return `tool ${a.span.toolName} vs ${b.span.toolName}`\n }\n if (a.span.kind === 'llm' && b.span.kind === 'llm' && a.span.model !== b.span.model) {\n return `model ${a.span.model} vs ${b.span.model}`\n }\n return `name \"${a.span.name}\" vs \"${b.span.name}\"`\n}\n","/**\n * JudgeAgreementView — pairwise agreement between judges across the\n * corpus, grouped by dimension.\n *\n * Output drives two workflows:\n * - Judge robustness audit: \"does Claude agree with GPT at κ ≥ 0.6?\"\n * - Calibration tracking: κ vs golden human labels over time (by\n * providing a `humanGoldenJudgeId`).\n */\n\nimport { interRaterReliability, pearsonR } from '../statistics'\nimport type { JudgeSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface JudgePair {\n judgeA: string\n judgeB: string\n dimension: string\n /** Number of (targetSpanId, dimension) tuples both judges scored. */\n commonItems: number\n pearson: number\n krippendorff: number\n}\n\nexport interface JudgeAgreementReport {\n pairs: JudgePair[]\n dimensions: string[]\n judgeIds: string[]\n}\n\nexport async function judgeAgreementView(store: TraceStore): Promise<JudgeAgreementReport> {\n const all = (await store.spans({ kind: 'judge' })).filter(\n (s): s is JudgeSpan => s.kind === 'judge',\n )\n if (all.length === 0) return { pairs: [], dimensions: [], judgeIds: [] }\n\n const byDimension = new Map<string, JudgeSpan[]>()\n for (const s of all) {\n const arr = byDimension.get(s.dimension) ?? []\n arr.push(s)\n byDimension.set(s.dimension, arr)\n }\n\n const judgeIds = [...new Set(all.map((s) => s.judgeId))].sort()\n const pairs: JudgePair[] = []\n for (const [dim, spans] of byDimension) {\n const byJudge = new Map<string, Map<string, number>>()\n for (const s of spans) {\n const m = byJudge.get(s.judgeId) ?? new Map<string, number>()\n m.set(s.targetSpanId, s.score)\n byJudge.set(s.judgeId, m)\n }\n const judgesHere = [...byJudge.keys()]\n for (let i = 0; i < judgesHere.length; i++) {\n for (let j = i + 1; j < judgesHere.length; j++) {\n const judgeI = judgesHere[i]!\n const judgeJ = judgesHere[j]!\n const a = byJudge.get(judgeI)!\n const b = byJudge.get(judgeJ)!\n const common: Array<[number, number]> = []\n for (const [target, scoreA] of a) {\n const scoreB = b.get(target)\n if (scoreB !== undefined) common.push([scoreA, scoreB])\n }\n if (common.length < 2) continue\n const judgeScores = common.map(\n ([scoreA, scoreB]) =>\n [\n { judgeName: judgeI, dimension: dim, score: scoreA, reasoning: '' },\n { judgeName: judgeJ, dimension: dim, score: scoreB, reasoning: '' },\n ] as const,\n )\n const k = interRaterReliability(\n judgeScores[0]!.map((_, k2) => judgeScores.map((pair) => pair[k2]!)),\n )\n pairs.push({\n judgeA: judgeI,\n judgeB: judgeJ,\n dimension: dim,\n commonItems: common.length,\n pearson: pearsonR(\n common.map((c) => c[0]),\n common.map((c) => c[1]),\n ),\n krippendorff: k,\n })\n }\n }\n }\n\n return {\n pairs: pairs.sort((a, b) => b.commonItems - a.commonItems),\n dimensions: [...byDimension.keys()].sort(),\n judgeIds,\n }\n}\n","/**\n * RegressionView — compares a candidate slice to a baseline slice on a\n * named metric. Delegates the statistics (Welch's t-test, Cohen's d,\n * IQR stability) to `baseline.ts`.\n *\n * This is the entry point for CI regression gates: \"given runs tagged\n * release=A and release=B, did any metric regress?\"\n */\n\nimport { type BaselineOptions, type BaselineReport, compareToBaseline } from '../baseline'\nimport { aggregateLlm, llmSpans, runFailureClass } from '../trace/query'\nimport type { Run } from '../trace/schema'\nimport type { RunFilter, TraceStore } from '../trace/store'\n\nexport interface RegressionSpec {\n metric: string\n higherIsBetter: boolean\n /** Extract a scalar from a run. Default extractors handle common metrics. */\n extract?: (run: Run, store: TraceStore) => Promise<number | null>\n}\n\nexport interface RegressionOptions extends BaselineOptions {\n baseline: RunFilter\n candidate: RunFilter\n}\n\nexport async function regressionView(\n store: TraceStore,\n metrics: RegressionSpec[],\n options: RegressionOptions,\n): Promise<BaselineReport> {\n const baselineRuns = await store.listRuns(options.baseline)\n const candidateRuns = await store.listRuns(options.candidate)\n const samples = await Promise.all(\n metrics.map(async (m) => {\n const extract = m.extract ?? defaultExtract(m.metric)\n const baseline = await extractAll(baselineRuns, extract, store)\n const candidate = await extractAll(candidateRuns, extract, store)\n return { metric: m.metric, higherIsBetter: m.higherIsBetter, baseline, candidate }\n }),\n )\n return compareToBaseline(samples, options)\n}\n\nasync function extractAll(\n runs: Run[],\n extract: (r: Run, s: TraceStore) => Promise<number | null>,\n store: TraceStore,\n): Promise<number[]> {\n const out: number[] = []\n for (const r of runs) {\n const v = await extract(r, store)\n if (v !== null && Number.isFinite(v)) out.push(v)\n }\n return out\n}\n\nfunction defaultExtract(metric: string): (run: Run, store: TraceStore) => Promise<number | null> {\n return async (run, store) => {\n switch (metric) {\n case 'score':\n case 'overallScore':\n return run.outcome?.score ?? null\n case 'pass':\n return run.outcome?.pass === true ? 1 : 0\n case 'durationMs':\n return run.endedAt && run.startedAt ? run.endedAt - run.startedAt : null\n case 'costUsd': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).costUsd\n }\n case 'inputTokens': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).inputTokens\n }\n case 'outputTokens': {\n const llm = await llmSpans(store, run.runId)\n return aggregateLlm(llm).outputTokens\n }\n case 'failureClass': {\n return runFailureClass(run) === 'success' ? 1 : 0\n }\n default:\n return null\n }\n }\n}\n","/**\n * StuckLoopView — detects when an agent calls the same tool with the\n * same (or structurally similar) arguments ≥ N times in a short window.\n *\n * Rationale: agents that loop are the number-one production failure\n * mode on long-horizon flows. The view returns (runId, toolName,\n * argHash, occurrences, windowMs) for each detected loop plus a\n * fraction of runs affected.\n */\n\nimport { executionTrackByLane } from '../trace/execution-tracks'\nimport { argHash, hasCapturedToolArgs } from '../trace/query'\nimport { isToolSpan, type Span, type ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nconst DEFAULT_MAX_WINDOW_MS = 60_000\nconst DEFAULT_MAX_INTERVENING_TOOL_CALLS = 0\n\ninterface ScopedToolCall {\n span: ToolSpan\n scopeSpanId: string | null\n laneSpanId: string | null\n}\n\ninterface IndexedToolCall extends ScopedToolCall {\n toolCallIndex: number\n trackId: string\n}\n\nexport interface StuckLoopFinding {\n runId: string\n toolName: string\n argHash: string\n /** Calls in this episode's densest qualifying interval, not the whole-run total. */\n occurrences: number\n spanIds: string[]\n /** Nearest agent ancestor, or the direct parent when ancestry is incomplete. */\n scopeSpanId?: string\n /** Milliseconds between first and last call in the loop. */\n windowMs: number\n}\n\nexport interface StuckLoopReport {\n findings: StuckLoopFinding[]\n affectedRunRatio: number\n totalRuns: number\n}\n\nexport interface StuckLoopOptions {\n /** Minimum call count to flag a loop (default 3). */\n minOccurrences?: number\n /** Maximum time between the first and last repeated call (default 60 seconds). */\n maxWindowMs?: number\n /**\n * Maximum other tool calls allowed between adjacent repeats (default 0).\n * Set to 1 to detect alternating patterns such as A,B,A,B,A.\n */\n maxInterveningToolCalls?: number\n /** Filter to a specific runId; omit to scan the entire corpus. */\n runId?: string\n}\n\nexport async function stuckLoopView(\n store: TraceStore,\n options: StuckLoopOptions = {},\n): Promise<StuckLoopReport> {\n const minOccurrences = options.minOccurrences ?? 3\n const maxWindowMs = options.maxWindowMs ?? DEFAULT_MAX_WINDOW_MS\n const maxInterveningToolCalls =\n options.maxInterveningToolCalls ?? DEFAULT_MAX_INTERVENING_TOOL_CALLS\n if (!Number.isInteger(minOccurrences) || minOccurrences < 1) {\n throw new RangeError('minOccurrences must be a positive integer')\n }\n if (!Number.isFinite(maxWindowMs) || maxWindowMs < 0) {\n throw new RangeError('maxWindowMs must be a finite non-negative number')\n }\n if (!Number.isInteger(maxInterveningToolCalls) || maxInterveningToolCalls < 0) {\n throw new RangeError('maxInterveningToolCalls must be a non-negative integer')\n }\n const runs = options.runId\n ? [{ runId: options.runId }]\n : (await store.listRuns()).map((r) => ({ runId: r.runId }))\n\n const findings: StuckLoopFinding[] = []\n for (const { runId } of runs) {\n const spans = await store.spans({ runId })\n const spansById = new Map(spans.map((span) => [span.spanId, span]))\n const scopedTools: ScopedToolCall[] = spans\n .filter(isToolSpan)\n .map((span, sourceIndex) => ({ span, sourceIndex }))\n .sort((a, b) => a.span.startedAt - b.span.startedAt || a.sourceIndex - b.sourceIndex)\n .map(({ span }) => ({ span, ...executionScope(span, spansById) }))\n const trackByLane = executionTrackByLane(\n scopedTools.map((call) => {\n const direct = call.laneSpanId === null || call.laneSpanId === call.scopeSpanId\n const timed = direct\n ? call.span\n : call.laneSpanId\n ? spansById.get(call.laneSpanId)\n : undefined\n return {\n key: executionKey(call),\n scopeKey: JSON.stringify(call.scopeSpanId),\n start: timed?.startedAt ?? null,\n end: timed?.endedAt ?? null,\n }\n }),\n )\n const nextToolIndexByTrack = new Map<string, number>()\n const orderedTools: IndexedToolCall[] = scopedTools.map((call) => {\n const trackId = trackByLane.get(executionKey(call))!\n const toolCallIndex = nextToolIndexByTrack.get(trackId) ?? 0\n nextToolIndexByTrack.set(trackId, toolCallIndex + 1)\n return { ...call, toolCallIndex, trackId }\n })\n const byKey = new Map<\n string,\n {\n calls: IndexedToolCall[]\n argHash: string\n toolName: string\n scopeSpanId: string | null\n }\n >()\n for (const call of orderedTools) {\n if (!hasCapturedToolArgs(call.span)) continue\n const h = argHash(call.span.args)\n const key = JSON.stringify([call.trackId, call.span.toolName, h])\n const bucket = byKey.get(key) ?? {\n calls: [],\n argHash: h,\n toolName: call.span.toolName,\n scopeSpanId: call.scopeSpanId,\n }\n bucket.calls.push(call)\n byKey.set(key, bucket)\n }\n for (const { calls, argHash: h, toolName, scopeSpanId } of byKey.values()) {\n if (calls.length < minOccurrences) continue\n let episodeStart = 0\n for (let episodeEnd = 1; episodeEnd <= calls.length; episodeEnd += 1) {\n const previous = calls[episodeEnd - 1]!\n const next = calls[episodeEnd]\n const episodeEnded =\n next === undefined ||\n next.span.startedAt - previous.span.startedAt > maxWindowMs ||\n next.toolCallIndex - previous.toolCallIndex - 1 > maxInterveningToolCalls ||\n !callsAreSerial(previous, next)\n if (!episodeEnded) continue\n\n const episode = calls.slice(episodeStart, episodeEnd)\n let left = 0\n let bestStart = 0\n let bestEnd = -1\n for (let right = 0; right < episode.length; right += 1) {\n while (episode[right]!.span.startedAt - episode[left]!.span.startedAt > maxWindowMs) {\n left += 1\n }\n if (right - left > bestEnd - bestStart) {\n bestStart = left\n bestEnd = right\n }\n }\n if (bestEnd - bestStart + 1 >= minOccurrences) {\n const loop = episode.slice(bestStart, bestEnd + 1)\n const first = loop[0]!.span.startedAt\n const last = loop[loop.length - 1]!.span.startedAt\n findings.push({\n runId,\n toolName,\n argHash: h,\n occurrences: loop.length,\n spanIds: loop.map((call) => call.span.spanId),\n ...(scopeSpanId ? { scopeSpanId } : {}),\n windowMs: last - first,\n })\n }\n episodeStart = episodeEnd\n }\n }\n }\n\n const affectedRuns = new Set(findings.map((f) => f.runId))\n return {\n findings,\n affectedRunRatio: runs.length > 0 ? affectedRuns.size / runs.length : 0,\n totalRuns: runs.length,\n }\n}\n\nfunction laneKey(call: Pick<ScopedToolCall, 'scopeSpanId' | 'laneSpanId'>): string {\n return JSON.stringify([call.scopeSpanId, call.laneSpanId])\n}\n\nfunction executionKey(call: ScopedToolCall): string {\n return call.laneSpanId === null || call.laneSpanId === call.scopeSpanId\n ? JSON.stringify([call.scopeSpanId, call.span.spanId])\n : laneKey(call)\n}\n\nfunction executionScope(\n span: ToolSpan,\n spansById: ReadonlyMap<string, Span>,\n): { scopeSpanId: string | null; laneSpanId: string | null } {\n const directParent = span.parentSpanId\n if (!directParent) return { scopeSpanId: null, laneSpanId: null }\n\n let currentId: string | undefined = directParent\n let laneSpanId: string | null = null\n const seen = new Set<string>()\n while (currentId && !seen.has(currentId)) {\n seen.add(currentId)\n const current = spansById.get(currentId)\n if (!current) return { scopeSpanId: directParent, laneSpanId: directParent }\n if (current.kind === 'agent') {\n return { scopeSpanId: current.spanId, laneSpanId: laneSpanId ?? current.spanId }\n }\n laneSpanId = current.spanId\n currentId = current.parentSpanId\n }\n return { scopeSpanId: directParent, laneSpanId: directParent }\n}\n\nfunction callsAreSerial(previous: IndexedToolCall, next: IndexedToolCall): boolean {\n return previous.span.endedAt !== undefined && previous.span.endedAt <= next.span.startedAt\n}\n","/**\n * ToolWasteView — fraction of tool calls whose results weren't used\n * downstream. Without a \"used\" signal we fall back to structural\n * proxies: error calls, duplicate calls, and tool calls followed by\n * zero subsequent LLM spans are all considered waste.\n *\n * Consumers can pass a `usageOracle` that inspects a tool span and\n * returns true iff the tool's result appears in a later LLM message,\n * artifact, or state mutation — that's the canonical definition; the\n * default heuristic is a reasonable fallback.\n */\n\nimport { computeToolUseMetrics } from '../tool-use-metrics'\nimport { llmSpans, toolSpans } from '../trace/query'\nimport type { LlmSpan, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface ToolWasteFinding {\n runId: string\n wastedCalls: number\n totalCalls: number\n wasteRate: number\n}\n\nexport interface ToolWasteReport {\n byRun: ToolWasteFinding[]\n overallWasteRate: number\n}\n\nexport interface ToolWasteOptions {\n runId?: string\n usageOracle?: (tool: ToolSpan, later: { llm: Awaited<ReturnType<typeof llmSpans>> }) => boolean\n}\n\nexport async function toolWasteView(\n store: TraceStore,\n options: ToolWasteOptions = {},\n): Promise<ToolWasteReport> {\n const runs = options.runId ? [options.runId] : (await store.listRuns()).map((r) => r.runId)\n\n const byRun: ToolWasteFinding[] = []\n let totalCalls = 0\n let totalWasted = 0\n for (const runId of runs) {\n const tools = await toolSpans(store, runId)\n if (tools.length === 0) {\n byRun.push({ runId, wastedCalls: 0, totalCalls: 0, wasteRate: 0 })\n continue\n }\n const llms = await llmSpans(store, runId)\n // Sort LLM spans once by start time, then build a suffix index of the\n // concatenated message text. `suffixText[i]` is the haystack of every\n // string message content in spans[i..]. Per tool we binary-search the\n // first span started strictly after the tool, then test that one suffix\n // — turning the per-tool O(llms × messages × content) scan into a single\n // O(log llms) lookup over precomputed text.\n const sortedLlm = [...llms].sort((a, b) => a.startedAt - b.startedAt)\n const startTimes = sortedLlm.map((l) => l.startedAt)\n const suffixText = buildSuffixText(sortedLlm)\n let wasted = 0\n for (const t of tools) {\n if (t.status === 'error') {\n wasted++\n continue\n }\n // First LLM span started strictly after this tool (upper-bound search).\n const cutoff = upperBound(startTimes, t.startedAt)\n if (options.usageOracle) {\n if (!options.usageOracle(t, { llm: sortedLlm.slice(cutoff) })) wasted++\n } else {\n // Default heuristic: a tool whose result is NOT mentioned in any\n // later LLM input message is likely wasted. An empty/null result has\n // no payload to propagate downstream — there is nothing to find in a\n // later message, so it is not evidence of waste; skip it.\n const resultStr = stringify(t.result)\n if (resultStr === '') continue\n const haystack = suffixText[cutoff] ?? ''\n const used = haystack.includes(resultStr.slice(0, 120))\n if (!used) wasted++\n }\n }\n const wasteRate = wasted / tools.length\n byRun.push({ runId, wastedCalls: wasted, totalCalls: tools.length, wasteRate })\n totalCalls += tools.length\n totalWasted += wasted\n }\n return { byRun, overallWasteRate: totalCalls > 0 ? totalWasted / totalCalls : 0 }\n}\n\n/**\n * Build per-position suffix haystacks: result[i] is the concatenation of every\n * string message content in spans[i..end]. Built back-to-front so each entry\n * reuses the next one — O(total message text) rather than O(spans²).\n */\nfunction buildSuffixText(spans: LlmSpan[]): string[] {\n const result = new Array<string>(spans.length + 1)\n result[spans.length] = ''\n for (let i = spans.length - 1; i >= 0; i--) {\n const own = spans[i]!.messages.map((m) =>\n typeof m.content === 'string' ? m.content : '',\n ).join('\\n')\n result[i] = `${own}\\n${result[i + 1]}`\n }\n return result\n}\n\n/** Index of the first element strictly greater than `target` in a sorted array. */\nfunction upperBound(sorted: number[], target: number): number {\n let lo = 0\n let hi = sorted.length\n while (lo < hi) {\n const mid = (lo + hi) >>> 1\n if (sorted[mid]! <= target) lo = mid + 1\n else hi = mid\n }\n return lo\n}\n\nfunction stringify(v: unknown): string {\n if (v === null || v === undefined) return ''\n if (typeof v === 'string') return v\n try {\n return JSON.stringify(v)\n } catch {\n return String(v)\n }\n}\n\n// Re-export for convenience in consumers that want both descriptive and usage metrics.\nexport { computeToolUseMetrics }\n"],"mappings":";;;;;;;;AA8BA,eAAsB,iBACpB,OACA,UAAuD,CAAC,GAC3B;CAC7B,MAAM,OAAO,MAAM,MAAM,SAAS;EAChC,YAAY,QAAQ;EACpB,WAAW,QAAQ;CACrB,CAAC;CACD,MAAM,WAAkC,CAAC;CACzC,MAAM,cAAsC,CAAC;CAC7C,MAAM,aAAqC,CAAC;CAC5C,MAAM,YAAoC,CAAC;CAE3C,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,UAAU,MAAM,MAAM,OAAO,IAAI,KAAK;EAC5C,KAAK,MAAM,KAAK,SAAS;GACvB,IAAI,CAAC,EAAE,UAAU;GACjB,MAAM,cAAc,EAAE,QAAQ,IAAI,EAAE,WAAW,EAAE,QAAQ;GACzD,SAAS,KAAK;IACZ,OAAO,IAAI;IACX,YAAY,IAAI;IAChB,WAAW,IAAI;IACf,WAAW,EAAE;IACb,OAAO,EAAE;IACT,UAAU,EAAE;IACZ;IACA,WAAW,EAAE;GACf,CAAC;GACD,YAAY,EAAE,cAAc,YAAY,EAAE,cAAc,KAAK;GAC7D,WAAW,IAAI,eAAe,WAAW,IAAI,eAAe,KAAK;GACjE,IAAI,IAAI,WAAW,UAAU,IAAI,cAAc,UAAU,IAAI,cAAc,KAAK;EAClF;CACF;CAEA,MAAM,eAAe,IAAI,IAAI,SAAS,KAAK,MAAM,EAAE,KAAK,CAAC;CACzD,OAAO;EACL;EACA;EACA;EACA;EACA,WAAW,KAAK;EAChB,kBAAkB,KAAK,SAAS,IAAI,aAAa,OAAO,KAAK,SAAS;CACxE;AACF;;;;;;;;;;ACnCA,eAAsB,mBACpB,OACA,UAA8D,CAAC,GAChC;CAC/B,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,UAAU,QAAQ,kBAAkB;CAC1C,MAAM,OAAO,MAAM,MAAM,SAAS;CAGlC,MAAM,2BAAW,IAAI,IAAyB;CAC9C,IAAI,gBAAgB;CAEpB,KAAK,MAAM,OAAO,MAAM;EACtB,IAAI,IAAI,WAAW,eAAe,IAAI,SAAS,SAAS,OAAO;EAC/D;EACA,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,OAAO,IAAI,MAAM,CAAC;EAEpD,MAAM,MAAM,gBAAgB;GAAE;GAAK;GAAO,QAAA,MADrB,MAAM,OAAO,EAAE,OAAO,IAAI,MAAM,CAAC;EACL,GAAG,KAAK;EAEzD,IAAI;EACJ,IAAI;EACJ,IAAI;EACJ,IAAI,IAAI,eAAe;GACrB,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,WAAW,IAAI,aAAa;GAC7D,IAAI,MAAM,SAAS,QAAQ;IACzB,WAAW,KAAK;IAChB,IAAI,oBAAoB,IAAI,GAAG,YAAY,QAAQ,KAAK,IAAI,CAAC,CAAC,MAAM,GAAG,EAAE;GAC3E,OAAO,IAAI,MAAM,SAAS,SACxB,YAAY,KAAK;EAErB;EAEA,IAAI,CAAC,UAAU;GAEb,MAAM,WAAU,MADC,UAAU,OAAO,IAAI,KAAK,EAAA,CACxB,QAAQ,MAAM,EAAE,WAAW,OAAO,CAAC,CAAC,IAAI;GAC3D,IAAI,SAAS;IACX,WAAW,QAAQ;IACnB,IAAI,oBAAoB,OAAO,GAAG,YAAY,QAAQ,QAAQ,IAAI,CAAC,CAAC,MAAM,GAAG,EAAE;GACjF;EACF;EAIA,IAAI,CAAC,WAAW;GACd,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,SAAS,WAAW,OAAO,EAAE,cAAc,QAAQ;GACrF,IAAI,OAAO,SAAS,SAAS,YAAY,MAAM;EACjD;EAEA,MAAM,MAAM,GAAG,IAAI,aAAa,GAAG,YAAY,GAAG,GAAG,aAAa,GAAG,GAAG,aAAa;EACrF,IAAI,UAAU,SAAS,IAAI,GAAG;EAC9B,IAAI,CAAC,SAAS;GACZ,UAAU;IACR,cAAc,IAAI;IAClB;IACA;IACA;IACA,UAAU;IACV,aAAa,CAAC;IACd,cAAc,IAAI;IAClB,cAAc,kBAAkB,KAAK,KAAK,IAAI;GAChD;GACA,SAAS,IAAI,KAAK,OAAO;EAC3B;EACA,QAAQ;EACR,IAAI,CAAC,QAAQ,YAAY,SAAS,IAAI,UAAU,GAAG,QAAQ,YAAY,KAAK,IAAI,UAAU;CAC5F;CAMA,OAAO;EAAE,UAJG,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAC/B,QAAQ,MAAM,EAAE,YAAY,OAAO,CAAC,CACpC,MAAM,GAAG,MAAM,EAAE,WAAW,EAAE,QAEZ;EAAG;EAAe,WAAW,KAAK;CAAO;AAChE;AAEA,SAAS,kBAAkB,OAAmC;CAE5D,OADgB,MAAM,MAAM,MAAM,EAAE,WAAW,OAClC,CAAC,EAAE;AAClB;;;ACvFA,eAAsB,oBACpB,OACA,MACA,MACA,UAA6B,CAAC,GACH;CAC3B,MAAM,CAAC,GAAG,KAAK,MAAM,QAAQ,IAAI,CAAC,gBAAgB,OAAO,IAAI,GAAG,gBAAgB,OAAO,IAAI,CAAC,CAAC;CAC7F,MAAM,KAAK,QAAQ,cAAc;CACjC,MAAM,SAAS,KAAK,IAAI,EAAE,MAAM,QAAQ,EAAE,MAAM,MAAM;CACtD,KAAK,IAAI,IAAI,GAAG,IAAI,QAAQ,KAAK;EAC/B,MAAM,QAAQ,EAAE,MAAM;EACtB,MAAM,QAAQ,EAAE,MAAM;EACtB,IAAI,CAAC,GAAG,OAAO,KAAK,GAClB,OAAO;GACL;GACA;GACA,sBAAsB;GACtB;GACA;GACA,QAAQ,mBAAmB,OAAO,KAAK;GACvC,iBAAiB;EACnB;CAEJ;CACA,IAAI,EAAE,MAAM,WAAW,EAAE,MAAM,QAC7B,OAAO;EAAE;EAAM;EAAM,sBAAsB;EAAM,iBAAiB;CAAO;CAG3E,MAAM,SADqB,EAAE,MAAM,SAAS,EAAE,MAAM,SAAS,IAAI,EAAA,CAC5C,MAAM,SAAS;CAGpC,MAAM,SACJ,WAAW,IACP,0CAA0C,MAAM,gCAChD,sBAAsB,MAAM,4BAA4B,SAAS;CACvE,OAAO;EACL;EACA;EACA,sBAAsB;EACtB,OAAO,EAAE,MAAM;EACf,OAAO,EAAE,MAAM;EACf;EACA,iBAAiB;CACnB;AACF;AAEA,SAAS,kBAAkB,GAAmB,GAA4B;CACxE,IAAI,EAAE,KAAK,SAAS,EAAE,KAAK,MAAM,OAAO;CACxC,IAAI,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,SAAS,QAAQ,OAAO,EAAE,KAAK,aAAa,EAAE,KAAK;CACxF,IAAI,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,OAAO,OAAO,EAAE,KAAK,UAAU,EAAE,KAAK;CACnF,IAAI,EAAE,KAAK,SAAS,WAAW,EAAE,KAAK,SAAS,SAC7C,OAAO,EAAE,KAAK,cAAc,EAAE,KAAK;CACrC,OAAO,EAAE,KAAK,SAAS,EAAE,KAAK;AAChC;AAEA,SAAS,mBAAmB,GAAmB,GAA2B;CACxE,IAAI,EAAE,KAAK,SAAS,EAAE,KAAK,MAAM,OAAO,QAAQ,EAAE,KAAK,KAAK,MAAM,EAAE,KAAK;CACzE,IAAI,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,SAAS,UAAU,EAAE,KAAK,aAAa,EAAE,KAAK,UACjF,OAAO,QAAQ,EAAE,KAAK,SAAS,MAAM,EAAE,KAAK;CAE9C,IAAI,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,UAAU,EAAE,KAAK,OAC5E,OAAO,SAAS,EAAE,KAAK,MAAM,MAAM,EAAE,KAAK;CAE5C,OAAO,SAAS,EAAE,KAAK,KAAK,QAAQ,EAAE,KAAK,KAAK;AAClD;;;;;;;;;;;;AC9DA,eAAsB,mBAAmB,OAAkD;CACzF,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,MAAM,QAAQ,CAAC,EAAA,CAAG,QAChD,MAAsB,EAAE,SAAS,OACpC;CACA,IAAI,IAAI,WAAW,GAAG,OAAO;EAAE,OAAO,CAAC;EAAG,YAAY,CAAC;EAAG,UAAU,CAAC;CAAE;CAEvE,MAAM,8BAAc,IAAI,IAAyB;CACjD,KAAK,MAAM,KAAK,KAAK;EACnB,MAAM,MAAM,YAAY,IAAI,EAAE,SAAS,KAAK,CAAC;EAC7C,IAAI,KAAK,CAAC;EACV,YAAY,IAAI,EAAE,WAAW,GAAG;CAClC;CAEA,MAAM,WAAW,CAAC,GAAG,IAAI,IAAI,IAAI,KAAK,MAAM,EAAE,OAAO,CAAC,CAAC,CAAC,CAAC,KAAK;CAC9D,MAAM,QAAqB,CAAC;CAC5B,KAAK,MAAM,CAAC,KAAK,UAAU,aAAa;EACtC,MAAM,0BAAU,IAAI,IAAiC;EACrD,KAAK,MAAM,KAAK,OAAO;GACrB,MAAM,IAAI,QAAQ,IAAI,EAAE,OAAO,qBAAK,IAAI,IAAoB;GAC5D,EAAE,IAAI,EAAE,cAAc,EAAE,KAAK;GAC7B,QAAQ,IAAI,EAAE,SAAS,CAAC;EAC1B;EACA,MAAM,aAAa,CAAC,GAAG,QAAQ,KAAK,CAAC;EACrC,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,QAAQ,KACrC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,WAAW,QAAQ,KAAK;GAC9C,MAAM,SAAS,WAAW;GAC1B,MAAM,SAAS,WAAW;GAC1B,MAAM,IAAI,QAAQ,IAAI,MAAM;GAC5B,MAAM,IAAI,QAAQ,IAAI,MAAM;GAC5B,MAAM,SAAkC,CAAC;GACzC,KAAK,MAAM,CAAC,QAAQ,WAAW,GAAG;IAChC,MAAM,SAAS,EAAE,IAAI,MAAM;IAC3B,IAAI,WAAW,KAAA,GAAW,OAAO,KAAK,CAAC,QAAQ,MAAM,CAAC;GACxD;GACA,IAAI,OAAO,SAAS,GAAG;GACvB,MAAM,cAAc,OAAO,KACxB,CAAC,QAAQ,YACR,CACE;IAAE,WAAW;IAAQ,WAAW;IAAK,OAAO;IAAQ,WAAW;GAAG,GAClE;IAAE,WAAW;IAAQ,WAAW;IAAK,OAAO;IAAQ,WAAW;GAAG,CACpE,CACJ;GACA,MAAM,IAAI,sBACR,YAAY,EAAE,CAAE,KAAK,GAAG,OAAO,YAAY,KAAK,SAAS,KAAK,GAAI,CAAC,CACrE;GACA,MAAM,KAAK;IACT,QAAQ;IACR,QAAQ;IACR,WAAW;IACX,aAAa,OAAO;IACpB,SAAS,SACP,OAAO,KAAK,MAAM,EAAE,EAAE,GACtB,OAAO,KAAK,MAAM,EAAE,EAAE,CACxB;IACA,cAAc;GAChB,CAAC;EACH;CAEJ;CAEA,OAAO;EACL,OAAO,MAAM,MAAM,GAAG,MAAM,EAAE,cAAc,EAAE,WAAW;EACzD,YAAY,CAAC,GAAG,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK;EACzC;CACF;AACF;;;;;;;;;;;ACrEA,eAAsB,eACpB,OACA,SACA,SACyB;CACzB,MAAM,eAAe,MAAM,MAAM,SAAS,QAAQ,QAAQ;CAC1D,MAAM,gBAAgB,MAAM,MAAM,SAAS,QAAQ,SAAS;CAS5D,OAAO,kBAAkB,MARH,QAAQ,IAC5B,QAAQ,IAAI,OAAO,MAAM;EACvB,MAAM,UAAU,EAAE,WAAW,eAAe,EAAE,MAAM;EACpD,MAAM,WAAW,MAAM,WAAW,cAAc,SAAS,KAAK;EAC9D,MAAM,YAAY,MAAM,WAAW,eAAe,SAAS,KAAK;EAChE,OAAO;GAAE,QAAQ,EAAE;GAAQ,gBAAgB,EAAE;GAAgB;GAAU;EAAU;CACnF,CAAC,CACH,GACkC,OAAO;AAC3C;AAEA,eAAe,WACb,MACA,SACA,OACmB;CACnB,MAAM,MAAgB,CAAC;CACvB,KAAK,MAAM,KAAK,MAAM;EACpB,MAAM,IAAI,MAAM,QAAQ,GAAG,KAAK;EAChC,IAAI,MAAM,QAAQ,OAAO,SAAS,CAAC,GAAG,IAAI,KAAK,CAAC;CAClD;CACA,OAAO;AACT;AAEA,SAAS,eAAe,QAAyE;CAC/F,OAAO,OAAO,KAAK,UAAU;EAC3B,QAAQ,QAAR;GACE,KAAK;GACL,KAAK,gBACH,OAAO,IAAI,SAAS,SAAS;GAC/B,KAAK,QACH,OAAO,IAAI,SAAS,SAAS,OAAO,IAAI;GAC1C,KAAK,cACH,OAAO,IAAI,WAAW,IAAI,YAAY,IAAI,UAAU,IAAI,YAAY;GACtE,KAAK,WAEH,OAAO,aAAa,MADF,SAAS,OAAO,IAAI,KAAK,CACpB,CAAC,CAAC;GAE3B,KAAK,eAEH,OAAO,aAAa,MADF,SAAS,OAAO,IAAI,KAAK,CACpB,CAAC,CAAC;GAE3B,KAAK,gBAEH,OAAO,aAAa,MADF,SAAS,OAAO,IAAI,KAAK,CACpB,CAAC,CAAC;GAE3B,KAAK,gBACH,OAAO,gBAAgB,GAAG,MAAM,YAAY,IAAI;GAElD,SACE,OAAO;EACX;CACF;AACF;;;;;;;;;;;;ACvEA,MAAM,wBAAwB;AAC9B,MAAM,qCAAqC;AA8C3C,eAAsB,cACpB,OACA,UAA4B,CAAC,GACH;CAC1B,MAAM,iBAAiB,QAAQ,kBAAkB;CACjD,MAAM,cAAc,QAAQ,eAAe;CAC3C,MAAM,0BACJ,QAAQ,2BAA2B;CACrC,IAAI,CAAC,OAAO,UAAU,cAAc,KAAK,iBAAiB,GACxD,MAAM,IAAI,WAAW,2CAA2C;CAElE,IAAI,CAAC,OAAO,SAAS,WAAW,KAAK,cAAc,GACjD,MAAM,IAAI,WAAW,kDAAkD;CAEzE,IAAI,CAAC,OAAO,UAAU,uBAAuB,KAAK,0BAA0B,GAC1E,MAAM,IAAI,WAAW,wDAAwD;CAE/E,MAAM,OAAO,QAAQ,QACjB,CAAC,EAAE,OAAO,QAAQ,MAAM,CAAC,KACxB,MAAM,MAAM,SAAS,EAAA,CAAG,KAAK,OAAO,EAAE,OAAO,EAAE,MAAM,EAAE;CAE5D,MAAM,WAA+B,CAAC;CACtC,KAAK,MAAM,EAAE,WAAW,MAAM;EAC5B,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,MAAM,CAAC;EACzC,MAAM,YAAY,IAAI,IAAI,MAAM,KAAK,SAAS,CAAC,KAAK,QAAQ,IAAI,CAAC,CAAC;EAClE,MAAM,cAAgC,MACnC,OAAO,UAAU,CAAC,CAClB,KAAK,MAAM,iBAAiB;GAAE;GAAM;EAAY,EAAE,CAAC,CACnD,MAAM,GAAG,MAAM,EAAE,KAAK,YAAY,EAAE,KAAK,aAAa,EAAE,cAAc,EAAE,WAAW,CAAC,CACpF,KAAK,EAAE,YAAY;GAAE;GAAM,GAAG,eAAe,MAAM,SAAS;EAAE,EAAE;EACnE,MAAM,cAAc,qBAClB,YAAY,KAAK,SAAS;GAExB,MAAM,QADS,KAAK,eAAe,QAAQ,KAAK,eAAe,KAAK,cAEhE,KAAK,OACL,KAAK,aACH,UAAU,IAAI,KAAK,UAAU,IAC7B,KAAA;GACN,OAAO;IACL,KAAK,aAAa,IAAI;IACtB,UAAU,KAAK,UAAU,KAAK,WAAW;IACzC,OAAO,OAAO,aAAa;IAC3B,KAAK,OAAO,WAAW;GACzB;EACF,CAAC,CACH;EACA,MAAM,uCAAuB,IAAI,IAAoB;EACrD,MAAM,eAAkC,YAAY,KAAK,SAAS;GAChE,MAAM,UAAU,YAAY,IAAI,aAAa,IAAI,CAAC;GAClD,MAAM,gBAAgB,qBAAqB,IAAI,OAAO,KAAK;GAC3D,qBAAqB,IAAI,SAAS,gBAAgB,CAAC;GACnD,OAAO;IAAE,GAAG;IAAM;IAAe;GAAQ;EAC3C,CAAC;EACD,MAAM,wBAAQ,IAAI,IAQhB;EACF,KAAK,MAAM,QAAQ,cAAc;GAC/B,IAAI,CAAC,oBAAoB,KAAK,IAAI,GAAG;GACrC,MAAM,IAAI,QAAQ,KAAK,KAAK,IAAI;GAChC,MAAM,MAAM,KAAK,UAAU;IAAC,KAAK;IAAS,KAAK,KAAK;IAAU;GAAC,CAAC;GAChE,MAAM,SAAS,MAAM,IAAI,GAAG,KAAK;IAC/B,OAAO,CAAC;IACR,SAAS;IACT,UAAU,KAAK,KAAK;IACpB,aAAa,KAAK;GACpB;GACA,OAAO,MAAM,KAAK,IAAI;GACtB,MAAM,IAAI,KAAK,MAAM;EACvB;EACA,KAAK,MAAM,EAAE,OAAO,SAAS,GAAG,UAAU,iBAAiB,MAAM,OAAO,GAAG;GACzE,IAAI,MAAM,SAAS,gBAAgB;GACnC,IAAI,eAAe;GACnB,KAAK,IAAI,aAAa,GAAG,cAAc,MAAM,QAAQ,cAAc,GAAG;IACpE,MAAM,WAAW,MAAM,aAAa;IACpC,MAAM,OAAO,MAAM;IAMnB,IAAI,EAJF,SAAS,KAAA,KACT,KAAK,KAAK,YAAY,SAAS,KAAK,YAAY,eAChD,KAAK,gBAAgB,SAAS,gBAAgB,IAAI,2BAClD,CAAC,eAAe,UAAU,IAAI,IACb;IAEnB,MAAM,UAAU,MAAM,MAAM,cAAc,UAAU;IACpD,IAAI,OAAO;IACX,IAAI,YAAY;IAChB,IAAI,UAAU;IACd,KAAK,IAAI,QAAQ,GAAG,QAAQ,QAAQ,QAAQ,SAAS,GAAG;KACtD,OAAO,QAAQ,MAAM,CAAE,KAAK,YAAY,QAAQ,KAAK,CAAE,KAAK,YAAY,aACtE,QAAQ;KAEV,IAAI,QAAQ,OAAO,UAAU,WAAW;MACtC,YAAY;MACZ,UAAU;KACZ;IACF;IACA,IAAI,UAAU,YAAY,KAAK,gBAAgB;KAC7C,MAAM,OAAO,QAAQ,MAAM,WAAW,UAAU,CAAC;KACjD,MAAM,QAAQ,KAAK,EAAE,CAAE,KAAK;KAC5B,MAAM,OAAO,KAAK,KAAK,SAAS,EAAE,CAAE,KAAK;KACzC,SAAS,KAAK;MACZ;MACA;MACA,SAAS;MACT,aAAa,KAAK;MAClB,SAAS,KAAK,KAAK,SAAS,KAAK,KAAK,MAAM;MAC5C,GAAI,cAAc,EAAE,YAAY,IAAI,CAAC;MACrC,UAAU,OAAO;KACnB,CAAC;IACH;IACA,eAAe;GACjB;EACF;CACF;CAEA,MAAM,eAAe,IAAI,IAAI,SAAS,KAAK,MAAM,EAAE,KAAK,CAAC;CACzD,OAAO;EACL;EACA,kBAAkB,KAAK,SAAS,IAAI,aAAa,OAAO,KAAK,SAAS;EACtE,WAAW,KAAK;CAClB;AACF;AAEA,SAAS,QAAQ,MAAkE;CACjF,OAAO,KAAK,UAAU,CAAC,KAAK,aAAa,KAAK,UAAU,CAAC;AAC3D;AAEA,SAAS,aAAa,MAA8B;CAClD,OAAO,KAAK,eAAe,QAAQ,KAAK,eAAe,KAAK,cACxD,KAAK,UAAU,CAAC,KAAK,aAAa,KAAK,KAAK,MAAM,CAAC,IACnD,QAAQ,IAAI;AAClB;AAEA,SAAS,eACP,MACA,WAC2D;CAC3D,MAAM,eAAe,KAAK;CAC1B,IAAI,CAAC,cAAc,OAAO;EAAE,aAAa;EAAM,YAAY;CAAK;CAEhE,IAAI,YAAgC;CACpC,IAAI,aAA4B;CAChC,MAAM,uBAAO,IAAI,IAAY;CAC7B,OAAO,aAAa,CAAC,KAAK,IAAI,SAAS,GAAG;EACxC,KAAK,IAAI,SAAS;EAClB,MAAM,UAAU,UAAU,IAAI,SAAS;EACvC,IAAI,CAAC,SAAS,OAAO;GAAE,aAAa;GAAc,YAAY;EAAa;EAC3E,IAAI,QAAQ,SAAS,SACnB,OAAO;GAAE,aAAa,QAAQ;GAAQ,YAAY,cAAc,QAAQ;EAAO;EAEjF,aAAa,QAAQ;EACrB,YAAY,QAAQ;CACtB;CACA,OAAO;EAAE,aAAa;EAAc,YAAY;CAAa;AAC/D;AAEA,SAAS,eAAe,UAA2B,MAAgC;CACjF,OAAO,SAAS,KAAK,YAAY,KAAA,KAAa,SAAS,KAAK,WAAW,KAAK,KAAK;AACnF;;;;;;;;;;;;;;AC/LA,eAAsB,cACpB,OACA,UAA4B,CAAC,GACH;CAC1B,MAAM,OAAO,QAAQ,QAAQ,CAAC,QAAQ,KAAK,KAAK,MAAM,MAAM,SAAS,EAAA,CAAG,KAAK,MAAM,EAAE,KAAK;CAE1F,MAAM,QAA4B,CAAC;CACnC,IAAI,aAAa;CACjB,IAAI,cAAc;CAClB,KAAK,MAAM,SAAS,MAAM;EACxB,MAAM,QAAQ,MAAM,UAAU,OAAO,KAAK;EAC1C,IAAI,MAAM,WAAW,GAAG;GACtB,MAAM,KAAK;IAAE;IAAO,aAAa;IAAG,YAAY;IAAG,WAAW;GAAE,CAAC;GACjE;EACF;EAQA,MAAM,YAAY,CAAC,GAAG,MAPH,SAAS,OAAO,KAAK,CAOd,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;EACpE,MAAM,aAAa,UAAU,KAAK,MAAM,EAAE,SAAS;EACnD,MAAM,aAAa,gBAAgB,SAAS;EAC5C,IAAI,SAAS;EACb,KAAK,MAAM,KAAK,OAAO;GACrB,IAAI,EAAE,WAAW,SAAS;IACxB;IACA;GACF;GAEA,MAAM,SAAS,WAAW,YAAY,EAAE,SAAS;GACjD,IAAI,QAAQ,aACN;QAAA,CAAC,QAAQ,YAAY,GAAG,EAAE,KAAK,UAAU,MAAM,MAAM,EAAE,CAAC,GAAG;GAAA,OAC1D;IAKL,MAAM,YAAY,UAAU,EAAE,MAAM;IACpC,IAAI,cAAc,IAAI;IAGtB,IAAI,EAFa,WAAW,WAAW,GAAA,CACjB,SAAS,UAAU,MAAM,GAAG,GAAG,CAC7C,GAAG;GACb;EACF;EACA,MAAM,YAAY,SAAS,MAAM;EACjC,MAAM,KAAK;GAAE;GAAO,aAAa;GAAQ,YAAY,MAAM;GAAQ;EAAU,CAAC;EAC9E,cAAc,MAAM;EACpB,eAAe;CACjB;CACA,OAAO;EAAE;EAAO,kBAAkB,aAAa,IAAI,cAAc,aAAa;CAAE;AAClF;;;;;;AAOA,SAAS,gBAAgB,OAA4B;CACnD,MAAM,SAAS,IAAI,MAAc,MAAM,SAAS,CAAC;CACjD,OAAO,MAAM,UAAU;CACvB,KAAK,IAAI,IAAI,MAAM,SAAS,GAAG,KAAK,GAAG,KAIrC,OAAO,KAAK,GAHA,MAAM,EAAE,CAAE,SAAS,KAAK,MAClC,OAAO,EAAE,YAAY,WAAW,EAAE,UAAU,EAC9C,CAAC,CAAC,KAAK,IACU,EAAE,IAAI,OAAO,IAAI;CAEpC,OAAO;AACT;;AAGA,SAAS,WAAW,QAAkB,QAAwB;CAC5D,IAAI,KAAK;CACT,IAAI,KAAK,OAAO;CAChB,OAAO,KAAK,IAAI;EACd,MAAM,MAAO,KAAK,OAAQ;EAC1B,IAAI,OAAO,QAAS,QAAQ,KAAK,MAAM;OAClC,KAAK;CACZ;CACA,OAAO;AACT;AAEA,SAAS,UAAU,GAAoB;CACrC,IAAI,MAAM,QAAQ,MAAM,KAAA,GAAW,OAAO;CAC1C,IAAI,OAAO,MAAM,UAAU,OAAO;CAClC,IAAI;EACF,OAAO,KAAK,UAAU,CAAC;CACzB,QAAQ;EACN,OAAO,OAAO,CAAC;CACjB;AACF"}
@@ -2,7 +2,7 @@ import { c as VerificationError, s as ValidationError } from "./errors-8YnH8WlF.
2
2
  import { r as observedSplitScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
3
3
  import { t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
4
4
  import { s as validateRunRecord } from "./run-record-BIwU2wdV.js";
5
- import { a as summaryTable } from "./summary-report-Ci17nIdU.js";
5
+ import { a as summaryTable } from "./summary-report-BxtossFi.js";
6
6
  //#region src/release-confidence.ts
7
7
  const DEFAULT_THRESHOLDS = {
8
8
  requireCorpus: true,
@@ -600,4 +600,4 @@ function duration(value) {
600
600
  //#endregion
601
601
  export { evaluateReleaseConfidence as a, assertReleaseConfidence as i, bootstrapCi as n, judgeReplayGate as r, renderReleaseReport as t };
602
602
 
603
- //# sourceMappingURL=release-report-wuilQkvK.js.map
603
+ //# sourceMappingURL=release-report-BVZBmRZp.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"release-report-wuilQkvK.js","names":[],"sources":["../src/release-confidence.ts","../src/promotion-gate.ts","../src/release-report.ts"],"sourcesContent":["/**\n * Release confidence gate.\n *\n * This is the production-facing composition layer over the lower-level\n * primitives:\n * - Dataset manifests prove corpus/version coverage.\n * - RunRecord rows prove reproducible search/holdout outcomes.\n * - Multi-shot trace evidence carries turn counts and ASI diagnostics.\n * - HeldOutGate decisions remain the paired promotion authority.\n *\n * The gate is intentionally pure and conservative. Missing declared evidence\n * fails closed instead of being treated as a neutral zero.\n */\n\nimport type { DatasetManifest, DatasetScenario, DatasetSplit } from './dataset'\nimport { ValidationError, VerificationError } from './errors'\nimport type { GateDecision } from './held-out-gate'\nimport { isRealnessGated, observedSplitScore } from './rollout/reward'\nimport { type RunRecord, type RunSplitTag, validateRunRecord } from './run-record'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Severity of an actionable finding attached to a run/trace. */\nexport type AsiSeverity = 'info' | 'warning' | 'error' | 'critical'\n\n/** Actionable side-info — a diagnosed finding the loop can act on. */\nexport interface ActionableSideInfo {\n /** Stable expectation/check id when available. */\n expectationId?: string\n /** Human-readable diagnosis of what happened. */\n message: string\n severity?: AsiSeverity\n /** Concrete trace excerpt, file path, tool call, screenshot id, etc. */\n evidence?: string\n /** Prompt/tool/context surface likely responsible. */\n responsibleSurface?: string\n /** Suggested fix in natural language. */\n suggestion?: string\n /** Whether this expectation was satisfied. Defaults to false for ASI rows. */\n matched?: boolean\n metadata?: Record<string, unknown>\n}\n\nexport type ReleaseConfidenceStatus = 'pass' | 'warn' | 'fail'\nexport type ReleaseConfidenceAxisName =\n | 'corpus'\n | 'quality'\n | 'reliability'\n | 'generalization'\n | 'diagnostics'\n | 'efficiency'\n\nexport interface ReleaseTraceEvidence {\n scenarioId: string\n candidateId?: string\n split?: RunSplitTag\n score?: number\n ok?: boolean\n turnCount?: number\n costUsd?: number\n durationMs?: number\n /** Canonical task-failure class. Free-form detail belongs in ASI. */\n failureClass?: FailureClass\n asi?: ActionableSideInfo[]\n metadata?: Record<string, unknown>\n}\n\nexport interface ReleaseConfidenceThresholds {\n /** Require a Dataset manifest or explicit scenarios. Default true. */\n requireCorpus?: boolean\n minScenarioCount?: number\n minSearchRuns?: number\n minHoldoutRuns?: number\n /** Require at least one holdout scenario/run. Default true. */\n requireHoldout?: boolean\n minPassRate?: number\n minMeanScore?: number\n /** Search mean may exceed holdout mean by at most this much. */\n maxOverfitGap?: number\n maxMeanCostUsd?: number\n maxP95WallMs?: number\n /** Low-score/failed rows must carry ASI. Default true. */\n requireAsiForFailures?: boolean\n /** Score below this is considered a failure for ASI coverage. Default 0.5. */\n failureScoreThreshold?: number\n}\n\nexport interface ReleaseConfidenceInput {\n target: string\n candidateId?: string\n baselineId?: string\n dataset?: DatasetManifest\n scenarios?: readonly DatasetScenario[]\n runs?: readonly RunRecord[]\n traces?: readonly ReleaseTraceEvidence[]\n gateDecision?: GateDecision | null\n thresholds?: ReleaseConfidenceThresholds\n}\n\nexport interface ReleaseConfidenceAxis {\n name: ReleaseConfidenceAxisName\n status: ReleaseConfidenceStatus\n score: number | null\n detail: string\n}\n\nexport interface ReleaseConfidenceIssue {\n axis: ReleaseConfidenceAxisName\n severity: 'critical' | 'warning'\n code: string\n detail: string\n}\n\nexport interface ReleaseConfidenceMetrics {\n scenarioCount: number\n /** Search rows with a finite search score. */\n searchRuns: number\n /** Holdout rows with a finite holdout score. */\n holdoutRuns: number\n /** Runs with neither a split-matched score nor an explicit task failure. */\n unscoredRuns: number\n /** Run rows, or trace rows when no runs exist, with no classified terminal result. */\n unclassifiedTerminalRuns: number\n /** Run rows, or trace rows when no runs exist, that ended unsuccessfully. */\n terminalFailureRuns: number\n /** Success fraction when every run or fallback trace row has a classified result. */\n reliabilityRate: number | null\n passRate: number | null\n meanScore: number | null\n searchMeanScore: number | null\n holdoutMeanScore: number | null\n overfitGap: number | null\n meanCostUsd: number | null\n p95WallMs: number | null\n failedRows: number\n failuresWithAsi: number\n singleShotTraces: number\n multiShotTraces: number\n splitCounts: Record<DatasetSplit, number>\n domainCounts: Record<string, number>\n failureClassCounts: Partial<Record<FailureClass, number>>\n responsibleSurfaceCounts: Record<string, number>\n /**\n * Runs excluded from `passRate` because the authenticity gate flagged them as\n * gamed. Surfaced, never silent: a release whose pass rate is computed over a\n * shrunken denominator has to say by how much, or the exclusion is just a\n * different way of hiding the same runs.\n */\n realnessGatedRuns: number\n}\n\nexport interface ReleaseConfidenceScorecard {\n target: string\n candidateId: string | null\n baselineId: string | null\n status: ReleaseConfidenceStatus\n promote: boolean\n axes: ReleaseConfidenceAxis[]\n issues: ReleaseConfidenceIssue[]\n metrics: ReleaseConfidenceMetrics\n dataset: DatasetManifest | null\n gateDecision: GateDecision | null\n summary: string\n}\n\nconst DEFAULT_THRESHOLDS: Required<ReleaseConfidenceThresholds> = {\n requireCorpus: true,\n minScenarioCount: 1,\n minSearchRuns: 1,\n minHoldoutRuns: 1,\n requireHoldout: true,\n minPassRate: 0.8,\n minMeanScore: 0.7,\n maxOverfitGap: 0.15,\n maxMeanCostUsd: Number.POSITIVE_INFINITY,\n maxP95WallMs: Number.POSITIVE_INFINITY,\n requireAsiForFailures: true,\n failureScoreThreshold: 0.5,\n}\n\nexport function evaluateReleaseConfidence(\n input: ReleaseConfidenceInput,\n): ReleaseConfidenceScorecard {\n const thresholds = { ...DEFAULT_THRESHOLDS, ...input.thresholds }\n const candidateId = input.candidateId ?? null\n const runs = filterCandidate(\n (input.runs ?? []).map(validateRunRecord),\n candidateId,\n input.baselineId,\n )\n const traces = filterTraceCandidate(\n (input.traces ?? []).map(validateReleaseTraceEvidence),\n candidateId,\n input.baselineId,\n )\n const scenarios = input.scenarios ?? []\n const scenarioCount = input.dataset?.scenarioCount ?? scenarios.length\n const splitCounts = input.dataset?.splitCounts ?? countScenarioSplits(scenarios)\n const searchScores = scoresFor(runs, 'search')\n const holdoutScores = scoresFor(runs, 'holdout')\n const runScores = runs.map(runSplitScore).filter(isFiniteNumber)\n const traceScores = traces.map((t) => t.score).filter(isFiniteNumber)\n const scoreUniverse = runs.length > 0 ? runScores : traceScores\n const qualityRuns = runs.filter((run) => runSplitScore(run) !== undefined)\n const unscoredRuns = runs.filter(\n (run) =>\n runSplitScore(run) === undefined &&\n !hasExplicitTaskFailure(run) &&\n !isFailedTerminalOutcome(run.terminalOutcome),\n ).length\n // Realness-gated runs are EXCLUDED from the pass rate — numerator AND\n // denominator — and the count of what was dropped ships beside the rate as\n // `metrics.realnessGatedRuns`. Counting a gamed run as a pass made faking a\n // success the cheapest way to improve a release scorecard; scoring it 0\n // instead would be the other error, silently deflating the rate with a run\n // the gate says carries no usable verdict at all.\n const honestRuns = runs.filter((run) => !isRealnessGated(run))\n const passOutcomes =\n runs.length > 0\n ? honestRuns.map((run) => runPassOutcome(run, thresholds.failureScoreThreshold))\n : traces.map((trace) => tracePassOutcome(trace, thresholds.failureScoreThreshold))\n const reliabilityRows =\n runs.length > 0\n ? runs.map((run) => terminalSuccess(run.terminalOutcome))\n : traces.map((trace) => trace.ok)\n const unclassifiedTerminalRuns = reliabilityRows.filter((outcome) => outcome === undefined).length\n const terminalFailureRuns = reliabilityRows.filter((outcome) => outcome === false).length\n const reliabilityRate =\n reliabilityRows.length === 0 || unclassifiedTerminalRuns > 0\n ? null\n : (reliabilityRows.length - terminalFailureRuns) / reliabilityRows.length\n const searchRuns = qualityRuns.filter((r) => r.splitTag === 'search').length\n const holdoutRuns = qualityRuns.filter((r) => r.splitTag === 'holdout').length\n const failed = failedRows(runs, traces, thresholds.failureScoreThreshold)\n const searchMeanScore = meanOrNull(searchScores)\n const holdoutMeanScore = meanOrNull(holdoutScores)\n const runCosts = runs.flatMap((run) =>\n run.costProvenance.kind === 'uncaptured' ? [] : [run.costProvenance.usd],\n )\n const traceCosts = traces.map((trace) => trace.costUsd).filter(isFiniteNumber)\n const meanCostUsd =\n runs.length > 0\n ? runCosts.length === runs.length\n ? meanOrNull(runCosts)\n : null\n : meanOrNull(traceCosts)\n const wallTimes =\n runs.length > 0\n ? runs.map((run) => run.wallMs)\n : traces.map((trace) => trace.durationMs).filter(isFiniteNumber)\n const metrics: ReleaseConfidenceMetrics = {\n scenarioCount,\n searchRuns,\n holdoutRuns,\n unscoredRuns,\n unclassifiedTerminalRuns,\n terminalFailureRuns,\n reliabilityRate,\n passRate: passOutcomeRate(passOutcomes),\n realnessGatedRuns: runs.length - honestRuns.length,\n meanScore: meanOrNull(scoreUniverse),\n searchMeanScore,\n holdoutMeanScore,\n overfitGap: diffOrNull(searchMeanScore, holdoutMeanScore),\n meanCostUsd,\n p95WallMs: percentileOrNull(wallTimes, 0.95),\n failedRows: failed.length,\n failuresWithAsi: failed.filter((row) => row.hasAsi).length,\n singleShotTraces: traces.filter((t) => t.turnCount === 1).length,\n multiShotTraces: traces.filter((t) => (t.turnCount ?? 0) > 1).length,\n splitCounts,\n domainCounts: countDomains(scenarios),\n failureClassCounts: countFailureClasses(runs, traces, thresholds.failureScoreThreshold),\n responsibleSurfaceCounts: countResponsibleSurfaces(traces),\n }\n\n const issues: ReleaseConfidenceIssue[] = []\n checkCorpus(input, thresholds, metrics, issues)\n checkQuality(thresholds, metrics, issues)\n checkReliability(metrics, issues)\n checkGeneralization(input.gateDecision ?? null, thresholds, metrics, issues)\n checkDiagnostics(thresholds, metrics, issues)\n checkEfficiency(thresholds, metrics, issues)\n\n const axes = buildAxes(metrics, thresholds, issues)\n const status = issues.some((i) => i.severity === 'critical')\n ? 'fail'\n : issues.length > 0\n ? 'warn'\n : 'pass'\n\n return {\n target: input.target,\n candidateId,\n baselineId: input.baselineId ?? null,\n status,\n promote: status === 'pass' && (input.gateDecision ? input.gateDecision.promote : true),\n axes,\n issues,\n metrics,\n dataset: input.dataset ?? null,\n gateDecision: input.gateDecision ?? null,\n summary: renderSummary(input.target, status, metrics, issues),\n }\n}\n\nexport function assertReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard {\n const scorecard = evaluateReleaseConfidence(input)\n if (scorecard.status === 'fail') {\n throw new VerificationError(scorecard.summary)\n }\n return scorecard\n}\n\nfunction filterCandidate(\n runs: readonly RunRecord[],\n candidateId: string | null,\n baselineId?: string,\n): RunRecord[] {\n if (candidateId) return runs.filter((r) => r.candidateId === candidateId)\n if (baselineId) return runs.filter((r) => r.candidateId !== baselineId)\n return [...runs]\n}\n\nfunction filterTraceCandidate(\n traces: readonly ReleaseTraceEvidence[],\n candidateId: string | null,\n baselineId?: string,\n): ReleaseTraceEvidence[] {\n if (candidateId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId === candidateId)\n if (baselineId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId !== baselineId)\n return [...traces]\n}\n\nfunction validateReleaseTraceEvidence(\n trace: ReleaseTraceEvidence,\n index: number,\n): ReleaseTraceEvidence {\n const value = trace as unknown as Record<string, unknown>\n if (Object.hasOwn(value, 'failureMode')) {\n throw new ValidationError(\n `traces[${index}].failureMode is not supported; use canonical failureClass`,\n )\n }\n if (trace.failureClass !== undefined && !FAILURE_CLASSES.includes(trace.failureClass)) {\n throw new ValidationError(\n `traces[${index}].failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n )\n }\n return trace\n}\n\nfunction checkCorpus(\n input: ReleaseConfidenceInput,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireCorpus && !input.dataset && (input.scenarios?.length ?? 0) === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_corpus',\n detail: 'No Dataset manifest or scenarios supplied.',\n })\n }\n if (metrics.scenarioCount < thresholds.minScenarioCount) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'few_scenarios',\n detail: `${metrics.scenarioCount} scenario(s) < min ${thresholds.minScenarioCount}.`,\n })\n }\n if (thresholds.requireHoldout && metrics.splitCounts.holdout === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_holdout_split',\n detail: 'Corpus has no holdout scenarios.',\n })\n }\n}\n\nfunction checkQuality(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.searchRuns < thresholds.minSearchRuns) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'few_search_runs',\n detail: `${metrics.searchRuns} search run(s) < min ${thresholds.minSearchRuns}.`,\n })\n }\n if (metrics.unscoredRuns > 0) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'unscored_runs',\n detail: `${metrics.unscoredRuns} supplied run(s) have no task result.`,\n })\n }\n if (metrics.passRate === null || metrics.meanScore === null) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'missing_quality_scores',\n detail: 'No task-quality scores are available for pass-rate and mean-score checks.',\n })\n }\n if (metrics.passRate !== null && metrics.passRate < thresholds.minPassRate) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_pass_rate',\n detail: `passRate ${fmt(metrics.passRate)} < ${fmt(thresholds.minPassRate)}.`,\n })\n }\n if (metrics.meanScore !== null && metrics.meanScore < thresholds.minMeanScore) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_mean_score',\n detail: `meanScore ${fmt(metrics.meanScore)} < ${fmt(thresholds.minMeanScore)}.`,\n })\n }\n}\n\nfunction checkReliability(\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.reliabilityRate === null) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'missing_reliability_evidence',\n detail:\n metrics.unclassifiedTerminalRuns > 0\n ? `${metrics.unclassifiedTerminalRuns} supplied run(s) have no classified terminal result.`\n : 'No classified terminal results are available.',\n })\n }\n if (metrics.terminalFailureRuns > 0) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'terminal_run_failures',\n detail: `${metrics.terminalFailureRuns} run(s) ended failed, cancelled, or incomplete.`,\n })\n }\n}\n\nfunction checkGeneralization(\n gateDecision: GateDecision | null,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireHoldout && metrics.holdoutRuns < thresholds.minHoldoutRuns) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'few_holdout_runs',\n detail: `${metrics.holdoutRuns} holdout run(s) < min ${thresholds.minHoldoutRuns}.`,\n })\n }\n if (metrics.overfitGap !== null && metrics.overfitGap > thresholds.maxOverfitGap) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'overfit_gap',\n detail: `search-holdout gap ${fmt(metrics.overfitGap)} > ${fmt(thresholds.maxOverfitGap)}.`,\n })\n }\n if (gateDecision && !gateDecision.promote) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: `gate_${gateDecision.rejectionCode ?? 'reject'}`,\n detail: gateDecision.reason,\n })\n }\n}\n\nfunction checkDiagnostics(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (!thresholds.requireAsiForFailures) return\n if (metrics.failedRows > metrics.failuresWithAsi) {\n issues.push({\n axis: 'diagnostics',\n severity: 'critical',\n code: 'missing_failure_asi',\n detail: `${metrics.failedRows - metrics.failuresWithAsi} failed row(s) have no actionable side information.`,\n })\n }\n}\n\nfunction checkEfficiency(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (Number.isFinite(thresholds.maxMeanCostUsd) && metrics.meanCostUsd === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_cost',\n detail: 'A finite cost limit was configured but no cost evidence is available.',\n })\n } else if (metrics.meanCostUsd !== null && metrics.meanCostUsd > thresholds.maxMeanCostUsd) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'cost_budget',\n detail: `meanCostUsd ${fmt(metrics.meanCostUsd)} > ${fmt(thresholds.maxMeanCostUsd)}.`,\n })\n }\n if (Number.isFinite(thresholds.maxP95WallMs) && metrics.p95WallMs === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_latency',\n detail: 'A finite latency limit was configured but no latency evidence is available.',\n })\n } else if (metrics.p95WallMs !== null && metrics.p95WallMs > thresholds.maxP95WallMs) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'latency_budget',\n detail: `p95WallMs ${fmt(metrics.p95WallMs)} > ${fmt(thresholds.maxP95WallMs)}.`,\n })\n }\n}\n\nfunction buildAxes(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n issues: ReleaseConfidenceIssue[],\n): ReleaseConfidenceAxis[] {\n return [\n axis(\n 'corpus',\n issues,\n bounded(metrics.scenarioCount / Math.max(1, thresholds.minScenarioCount)),\n `${metrics.scenarioCount} scenarios; holdout=${metrics.splitCounts.holdout}`,\n ),\n axis(\n 'quality',\n issues,\n metrics.passRate === null || metrics.meanScore === null\n ? null\n : Math.min(metrics.passRate, metrics.meanScore),\n `passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`,\n ),\n axis(\n 'reliability',\n issues,\n metrics.reliabilityRate,\n `successRate=${fmt(metrics.reliabilityRate)} terminalFailures=${metrics.terminalFailureRuns} unclassified=${metrics.unclassifiedTerminalRuns}`,\n ),\n axis(\n 'generalization',\n issues,\n gapScore(metrics.overfitGap, thresholds.maxOverfitGap),\n `holdoutRuns=${metrics.holdoutRuns} overfitGap=${fmt(metrics.overfitGap)}`,\n ),\n axis(\n 'diagnostics',\n issues,\n metrics.failedRows === 0 ? 1 : metrics.failuresWithAsi / metrics.failedRows,\n `failuresWithAsi=${metrics.failuresWithAsi}/${metrics.failedRows}`,\n ),\n axis(\n 'efficiency',\n issues,\n efficiencyScore(metrics, thresholds),\n `meanCostUsd=${fmt(metrics.meanCostUsd)} p95WallMs=${fmt(metrics.p95WallMs)}`,\n ),\n ]\n}\n\nfunction axis(\n name: ReleaseConfidenceAxisName,\n issues: ReleaseConfidenceIssue[],\n score: number | null,\n detail: string,\n): ReleaseConfidenceAxis {\n const own = issues.filter((i) => i.axis === name)\n const status = own.some((i) => i.severity === 'critical')\n ? 'fail'\n : own.length > 0\n ? 'warn'\n : 'pass'\n return { name, status, score: score === null ? null : bounded(score), detail }\n}\n\nfunction countScenarioSplits(scenarios: readonly DatasetScenario[]): Record<DatasetSplit, number> {\n const counts: Record<DatasetSplit, number> = { train: 0, dev: 0, test: 0, holdout: 0 }\n for (const scenario of scenarios) counts[scenario.split ?? 'train']++\n return counts\n}\n\nfunction countDomains(scenarios: readonly DatasetScenario[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const scenario of scenarios) {\n const domain = scenario.tags?.domain ?? scenario.tags?.category ?? 'uncategorized'\n out[domain] = (out[domain] ?? 0) + 1\n }\n return out\n}\n\nfunction countFailureClasses(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Partial<Record<FailureClass, number>> {\n const out: Partial<Record<FailureClass, number>> = {}\n for (const run of runs) {\n // Ungated: a failure-mode census counts what the runs REPORTED, which is\n // the only way a gamed run's inflated score is visible at all. It is not a\n // promotion number — `passRate` is, and that one excludes gated runs and\n // publishes the excluded count as `metrics.realnessGatedRuns`.\n if (runPassOutcome(run, threshold) === false) {\n const failureClass =\n run.failureClass !== undefined && run.failureClass !== 'success'\n ? run.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n const failureClass =\n trace.failureClass !== undefined && trace.failureClass !== 'success'\n ? trace.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction countResponsibleSurfaces(traces: readonly ReleaseTraceEvidence[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const trace of traces) {\n for (const asi of trace.asi ?? []) {\n const surface = asi.responsibleSurface ?? 'unknown'\n out[surface] = (out[surface] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction failedRows(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Array<{ hasAsi: boolean }> {\n const out: Array<{ hasAsi: boolean }> = []\n for (const run of runs) {\n if (runPassOutcome(run, threshold) === false) {\n const asiMetric = run.outcome.raw.asi\n out.push({ hasAsi: typeof asiMetric === 'number' && asiMetric > 0 })\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n out.push({ hasAsi: (trace.asi?.length ?? 0) > 0 })\n }\n }\n return out\n}\n\nfunction passOutcomeRate(outcomes: readonly (boolean | null)[]): number | null {\n const classified = outcomes.filter((outcome): outcome is boolean => outcome !== null)\n if (classified.length === 0) return null\n return classified.filter(Boolean).length / classified.length\n}\n\nfunction runPassOutcome(run: RunRecord, threshold: number): boolean | null {\n if (hasExplicitTaskFailure(run)) return false\n const score = runSplitScore(run)\n return score === undefined ? null : score >= threshold\n}\n\nfunction hasExplicitTaskFailure(run: RunRecord): boolean {\n return run.failureClass !== undefined && run.failureClass !== 'success'\n}\n\nfunction tracePassOutcome(trace: ReleaseTraceEvidence, threshold: number): boolean | null {\n if (trace.failureClass !== undefined && trace.failureClass !== 'success') return false\n if (trace.ok === false) return false\n if (isFiniteNumber(trace.score)) return trace.score >= threshold\n return trace.ok === true ? true : null\n}\n\nfunction isFailedTerminalOutcome(\n outcome: RunRecord['terminalOutcome'],\n): outcome is 'failed' | 'cancelled' | 'incomplete' {\n return outcome === 'failed' || outcome === 'cancelled' || outcome === 'incomplete'\n}\n\nfunction terminalSuccess(outcome: RunRecord['terminalOutcome']): boolean | undefined {\n if (outcome === 'succeeded') return true\n if (isFailedTerminalOutcome(outcome)) return false\n return undefined\n}\n\nfunction scoresFor(runs: readonly RunRecord[], split: RunSplitTag): number[] {\n return runs\n .filter((run) => run.splitTag === split)\n .map(runSplitScore)\n .filter(isFiniteNumber)\n}\n\n/**\n * RAW and split-exact (`observedSplitScore`): this feeds the per-split means,\n * the overfit gap, and the pass threshold — descriptions of what the runs\n * reported. A gamed run inflating them is visible next to `realnessGatedRuns`,\n * and the promotion number (`passRate`) excludes gated runs entirely.\n */\nfunction runSplitScore(run: RunRecord): number | undefined {\n const score = observedSplitScore(run, run.splitTag === 'holdout' ? 'holdout' : 'search')\n return isFiniteNumber(score) ? score : undefined\n}\n\nfunction meanOrNull(xs: readonly number[]): number | null {\n if (xs.length === 0) return null\n return xs.reduce((sum, x) => sum + x, 0) / xs.length\n}\n\nfunction percentileOrNull(xs: readonly number[], p: number): number | null {\n if (xs.length === 0) return null\n const sorted = [...xs].sort((a, b) => a - b)\n return sorted[Math.min(sorted.length - 1, Math.max(0, Math.ceil(p * sorted.length) - 1))]!\n}\n\nfunction isFiniteNumber(value: unknown): value is number {\n return typeof value === 'number' && Number.isFinite(value)\n}\n\nfunction diffOrNull(a: number | null, b: number | null): number | null {\n if (a === null || b === null) return null\n return a - b\n}\n\nfunction gapScore(gap: number | null, maxGap: number): number | null {\n if (gap === null) return null\n if (maxGap <= 0) return gap <= 0 ? 1 : 0\n return bounded(1 - Math.max(0, gap) / maxGap)\n}\n\nfunction efficiencyScore(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n): number | null {\n const cost = Number.isFinite(thresholds.maxMeanCostUsd)\n ? metrics.meanCostUsd === null\n ? null\n : bounded(thresholds.maxMeanCostUsd / Math.max(metrics.meanCostUsd, 1e-12))\n : 1\n const latency = Number.isFinite(thresholds.maxP95WallMs)\n ? metrics.p95WallMs === null\n ? null\n : bounded(thresholds.maxP95WallMs / Math.max(metrics.p95WallMs, 1e-12))\n : 1\n if (cost === null || latency === null) return null\n return Math.min(cost, latency)\n}\n\nfunction bounded(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n\nfunction renderSummary(\n target: string,\n status: ReleaseConfidenceStatus,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): string {\n const prefix = `release confidence ${status}: ${target}`\n const metricText = `scenarios=${metrics.scenarioCount} searchRuns=${metrics.searchRuns} holdoutRuns=${metrics.holdoutRuns} passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`\n if (issues.length === 0) return `${prefix}; ${metricText}`\n return `${prefix}; ${metricText}; issues=${issues.map((i) => i.code).join(',')}`\n}\n\nfunction fmt(x: number | null): string {\n if (x === null) return 'n/a'\n return x.toFixed(4)\n}\n","/**\n * Bootstrap-CI promotion gate.\n *\n * In any iterative-improvement loop (GEPA, prompt evolution, dataset\n * curation), the question is \"did this generation actually improve, or are\n * we celebrating noise?\". With small N and noisy outcomes, point-estimate\n * deltas lie. Bootstrap confidence intervals tell the operator whether the\n * delta is real before code or prompts get promoted.\n *\n * This module is pure functions — no I/O, no model calls. Easy to unit-test\n * and to compose into any verdict gate.\n *\n * Default gate:\n * - Bootstrap mean baseline vs candidate (1k resamples).\n * - Compute the delta distribution; pass if the lower CI bound > 0.\n * - Tunable confidence (default 95%) and resample count.\n *\n * Verdict semantics intentionally match the existing `experiments.jsonl`\n * vocabulary:\n * - ADVANCE: candidate's CI lower bound > baseline mean (real win)\n * - KEEP: overlap, but candidate point estimate >= baseline (neutral)\n * - REVERT: candidate's CI upper bound < baseline mean (real regression)\n * - INCONCLUSIVE: not enough samples or CI straddles zero with no signal\n */\n\nexport type Verdict = 'ADVANCE' | 'KEEP' | 'REVERT' | 'INCONCLUSIVE'\n\nexport interface BootstrapResult {\n baselineMean: number\n candidateMean: number\n /** candidateMean - baselineMean, point estimate. */\n delta: number\n /** Lower bound of the (1 - alpha) CI on the delta. */\n ciLower: number\n /** Upper bound of the (1 - alpha) CI on the delta. */\n ciUpper: number\n /** Number of bootstrap resamples used. */\n iterations: number\n alpha: number\n verdict: Verdict\n}\n\nexport interface BootstrapOptions {\n /** Confidence level alpha (default 0.05 → 95% CI). */\n alpha?: number\n /** Number of resamples (default 1000). */\n iterations?: number\n /**\n * Minimum total samples (baseline + candidate) below which we always\n * return INCONCLUSIVE — bootstrap with too few samples is meaningless.\n * Default 6 (combined).\n */\n minTotalSamples?: number\n /** RNG seed for reproducibility. Default: Math.random. */\n seed?: number\n}\n\n/**\n * Compute the bootstrap CI on (candidateMean - baselineMean) and a verdict.\n *\n * Uses simple percentile bootstrap on the difference of resampled means.\n * That's the standard non-parametric primitive — no distributional\n * assumptions, robust to skew, easy to reason about.\n */\nexport function bootstrapCi(\n baseline: number[],\n candidate: number[],\n options: BootstrapOptions = {},\n): BootstrapResult {\n const alpha = options.alpha ?? 0.05\n const iterations = options.iterations ?? 1000\n const minTotal = options.minTotalSamples ?? 6\n const rng = mulberry32(options.seed ?? hashSeed(baseline, candidate))\n\n const baselineMean = mean(baseline)\n const candidateMean = mean(candidate)\n const delta = candidateMean - baselineMean\n\n if (\n baseline.length + candidate.length < minTotal ||\n baseline.length === 0 ||\n candidate.length === 0\n ) {\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower: -Infinity,\n ciUpper: Infinity,\n iterations: 0,\n alpha,\n verdict: 'INCONCLUSIVE',\n }\n }\n\n const deltas: number[] = new Array(iterations)\n for (let i = 0; i < iterations; i++) {\n const bResample = resample(baseline, rng)\n const cResample = resample(candidate, rng)\n deltas[i] = mean(cResample) - mean(bResample)\n }\n deltas.sort((a, b) => a - b)\n const lowerIdx = Math.floor((alpha / 2) * iterations)\n const upperIdx = Math.floor((1 - alpha / 2) * iterations) - 1\n const ciLower = deltas[Math.max(0, lowerIdx)]!\n const ciUpper = deltas[Math.min(iterations - 1, upperIdx)]!\n\n let verdict: Verdict\n if (ciLower > 0) verdict = 'ADVANCE'\n else if (ciUpper < 0) verdict = 'REVERT'\n else if (delta >= 0) verdict = 'KEEP'\n else verdict = 'INCONCLUSIVE'\n\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower,\n ciUpper,\n iterations,\n alpha,\n verdict,\n }\n}\n\nfunction mean(xs: number[]): number {\n if (xs.length === 0) return 0\n let s = 0\n for (const x of xs) s += x\n return s / xs.length\n}\n\nfunction resample(xs: number[], rng: () => number): number[] {\n const out = new Array(xs.length)\n for (let i = 0; i < xs.length; i++) out[i] = xs[Math.floor(rng() * xs.length)]\n return out\n}\n\n/** Mulberry32 — fast deterministic PRNG. Stable across runs given the same seed. */\nfunction mulberry32(seed: number): () => number {\n let t = seed >>> 0\n return () => {\n t += 0x6d2b79f5\n let r = t\n r = Math.imul(r ^ (r >>> 15), r | 1)\n r ^= r + Math.imul(r ^ (r >>> 7), r | 61)\n return ((r ^ (r >>> 14)) >>> 0) / 4294967296\n }\n}\n\n/** Stable seed derived from the inputs — same data → same CI bounds. */\nfunction hashSeed(a: number[], b: number[]): number {\n let h = 2166136261\n for (const x of [...a, ...b]) {\n const view = new Float64Array([x])\n const bytes = new Uint8Array(view.buffer)\n for (const byte of bytes) {\n h ^= byte\n h = Math.imul(h, 16777619)\n }\n }\n return h >>> 0\n}\n\n/**\n * Judge-replay promotion gate.\n *\n * The cheap inner-loop judge that drives an evolution run is by definition\n * fast and noisy. When you're about to promote a winning variant to the\n * canonical default, you want a STRONGER judge (a more expensive model, a\n * human grader, a separately-trained reward model) to confirm the win\n * generalises beyond the inner loop.\n *\n * This helper takes raw winner + baseline outputs, scores both through the\n * stronger judge, and applies `bootstrapCi`. ADVANCE means the stronger\n * judge agrees the winner is real with the configured confidence. Doesn't\n * matter what shape your \"output\" is — pass a string, an object, anything\n * the judge can read.\n */\nexport interface JudgeReplayGateArgs<TOutput> {\n baselineOutputs: TOutput[]\n candidateOutputs: TOutput[]\n /** Stronger judge — async to allow LLM calls. Return a 0..N scalar score. */\n judge: (output: TOutput) => Promise<number> | number\n alpha?: number\n iterations?: number\n /** RNG seed for reproducibility. */\n seed?: number\n /** Maximum concurrent judge calls. Default 4. */\n judgeConcurrency?: number\n}\n\n/**\n * Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.\n */\nexport async function judgeReplayGate<TOutput>(\n args: JudgeReplayGateArgs<TOutput>,\n): Promise<BootstrapResult & { baselineSamples: number; candidateSamples: number }> {\n const concurrency = args.judgeConcurrency ?? 4\n const baselineScores = await scoreAll(args.baselineOutputs, args.judge, concurrency)\n const candidateScores = await scoreAll(args.candidateOutputs, args.judge, concurrency)\n const ci = bootstrapCi(baselineScores, candidateScores, {\n ...(args.alpha !== undefined ? { alpha: args.alpha } : {}),\n ...(args.iterations !== undefined ? { iterations: args.iterations } : {}),\n ...(args.seed !== undefined ? { seed: args.seed } : {}),\n })\n return {\n ...ci,\n baselineSamples: baselineScores.length,\n candidateSamples: candidateScores.length,\n }\n}\n\nasync function scoreAll<TOutput>(\n outputs: TOutput[],\n judge: (output: TOutput) => Promise<number> | number,\n concurrency: number,\n): Promise<number[]> {\n const results: number[] = new Array(outputs.length)\n let next = 0\n async function worker(): Promise<void> {\n while (true) {\n const i = next++\n if (i >= outputs.length) return\n const v = await judge(outputs[i]!)\n results[i] = Number.isFinite(v) ? v : 0\n }\n }\n await Promise.all(Array.from({ length: Math.max(1, concurrency) }, () => worker()))\n return results\n}\n","import type { ReleaseConfidenceScorecard } from './release-confidence'\nimport type { RunRecord } from './run-record'\nimport { summaryTable } from './summary-report'\n\nexport interface RenderReleaseReportOptions {\n title?: string\n runs?: readonly RunRecord[]\n comparator?: string\n traceAnalystFindings?: readonly string[]\n nextActions?: readonly string[]\n}\n\nexport function renderReleaseReport(\n scorecard: ReleaseConfidenceScorecard,\n options: RenderReleaseReportOptions = {},\n): string {\n const title = options.title ?? `Release Report: ${scorecard.target}`\n const lines: string[] = []\n lines.push(`# ${title}`)\n lines.push('')\n lines.push(`Status: **${scorecard.status.toUpperCase()}**`)\n lines.push(`Promote: **${scorecard.promote ? 'yes' : 'no'}**`)\n if (scorecard.candidateId) lines.push(`Candidate: \\`${scorecard.candidateId}\\``)\n if (scorecard.baselineId) lines.push(`Baseline: \\`${scorecard.baselineId}\\``)\n lines.push('')\n lines.push(scorecard.summary)\n lines.push('')\n\n lines.push('## Metrics')\n lines.push('')\n lines.push('| Metric | Value |')\n lines.push('|---|---:|')\n lines.push(`| Scenarios | ${scorecard.metrics.scenarioCount} |`)\n lines.push(`| Search runs | ${scorecard.metrics.searchRuns} |`)\n lines.push(`| Holdout runs | ${scorecard.metrics.holdoutRuns} |`)\n lines.push(`| Unscored runs | ${scorecard.metrics.unscoredRuns} |`)\n lines.push(`| Terminal failures | ${scorecard.metrics.terminalFailureRuns} |`)\n lines.push(`| Pass rate | ${pct(scorecard.metrics.passRate)} |`)\n lines.push(`| Mean score | ${num(scorecard.metrics.meanScore)} |`)\n lines.push(`| Search mean | ${num(scorecard.metrics.searchMeanScore)} |`)\n lines.push(`| Holdout mean | ${num(scorecard.metrics.holdoutMeanScore)} |`)\n lines.push(`| Overfit gap | ${num(scorecard.metrics.overfitGap)} |`)\n lines.push(`| Mean cost | $${num(scorecard.metrics.meanCostUsd)} |`)\n lines.push(`| p95 wall time | ${duration(scorecard.metrics.p95WallMs)} |`)\n lines.push('')\n\n if (scorecard.issues.length > 0) {\n lines.push('## Issues')\n lines.push('')\n for (const issue of scorecard.issues) {\n lines.push(`- **${issue.severity}** \\`${issue.code}\\` (${issue.axis}): ${issue.detail}`)\n }\n lines.push('')\n }\n\n const surfaces = entries(scorecard.metrics.responsibleSurfaceCounts)\n if (surfaces.length > 0) {\n lines.push('## Responsible Surfaces')\n lines.push('')\n for (const [surface, count] of surfaces) lines.push(`- ${surface}: ${count}`)\n lines.push('')\n }\n\n const failures = entries(scorecard.metrics.failureClassCounts)\n if (failures.length > 0) {\n lines.push('## Failure Classes')\n lines.push('')\n for (const [mode, count] of failures) lines.push(`- ${mode}: ${count}`)\n lines.push('')\n }\n\n if (options.runs && options.runs.length > 0) {\n lines.push('## Run Summary')\n lines.push('')\n lines.push(\n summaryTable([...options.runs], {\n comparator: options.comparator ?? scorecard.baselineId ?? undefined,\n split: 'holdout',\n }).markdown,\n )\n lines.push('')\n }\n\n if (options.traceAnalystFindings && options.traceAnalystFindings.length > 0) {\n lines.push('## TraceAnalyst Findings')\n lines.push('')\n for (const finding of options.traceAnalystFindings) lines.push(`- ${finding}`)\n lines.push('')\n }\n\n const nextActions = options.nextActions ?? defaultNextActions(scorecard)\n if (nextActions.length > 0) {\n lines.push('## Next Actions')\n lines.push('')\n for (const action of nextActions) lines.push(`- ${action}`)\n lines.push('')\n }\n\n return `${lines.join('\\n').trimEnd()}\\n`\n}\n\nfunction defaultNextActions(scorecard: ReleaseConfidenceScorecard): string[] {\n if (scorecard.promote) return ['Promote the candidate and keep canaries enabled.']\n return scorecard.issues\n .filter((issue) => issue.severity === 'critical')\n .map((issue) => `Resolve ${issue.code}: ${issue.detail}`)\n}\n\nfunction entries(values: Record<string, number>): Array<[string, number]> {\n return Object.entries(values)\n .filter(([, count]) => count > 0)\n .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))\n}\n\nfunction pct(value: number | null): string {\n return value === null ? 'n/a' : `${(value * 100).toFixed(1)}%`\n}\n\nfunction num(value: number | null): string {\n return value === null ? 'n/a' : value.toFixed(3)\n}\n\nfunction duration(value: number | null): string {\n return value === null ? 'n/a' : `${Math.round(value)} ms`\n}\n"],"mappings":";;;;;;AAoKA,MAAM,qBAA4D;CAChE,eAAe;CACf,kBAAkB;CAClB,eAAe;CACf,gBAAgB;CAChB,gBAAgB;CAChB,aAAa;CACb,cAAc;CACd,eAAe;CACf,gBAAgB,OAAO;CACvB,cAAc,OAAO;CACrB,uBAAuB;CACvB,uBAAuB;AACzB;AAEA,SAAgB,0BACd,OAC4B;CAC5B,MAAM,aAAa;EAAE,GAAG;EAAoB,GAAG,MAAM;CAAW;CAChE,MAAM,cAAc,MAAM,eAAe;CACzC,MAAM,OAAO,iBACV,MAAM,QAAQ,CAAC,EAAA,CAAG,IAAI,iBAAiB,GACxC,aACA,MAAM,UACR;CACA,MAAM,SAAS,sBACZ,MAAM,UAAU,CAAC,EAAA,CAAG,IAAI,4BAA4B,GACrD,aACA,MAAM,UACR;CACA,MAAM,YAAY,MAAM,aAAa,CAAC;CACtC,MAAM,gBAAgB,MAAM,SAAS,iBAAiB,UAAU;CAChE,MAAM,cAAc,MAAM,SAAS,eAAe,oBAAoB,SAAS;CAC/E,MAAM,eAAe,UAAU,MAAM,QAAQ;CAC7C,MAAM,gBAAgB,UAAU,MAAM,SAAS;CAC/C,MAAM,YAAY,KAAK,IAAI,aAAa,CAAC,CAAC,OAAO,cAAc;CAC/D,MAAM,cAAc,OAAO,KAAK,MAAM,EAAE,KAAK,CAAC,CAAC,OAAO,cAAc;CACpE,MAAM,gBAAgB,KAAK,SAAS,IAAI,YAAY;CACpD,MAAM,cAAc,KAAK,QAAQ,QAAQ,cAAc,GAAG,MAAM,KAAA,CAAS;CACzE,MAAM,eAAe,KAAK,QACvB,QACC,cAAc,GAAG,MAAM,KAAA,KACvB,CAAC,uBAAuB,GAAG,KAC3B,CAAC,wBAAwB,IAAI,eAAe,CAChD,CAAC,CAAC;CAOF,MAAM,aAAa,KAAK,QAAQ,QAAQ,CAAC,gBAAgB,GAAG,CAAC;CAC7D,MAAM,eACJ,KAAK,SAAS,IACV,WAAW,KAAK,QAAQ,eAAe,KAAK,WAAW,qBAAqB,CAAC,IAC7E,OAAO,KAAK,UAAU,iBAAiB,OAAO,WAAW,qBAAqB,CAAC;CACrF,MAAM,kBACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,gBAAgB,IAAI,eAAe,CAAC,IACtD,OAAO,KAAK,UAAU,MAAM,EAAE;CACpC,MAAM,2BAA2B,gBAAgB,QAAQ,YAAY,YAAY,KAAA,CAAS,CAAC,CAAC;CAC5F,MAAM,sBAAsB,gBAAgB,QAAQ,YAAY,YAAY,KAAK,CAAC,CAAC;CACnF,MAAM,kBACJ,gBAAgB,WAAW,KAAK,2BAA2B,IACvD,QACC,gBAAgB,SAAS,uBAAuB,gBAAgB;CACvE,MAAM,aAAa,YAAY,QAAQ,MAAM,EAAE,aAAa,QAAQ,CAAC,CAAC;CACtE,MAAM,cAAc,YAAY,QAAQ,MAAM,EAAE,aAAa,SAAS,CAAC,CAAC;CACxE,MAAM,SAAS,WAAW,MAAM,QAAQ,WAAW,qBAAqB;CACxE,MAAM,kBAAkB,WAAW,YAAY;CAC/C,MAAM,mBAAmB,WAAW,aAAa;CACjD,MAAM,WAAW,KAAK,SAAS,QAC7B,IAAI,eAAe,SAAS,eAAe,CAAC,IAAI,CAAC,IAAI,eAAe,GAAG,CACzE;CACA,MAAM,aAAa,OAAO,KAAK,UAAU,MAAM,OAAO,CAAC,CAAC,OAAO,cAAc;CAC7E,MAAM,cACJ,KAAK,SAAS,IACV,SAAS,WAAW,KAAK,SACvB,WAAW,QAAQ,IACnB,OACF,WAAW,UAAU;CAC3B,MAAM,YACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,IAAI,MAAM,IAC5B,OAAO,KAAK,UAAU,MAAM,UAAU,CAAC,CAAC,OAAO,cAAc;CACnE,MAAM,UAAoC;EACxC;EACA;EACA;EACA;EACA;EACA;EACA;EACA,UAAU,gBAAgB,YAAY;EACtC,mBAAmB,KAAK,SAAS,WAAW;EAC5C,WAAW,WAAW,aAAa;EACnC;EACA;EACA,YAAY,WAAW,iBAAiB,gBAAgB;EACxD;EACA,WAAW,iBAAiB,WAAW,GAAI;EAC3C,YAAY,OAAO;EACnB,iBAAiB,OAAO,QAAQ,QAAQ,IAAI,MAAM,CAAC,CAAC;EACpD,kBAAkB,OAAO,QAAQ,MAAM,EAAE,cAAc,CAAC,CAAC,CAAC;EAC1D,iBAAiB,OAAO,QAAQ,OAAO,EAAE,aAAa,KAAK,CAAC,CAAC,CAAC;EAC9D;EACA,cAAc,aAAa,SAAS;EACpC,oBAAoB,oBAAoB,MAAM,QAAQ,WAAW,qBAAqB;EACtF,0BAA0B,yBAAyB,MAAM;CAC3D;CAEA,MAAM,SAAmC,CAAC;CAC1C,YAAY,OAAO,YAAY,SAAS,MAAM;CAC9C,aAAa,YAAY,SAAS,MAAM;CACxC,iBAAiB,SAAS,MAAM;CAChC,oBAAoB,MAAM,gBAAgB,MAAM,YAAY,SAAS,MAAM;CAC3E,iBAAiB,YAAY,SAAS,MAAM;CAC5C,gBAAgB,YAAY,SAAS,MAAM;CAE3C,MAAM,OAAO,UAAU,SAAS,YAAY,MAAM;CAClD,MAAM,SAAS,OAAO,MAAM,MAAM,EAAE,aAAa,UAAU,IACvD,SACA,OAAO,SAAS,IACd,SACA;CAEN,OAAO;EACL,QAAQ,MAAM;EACd;EACA,YAAY,MAAM,cAAc;EAChC;EACA,SAAS,WAAW,WAAW,MAAM,eAAe,MAAM,aAAa,UAAU;EACjF;EACA;EACA;EACA,SAAS,MAAM,WAAW;EAC1B,cAAc,MAAM,gBAAgB;EACpC,SAAS,cAAc,MAAM,QAAQ,QAAQ,SAAS,MAAM;CAC9D;AACF;AAEA,SAAgB,wBAAwB,OAA2D;CACjG,MAAM,YAAY,0BAA0B,KAAK;CACjD,IAAI,UAAU,WAAW,QACvB,MAAM,IAAI,kBAAkB,UAAU,OAAO;CAE/C,OAAO;AACT;AAEA,SAAS,gBACP,MACA,aACA,YACa;CACb,IAAI,aAAa,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,WAAW;CACxE,IAAI,YAAY,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,UAAU;CACtE,OAAO,CAAC,GAAG,IAAI;AACjB;AAEA,SAAS,qBACP,QACA,aACA,YACwB;CACxB,IAAI,aACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,WAAW;CAC1F,IAAI,YACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,UAAU;CACzF,OAAO,CAAC,GAAG,MAAM;AACnB;AAEA,SAAS,6BACP,OACA,OACsB;CACtB,MAAM,QAAQ;CACd,IAAI,OAAO,OAAO,OAAO,aAAa,GACpC,MAAM,IAAI,gBACR,UAAU,MAAM,2DAClB;CAEF,IAAI,MAAM,iBAAiB,KAAA,KAAa,CAAC,gBAAgB,SAAS,MAAM,YAAY,GAClF,MAAM,IAAI,gBACR,UAAU,MAAM,gCAAgC,gBAAgB,KAAK,IAAI,GAC3E;CAEF,OAAO;AACT;AAEA,SAAS,YACP,OACA,YACA,SACA,QACM;CACN,IAAI,WAAW,iBAAiB,CAAC,MAAM,YAAY,MAAM,WAAW,UAAU,OAAO,GACnF,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,gBAAgB,WAAW,kBACrC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,cAAc,qBAAqB,WAAW,iBAAiB;CACpF,CAAC;CAEH,IAAI,WAAW,kBAAkB,QAAQ,YAAY,YAAY,GAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;AAEL;AAEA,SAAS,aACP,YACA,SACA,QACM;CACN,IAAI,QAAQ,aAAa,WAAW,eAClC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,WAAW,uBAAuB,WAAW,cAAc;CAChF,CAAC;CAEH,IAAI,QAAQ,eAAe,GACzB,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa;CAClC,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,cAAc,MACrD,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,WAAW,WAAW,aAC7D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,YAAY,IAAI,QAAQ,QAAQ,EAAE,KAAK,IAAI,WAAW,WAAW,EAAE;CAC7E,CAAC;CAEH,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,iBACP,SACA,QACM;CACN,IAAI,QAAQ,oBAAoB,MAC9B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QACE,QAAQ,2BAA2B,IAC/B,GAAG,QAAQ,yBAAyB,wDACpC;CACR,CAAC;CAEH,IAAI,QAAQ,sBAAsB,GAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,oBAAoB;CACzC,CAAC;AAEL;AAEA,SAAS,oBACP,cACA,YACA,SACA,QACM;CACN,IAAI,WAAW,kBAAkB,QAAQ,cAAc,WAAW,gBAChE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,YAAY,wBAAwB,WAAW,eAAe;CACnF,CAAC;CAEH,IAAI,QAAQ,eAAe,QAAQ,QAAQ,aAAa,WAAW,eACjE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,sBAAsB,IAAI,QAAQ,UAAU,EAAE,KAAK,IAAI,WAAW,aAAa,EAAE;CAC3F,CAAC;CAEH,IAAI,gBAAgB,CAAC,aAAa,SAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM,QAAQ,aAAa,iBAAiB;EAC5C,QAAQ,aAAa;CACvB,CAAC;AAEL;AAEA,SAAS,iBACP,YACA,SACA,QACM;CACN,IAAI,CAAC,WAAW,uBAAuB;CACvC,IAAI,QAAQ,aAAa,QAAQ,iBAC/B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa,QAAQ,gBAAgB;CAC1D,CAAC;AAEL;AAEA,SAAS,gBACP,YACA,SACA,QACM;CACN,IAAI,OAAO,SAAS,WAAW,cAAc,KAAK,QAAQ,gBAAgB,MACxE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,gBAAgB,QAAQ,QAAQ,cAAc,WAAW,gBAC1E,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,eAAe,IAAI,QAAQ,WAAW,EAAE,KAAK,IAAI,WAAW,cAAc,EAAE;CACtF,CAAC;CAEH,IAAI,OAAO,SAAS,WAAW,YAAY,KAAK,QAAQ,cAAc,MACpE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cACtE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,UACP,SACA,YACA,QACyB;CACzB,OAAO;EACL,KACE,UACA,QACA,QAAQ,QAAQ,gBAAgB,KAAK,IAAI,GAAG,WAAW,gBAAgB,CAAC,GACxE,GAAG,QAAQ,cAAc,sBAAsB,QAAQ,YAAY,SACrE;EACA,KACE,WACA,QACA,QAAQ,aAAa,QAAQ,QAAQ,cAAc,OAC/C,OACA,KAAK,IAAI,QAAQ,UAAU,QAAQ,SAAS,GAChD,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS,GACtE;EACA,KACE,eACA,QACA,QAAQ,iBACR,eAAe,IAAI,QAAQ,eAAe,EAAE,oBAAoB,QAAQ,oBAAoB,gBAAgB,QAAQ,0BACtH;EACA,KACE,kBACA,QACA,SAAS,QAAQ,YAAY,WAAW,aAAa,GACrD,eAAe,QAAQ,YAAY,cAAc,IAAI,QAAQ,UAAU,GACzE;EACA,KACE,eACA,QACA,QAAQ,eAAe,IAAI,IAAI,QAAQ,kBAAkB,QAAQ,YACjE,mBAAmB,QAAQ,gBAAgB,GAAG,QAAQ,YACxD;EACA,KACE,cACA,QACA,gBAAgB,SAAS,UAAU,GACnC,eAAe,IAAI,QAAQ,WAAW,EAAE,aAAa,IAAI,QAAQ,SAAS,GAC5E;CACF;AACF;AAEA,SAAS,KACP,MACA,QACA,OACA,QACuB;CACvB,MAAM,MAAM,OAAO,QAAQ,MAAM,EAAE,SAAS,IAAI;CAMhD,OAAO;EAAE;EAAM,QALA,IAAI,MAAM,MAAM,EAAE,aAAa,UAAU,IACpD,SACA,IAAI,SAAS,IACX,SACA;EACiB,OAAO,UAAU,OAAO,OAAO,QAAQ,KAAK;EAAG;CAAO;AAC/E;AAEA,SAAS,oBAAoB,WAAqE;CAChG,MAAM,SAAuC;EAAE,OAAO;EAAG,KAAK;EAAG,MAAM;EAAG,SAAS;CAAE;CACrF,KAAK,MAAM,YAAY,WAAW,OAAO,SAAS,SAAS,QAAQ;CACnE,OAAO;AACT;AAEA,SAAS,aAAa,WAA+D;CACnF,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,YAAY,WAAW;EAChC,MAAM,SAAS,SAAS,MAAM,UAAU,SAAS,MAAM,YAAY;EACnE,IAAI,WAAW,IAAI,WAAW,KAAK;CACrC;CACA,OAAO;AACT;AAEA,SAAS,oBACP,MACA,QACA,WACuC;CACvC,MAAM,MAA6C,CAAC;CACpD,KAAK,MAAM,OAAO,MAKhB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,eACJ,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,YACnD,IAAI,eACJ;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OAAO;EAChD,MAAM,eACJ,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,YACvD,MAAM,eACN;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,QAAiE;CACjG,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,SAAS,QAClB,KAAK,MAAM,OAAO,MAAM,OAAO,CAAC,GAAG;EACjC,MAAM,UAAU,IAAI,sBAAsB;EAC1C,IAAI,YAAY,IAAI,YAAY,KAAK;CACvC;CAEF,OAAO;AACT;AAEA,SAAS,WACP,MACA,QACA,WAC4B;CAC5B,MAAM,MAAkC,CAAC;CACzC,KAAK,MAAM,OAAO,MAChB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,YAAY,IAAI,QAAQ,IAAI;EAClC,IAAI,KAAK,EAAE,QAAQ,OAAO,cAAc,YAAY,YAAY,EAAE,CAAC;CACrE;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OACzC,IAAI,KAAK,EAAE,SAAS,MAAM,KAAK,UAAU,KAAK,EAAE,CAAC;CAGrD,OAAO;AACT;AAEA,SAAS,gBAAgB,UAAsD;CAC7E,MAAM,aAAa,SAAS,QAAQ,YAAgC,YAAY,IAAI;CACpF,IAAI,WAAW,WAAW,GAAG,OAAO;CACpC,OAAO,WAAW,OAAO,OAAO,CAAC,CAAC,SAAS,WAAW;AACxD;AAEA,SAAS,eAAe,KAAgB,WAAmC;CACzE,IAAI,uBAAuB,GAAG,GAAG,OAAO;CACxC,MAAM,QAAQ,cAAc,GAAG;CAC/B,OAAO,UAAU,KAAA,IAAY,OAAO,SAAS;AAC/C;AAEA,SAAS,uBAAuB,KAAyB;CACvD,OAAO,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB;AAChE;AAEA,SAAS,iBAAiB,OAA6B,WAAmC;CACxF,IAAI,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,WAAW,OAAO;CACjF,IAAI,MAAM,OAAO,OAAO,OAAO;CAC/B,IAAI,eAAe,MAAM,KAAK,GAAG,OAAO,MAAM,SAAS;CACvD,OAAO,MAAM,OAAO,OAAO,OAAO;AACpC;AAEA,SAAS,wBACP,SACkD;CAClD,OAAO,YAAY,YAAY,YAAY,eAAe,YAAY;AACxE;AAEA,SAAS,gBAAgB,SAA4D;CACnF,IAAI,YAAY,aAAa,OAAO;CACpC,IAAI,wBAAwB,OAAO,GAAG,OAAO;AAE/C;AAEA,SAAS,UAAU,MAA4B,OAA8B;CAC3E,OAAO,KACJ,QAAQ,QAAQ,IAAI,aAAa,KAAK,CAAC,CACvC,IAAI,aAAa,CAAC,CAClB,OAAO,cAAc;AAC1B;;;;;;;AAQA,SAAS,cAAc,KAAoC;CACzD,MAAM,QAAQ,mBAAmB,KAAK,IAAI,aAAa,YAAY,YAAY,QAAQ;CACvF,OAAO,eAAe,KAAK,IAAI,QAAQ,KAAA;AACzC;AAEA,SAAS,WAAW,IAAsC;CACxD,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,OAAO,GAAG,QAAQ,KAAK,MAAM,MAAM,GAAG,CAAC,IAAI,GAAG;AAChD;AAEA,SAAS,iBAAiB,IAAuB,GAA0B;CACzE,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,MAAM,SAAS,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,OAAO,OAAO,KAAK,IAAI,OAAO,SAAS,GAAG,KAAK,IAAI,GAAG,KAAK,KAAK,IAAI,OAAO,MAAM,IAAI,CAAC,CAAC;AACzF;AAEA,SAAS,eAAe,OAAiC;CACvD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK;AAC3D;AAEA,SAAS,WAAW,GAAkB,GAAiC;CACrE,IAAI,MAAM,QAAQ,MAAM,MAAM,OAAO;CACrC,OAAO,IAAI;AACb;AAEA,SAAS,SAAS,KAAoB,QAA+B;CACnE,IAAI,QAAQ,MAAM,OAAO;CACzB,IAAI,UAAU,GAAG,OAAO,OAAO,IAAI,IAAI;CACvC,OAAO,QAAQ,IAAI,KAAK,IAAI,GAAG,GAAG,IAAI,MAAM;AAC9C;AAEA,SAAS,gBACP,SACA,YACe;CACf,MAAM,OAAO,OAAO,SAAS,WAAW,cAAc,IAClD,QAAQ,gBAAgB,OACtB,OACA,QAAQ,WAAW,iBAAiB,KAAK,IAAI,QAAQ,aAAa,KAAK,CAAC,IAC1E;CACJ,MAAM,UAAU,OAAO,SAAS,WAAW,YAAY,IACnD,QAAQ,cAAc,OACpB,OACA,QAAQ,WAAW,eAAe,KAAK,IAAI,QAAQ,WAAW,KAAK,CAAC,IACtE;CACJ,IAAI,SAAS,QAAQ,YAAY,MAAM,OAAO;CAC9C,OAAO,KAAK,IAAI,MAAM,OAAO;AAC/B;AAEA,SAAS,QAAQ,GAAmB;CAClC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;AAEA,SAAS,cACP,QACA,QACA,SACA,QACQ;CACR,MAAM,SAAS,sBAAsB,OAAO,IAAI;CAChD,MAAM,aAAa,aAAa,QAAQ,cAAc,cAAc,QAAQ,WAAW,eAAe,QAAQ,YAAY,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS;CAC9L,IAAI,OAAO,WAAW,GAAG,OAAO,GAAG,OAAO,IAAI;CAC9C,OAAO,GAAG,OAAO,IAAI,WAAW,WAAW,OAAO,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,GAAG;AAC/E;AAEA,SAAS,IAAI,GAA0B;CACrC,IAAI,MAAM,MAAM,OAAO;CACvB,OAAO,EAAE,QAAQ,CAAC;AACpB;;;;;;;;;;AC9tBA,SAAgB,YACd,UACA,WACA,UAA4B,CAAC,GACZ;CACjB,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,aAAa,QAAQ,cAAc;CACzC,MAAM,WAAW,QAAQ,mBAAmB;CAC5C,MAAM,MAAM,WAAW,QAAQ,QAAQ,SAAS,UAAU,SAAS,CAAC;CAEpE,MAAM,eAAe,KAAK,QAAQ;CAClC,MAAM,gBAAgB,KAAK,SAAS;CACpC,MAAM,QAAQ,gBAAgB;CAE9B,IACE,SAAS,SAAS,UAAU,SAAS,YACrC,SAAS,WAAW,KACpB,UAAU,WAAW,GAErB,OAAO;EACL;EACA;EACA;EACA,SAAS;EACT,SAAS;EACT,YAAY;EACZ;EACA,SAAS;CACX;CAGF,MAAM,SAAmB,IAAI,MAAM,UAAU;CAC7C,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,YAAY,SAAS,UAAU,GAAG;EAExC,OAAO,KAAK,KADM,SAAS,WAAW,GACb,CAAC,IAAI,KAAK,SAAS;CAC9C;CACA,OAAO,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3B,MAAM,WAAW,KAAK,MAAO,QAAQ,IAAK,UAAU;CACpD,MAAM,WAAW,KAAK,OAAO,IAAI,QAAQ,KAAK,UAAU,IAAI;CAC5D,MAAM,UAAU,OAAO,KAAK,IAAI,GAAG,QAAQ;CAC3C,MAAM,UAAU,OAAO,KAAK,IAAI,aAAa,GAAG,QAAQ;CAExD,IAAI;CACJ,IAAI,UAAU,GAAG,UAAU;MACtB,IAAI,UAAU,GAAG,UAAU;MAC3B,IAAI,SAAS,GAAG,UAAU;MAC1B,UAAU;CAEf,OAAO;EACL;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,IAAI,KAAK;CACzB,OAAO,IAAI,GAAG;AAChB;AAEA,SAAS,SAAS,IAAc,KAA6B;CAC3D,MAAM,MAAM,IAAI,MAAM,GAAG,MAAM;CAC/B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,QAAQ,KAAK,IAAI,KAAK,GAAG,KAAK,MAAM,IAAI,IAAI,GAAG,MAAM;CAC5E,OAAO;AACT;;AAGA,SAAS,WAAW,MAA4B;CAC9C,IAAI,IAAI,SAAS;CACjB,aAAa;EACX,KAAK;EACL,IAAI,IAAI;EACR,IAAI,KAAK,KAAK,IAAK,MAAM,IAAK,IAAI,CAAC;EACnC,KAAK,IAAI,KAAK,KAAK,IAAK,MAAM,GAAI,IAAI,EAAE;EACxC,SAAS,IAAK,MAAM,QAAS,KAAK;CACpC;AACF;;AAGA,SAAS,SAAS,GAAa,GAAqB;CAClD,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,CAAC,GAAG,GAAG,GAAG,CAAC,GAAG;EAC5B,MAAM,OAAO,IAAI,aAAa,CAAC,CAAC,CAAC;EACjC,MAAM,QAAQ,IAAI,WAAW,KAAK,MAAM;EACxC,KAAK,MAAM,QAAQ,OAAO;GACxB,KAAK;GACL,IAAI,KAAK,KAAK,GAAG,QAAQ;EAC3B;CACF;CACA,OAAO,MAAM;AACf;;;;AAiCA,eAAsB,gBACpB,MACkF;CAClF,MAAM,cAAc,KAAK,oBAAoB;CAC7C,MAAM,iBAAiB,MAAM,SAAS,KAAK,iBAAiB,KAAK,OAAO,WAAW;CACnF,MAAM,kBAAkB,MAAM,SAAS,KAAK,kBAAkB,KAAK,OAAO,WAAW;CAMrF,OAAO;EACL,GANS,YAAY,gBAAgB,iBAAiB;GACtD,GAAI,KAAK,UAAU,KAAA,IAAY,EAAE,OAAO,KAAK,MAAM,IAAI,CAAC;GACxD,GAAI,KAAK,eAAe,KAAA,IAAY,EAAE,YAAY,KAAK,WAAW,IAAI,CAAC;GACvE,GAAI,KAAK,SAAS,KAAA,IAAY,EAAE,MAAM,KAAK,KAAK,IAAI,CAAC;EACvD,CAEM;EACJ,iBAAiB,eAAe;EAChC,kBAAkB,gBAAgB;CACpC;AACF;AAEA,eAAe,SACb,SACA,OACA,aACmB;CACnB,MAAM,UAAoB,IAAI,MAAM,QAAQ,MAAM;CAClD,IAAI,OAAO;CACX,eAAe,SAAwB;EACrC,OAAO,MAAM;GACX,MAAM,IAAI;GACV,IAAI,KAAK,QAAQ,QAAQ;GACzB,MAAM,IAAI,MAAM,MAAM,QAAQ,EAAG;GACjC,QAAQ,KAAK,OAAO,SAAS,CAAC,IAAI,IAAI;EACxC;CACF;CACA,MAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,GAAG,WAAW,EAAE,SAAS,OAAO,CAAC,CAAC;CAClF,OAAO;AACT;;;AC1NA,SAAgB,oBACd,WACA,UAAsC,CAAC,GAC/B;CACR,MAAM,QAAQ,QAAQ,SAAS,mBAAmB,UAAU;CAC5D,MAAM,QAAkB,CAAC;CACzB,MAAM,KAAK,KAAK,OAAO;CACvB,MAAM,KAAK,EAAE;CACb,MAAM,KAAK,aAAa,UAAU,OAAO,YAAY,EAAE,GAAG;CAC1D,MAAM,KAAK,cAAc,UAAU,UAAU,QAAQ,KAAK,GAAG;CAC7D,IAAI,UAAU,aAAa,MAAM,KAAK,gBAAgB,UAAU,YAAY,GAAG;CAC/E,IAAI,UAAU,YAAY,MAAM,KAAK,eAAe,UAAU,WAAW,GAAG;CAC5E,MAAM,KAAK,EAAE;CACb,MAAM,KAAK,UAAU,OAAO;CAC5B,MAAM,KAAK,EAAE;CAEb,MAAM,KAAK,YAAY;CACvB,MAAM,KAAK,EAAE;CACb,MAAM,KAAK,oBAAoB;CAC/B,MAAM,KAAK,YAAY;CACvB,MAAM,KAAK,iBAAiB,UAAU,QAAQ,cAAc,GAAG;CAC/D,MAAM,KAAK,mBAAmB,UAAU,QAAQ,WAAW,GAAG;CAC9D,MAAM,KAAK,oBAAoB,UAAU,QAAQ,YAAY,GAAG;CAChE,MAAM,KAAK,qBAAqB,UAAU,QAAQ,aAAa,GAAG;CAClE,MAAM,KAAK,yBAAyB,UAAU,QAAQ,oBAAoB,GAAG;CAC7E,MAAM,KAAK,iBAAiB,IAAI,UAAU,QAAQ,QAAQ,EAAE,GAAG;CAC/D,MAAM,KAAK,kBAAkB,IAAI,UAAU,QAAQ,SAAS,EAAE,GAAG;CACjE,MAAM,KAAK,mBAAmB,IAAI,UAAU,QAAQ,eAAe,EAAE,GAAG;CACxE,MAAM,KAAK,oBAAoB,IAAI,UAAU,QAAQ,gBAAgB,EAAE,GAAG;CAC1E,MAAM,KAAK,mBAAmB,IAAI,UAAU,QAAQ,UAAU,EAAE,GAAG;CACnE,MAAM,KAAK,kBAAkB,IAAI,UAAU,QAAQ,WAAW,EAAE,GAAG;CACnE,MAAM,KAAK,qBAAqB,SAAS,UAAU,QAAQ,SAAS,EAAE,GAAG;CACzE,MAAM,KAAK,EAAE;CAEb,IAAI,UAAU,OAAO,SAAS,GAAG;EAC/B,MAAM,KAAK,WAAW;EACtB,MAAM,KAAK,EAAE;EACb,KAAK,MAAM,SAAS,UAAU,QAC5B,MAAM,KAAK,OAAO,MAAM,SAAS,OAAO,MAAM,KAAK,MAAM,MAAM,KAAK,KAAK,MAAM,QAAQ;EAEzF,MAAM,KAAK,EAAE;CACf;CAEA,MAAM,WAAW,QAAQ,UAAU,QAAQ,wBAAwB;CACnE,IAAI,SAAS,SAAS,GAAG;EACvB,MAAM,KAAK,yBAAyB;EACpC,MAAM,KAAK,EAAE;EACb,KAAK,MAAM,CAAC,SAAS,UAAU,UAAU,MAAM,KAAK,KAAK,QAAQ,IAAI,OAAO;EAC5E,MAAM,KAAK,EAAE;CACf;CAEA,MAAM,WAAW,QAAQ,UAAU,QAAQ,kBAAkB;CAC7D,IAAI,SAAS,SAAS,GAAG;EACvB,MAAM,KAAK,oBAAoB;EAC/B,MAAM,KAAK,EAAE;EACb,KAAK,MAAM,CAAC,MAAM,UAAU,UAAU,MAAM,KAAK,KAAK,KAAK,IAAI,OAAO;EACtE,MAAM,KAAK,EAAE;CACf;CAEA,IAAI,QAAQ,QAAQ,QAAQ,KAAK,SAAS,GAAG;EAC3C,MAAM,KAAK,gBAAgB;EAC3B,MAAM,KAAK,EAAE;EACb,MAAM,KACJ,aAAa,CAAC,GAAG,QAAQ,IAAI,GAAG;GAC9B,YAAY,QAAQ,cAAc,UAAU,cAAc,KAAA;GAC1D,OAAO;EACT,CAAC,CAAC,CAAC,QACL;EACA,MAAM,KAAK,EAAE;CACf;CAEA,IAAI,QAAQ,wBAAwB,QAAQ,qBAAqB,SAAS,GAAG;EAC3E,MAAM,KAAK,0BAA0B;EACrC,MAAM,KAAK,EAAE;EACb,KAAK,MAAM,WAAW,QAAQ,sBAAsB,MAAM,KAAK,KAAK,SAAS;EAC7E,MAAM,KAAK,EAAE;CACf;CAEA,MAAM,cAAc,QAAQ,eAAe,mBAAmB,SAAS;CACvE,IAAI,YAAY,SAAS,GAAG;EAC1B,MAAM,KAAK,iBAAiB;EAC5B,MAAM,KAAK,EAAE;EACb,KAAK,MAAM,UAAU,aAAa,MAAM,KAAK,KAAK,QAAQ;EAC1D,MAAM,KAAK,EAAE;CACf;CAEA,OAAO,GAAG,MAAM,KAAK,IAAI,CAAC,CAAC,QAAQ,EAAE;AACvC;AAEA,SAAS,mBAAmB,WAAiD;CAC3E,IAAI,UAAU,SAAS,OAAO,CAAC,kDAAkD;CACjF,OAAO,UAAU,OACd,QAAQ,UAAU,MAAM,aAAa,UAAU,CAAC,CAChD,KAAK,UAAU,WAAW,MAAM,KAAK,IAAI,MAAM,QAAQ;AAC5D;AAEA,SAAS,QAAQ,QAAyD;CACxE,OAAO,OAAO,QAAQ,MAAM,CAAC,CAC1B,QAAQ,GAAG,WAAW,QAAQ,CAAC,CAAC,CAChC,MAAM,GAAG,MAAM,EAAE,KAAK,EAAE,MAAM,EAAE,EAAE,CAAC,cAAc,EAAE,EAAE,CAAC;AAC3D;AAEA,SAAS,IAAI,OAA8B;CACzC,OAAO,UAAU,OAAO,QAAQ,IAAI,QAAQ,IAAA,CAAK,QAAQ,CAAC,EAAE;AAC9D;AAEA,SAAS,IAAI,OAA8B;CACzC,OAAO,UAAU,OAAO,QAAQ,MAAM,QAAQ,CAAC;AACjD;AAEA,SAAS,SAAS,OAA8B;CAC9C,OAAO,UAAU,OAAO,QAAQ,GAAG,KAAK,MAAM,KAAK,EAAE;AACvD"}
1
+ {"version":3,"file":"release-report-BVZBmRZp.js","names":[],"sources":["../src/release-confidence.ts","../src/promotion-gate.ts","../src/release-report.ts"],"sourcesContent":["/**\n * Release confidence gate.\n *\n * This is the production-facing composition layer over the lower-level\n * primitives:\n * - Dataset manifests prove corpus/version coverage.\n * - RunRecord rows prove reproducible search/holdout outcomes.\n * - Multi-shot trace evidence carries turn counts and ASI diagnostics.\n * - HeldOutGate decisions remain the paired promotion authority.\n *\n * The gate is intentionally pure and conservative. Missing declared evidence\n * fails closed instead of being treated as a neutral zero.\n */\n\nimport type { DatasetManifest, DatasetScenario, DatasetSplit } from './dataset'\nimport { ValidationError, VerificationError } from './errors'\nimport type { GateDecision } from './held-out-gate'\nimport { isRealnessGated, observedSplitScore } from './rollout/reward'\nimport { type RunRecord, type RunSplitTag, validateRunRecord } from './run-record'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Severity of an actionable finding attached to a run/trace. */\nexport type AsiSeverity = 'info' | 'warning' | 'error' | 'critical'\n\n/** Actionable side-info — a diagnosed finding the loop can act on. */\nexport interface ActionableSideInfo {\n /** Stable expectation/check id when available. */\n expectationId?: string\n /** Human-readable diagnosis of what happened. */\n message: string\n severity?: AsiSeverity\n /** Concrete trace excerpt, file path, tool call, screenshot id, etc. */\n evidence?: string\n /** Prompt/tool/context surface likely responsible. */\n responsibleSurface?: string\n /** Suggested fix in natural language. */\n suggestion?: string\n /** Whether this expectation was satisfied. Defaults to false for ASI rows. */\n matched?: boolean\n metadata?: Record<string, unknown>\n}\n\nexport type ReleaseConfidenceStatus = 'pass' | 'warn' | 'fail'\nexport type ReleaseConfidenceAxisName =\n | 'corpus'\n | 'quality'\n | 'reliability'\n | 'generalization'\n | 'diagnostics'\n | 'efficiency'\n\nexport interface ReleaseTraceEvidence {\n scenarioId: string\n candidateId?: string\n split?: RunSplitTag\n score?: number\n ok?: boolean\n turnCount?: number\n costUsd?: number\n durationMs?: number\n /** Canonical task-failure class. Free-form detail belongs in ASI. */\n failureClass?: FailureClass\n asi?: ActionableSideInfo[]\n metadata?: Record<string, unknown>\n}\n\nexport interface ReleaseConfidenceThresholds {\n /** Require a Dataset manifest or explicit scenarios. Default true. */\n requireCorpus?: boolean\n minScenarioCount?: number\n minSearchRuns?: number\n minHoldoutRuns?: number\n /** Require at least one holdout scenario/run. Default true. */\n requireHoldout?: boolean\n minPassRate?: number\n minMeanScore?: number\n /** Search mean may exceed holdout mean by at most this much. */\n maxOverfitGap?: number\n maxMeanCostUsd?: number\n maxP95WallMs?: number\n /** Low-score/failed rows must carry ASI. Default true. */\n requireAsiForFailures?: boolean\n /** Score below this is considered a failure for ASI coverage. Default 0.5. */\n failureScoreThreshold?: number\n}\n\nexport interface ReleaseConfidenceInput {\n target: string\n candidateId?: string\n baselineId?: string\n dataset?: DatasetManifest\n scenarios?: readonly DatasetScenario[]\n runs?: readonly RunRecord[]\n traces?: readonly ReleaseTraceEvidence[]\n gateDecision?: GateDecision | null\n thresholds?: ReleaseConfidenceThresholds\n}\n\nexport interface ReleaseConfidenceAxis {\n name: ReleaseConfidenceAxisName\n status: ReleaseConfidenceStatus\n score: number | null\n detail: string\n}\n\nexport interface ReleaseConfidenceIssue {\n axis: ReleaseConfidenceAxisName\n severity: 'critical' | 'warning'\n code: string\n detail: string\n}\n\nexport interface ReleaseConfidenceMetrics {\n scenarioCount: number\n /** Search rows with a finite search score. */\n searchRuns: number\n /** Holdout rows with a finite holdout score. */\n holdoutRuns: number\n /** Runs with neither a split-matched score nor an explicit task failure. */\n unscoredRuns: number\n /** Run rows, or trace rows when no runs exist, with no classified terminal result. */\n unclassifiedTerminalRuns: number\n /** Run rows, or trace rows when no runs exist, that ended unsuccessfully. */\n terminalFailureRuns: number\n /** Success fraction when every run or fallback trace row has a classified result. */\n reliabilityRate: number | null\n passRate: number | null\n meanScore: number | null\n searchMeanScore: number | null\n holdoutMeanScore: number | null\n overfitGap: number | null\n meanCostUsd: number | null\n p95WallMs: number | null\n failedRows: number\n failuresWithAsi: number\n singleShotTraces: number\n multiShotTraces: number\n splitCounts: Record<DatasetSplit, number>\n domainCounts: Record<string, number>\n failureClassCounts: Partial<Record<FailureClass, number>>\n responsibleSurfaceCounts: Record<string, number>\n /**\n * Runs excluded from `passRate` because the authenticity gate flagged them as\n * gamed. Surfaced, never silent: a release whose pass rate is computed over a\n * shrunken denominator has to say by how much, or the exclusion is just a\n * different way of hiding the same runs.\n */\n realnessGatedRuns: number\n}\n\nexport interface ReleaseConfidenceScorecard {\n target: string\n candidateId: string | null\n baselineId: string | null\n status: ReleaseConfidenceStatus\n promote: boolean\n axes: ReleaseConfidenceAxis[]\n issues: ReleaseConfidenceIssue[]\n metrics: ReleaseConfidenceMetrics\n dataset: DatasetManifest | null\n gateDecision: GateDecision | null\n summary: string\n}\n\nconst DEFAULT_THRESHOLDS: Required<ReleaseConfidenceThresholds> = {\n requireCorpus: true,\n minScenarioCount: 1,\n minSearchRuns: 1,\n minHoldoutRuns: 1,\n requireHoldout: true,\n minPassRate: 0.8,\n minMeanScore: 0.7,\n maxOverfitGap: 0.15,\n maxMeanCostUsd: Number.POSITIVE_INFINITY,\n maxP95WallMs: Number.POSITIVE_INFINITY,\n requireAsiForFailures: true,\n failureScoreThreshold: 0.5,\n}\n\nexport function evaluateReleaseConfidence(\n input: ReleaseConfidenceInput,\n): ReleaseConfidenceScorecard {\n const thresholds = { ...DEFAULT_THRESHOLDS, ...input.thresholds }\n const candidateId = input.candidateId ?? null\n const runs = filterCandidate(\n (input.runs ?? []).map(validateRunRecord),\n candidateId,\n input.baselineId,\n )\n const traces = filterTraceCandidate(\n (input.traces ?? []).map(validateReleaseTraceEvidence),\n candidateId,\n input.baselineId,\n )\n const scenarios = input.scenarios ?? []\n const scenarioCount = input.dataset?.scenarioCount ?? scenarios.length\n const splitCounts = input.dataset?.splitCounts ?? countScenarioSplits(scenarios)\n const searchScores = scoresFor(runs, 'search')\n const holdoutScores = scoresFor(runs, 'holdout')\n const runScores = runs.map(runSplitScore).filter(isFiniteNumber)\n const traceScores = traces.map((t) => t.score).filter(isFiniteNumber)\n const scoreUniverse = runs.length > 0 ? runScores : traceScores\n const qualityRuns = runs.filter((run) => runSplitScore(run) !== undefined)\n const unscoredRuns = runs.filter(\n (run) =>\n runSplitScore(run) === undefined &&\n !hasExplicitTaskFailure(run) &&\n !isFailedTerminalOutcome(run.terminalOutcome),\n ).length\n // Realness-gated runs are EXCLUDED from the pass rate — numerator AND\n // denominator — and the count of what was dropped ships beside the rate as\n // `metrics.realnessGatedRuns`. Counting a gamed run as a pass made faking a\n // success the cheapest way to improve a release scorecard; scoring it 0\n // instead would be the other error, silently deflating the rate with a run\n // the gate says carries no usable verdict at all.\n const honestRuns = runs.filter((run) => !isRealnessGated(run))\n const passOutcomes =\n runs.length > 0\n ? honestRuns.map((run) => runPassOutcome(run, thresholds.failureScoreThreshold))\n : traces.map((trace) => tracePassOutcome(trace, thresholds.failureScoreThreshold))\n const reliabilityRows =\n runs.length > 0\n ? runs.map((run) => terminalSuccess(run.terminalOutcome))\n : traces.map((trace) => trace.ok)\n const unclassifiedTerminalRuns = reliabilityRows.filter((outcome) => outcome === undefined).length\n const terminalFailureRuns = reliabilityRows.filter((outcome) => outcome === false).length\n const reliabilityRate =\n reliabilityRows.length === 0 || unclassifiedTerminalRuns > 0\n ? null\n : (reliabilityRows.length - terminalFailureRuns) / reliabilityRows.length\n const searchRuns = qualityRuns.filter((r) => r.splitTag === 'search').length\n const holdoutRuns = qualityRuns.filter((r) => r.splitTag === 'holdout').length\n const failed = failedRows(runs, traces, thresholds.failureScoreThreshold)\n const searchMeanScore = meanOrNull(searchScores)\n const holdoutMeanScore = meanOrNull(holdoutScores)\n const runCosts = runs.flatMap((run) =>\n run.costProvenance.kind === 'uncaptured' ? [] : [run.costProvenance.usd],\n )\n const traceCosts = traces.map((trace) => trace.costUsd).filter(isFiniteNumber)\n const meanCostUsd =\n runs.length > 0\n ? runCosts.length === runs.length\n ? meanOrNull(runCosts)\n : null\n : meanOrNull(traceCosts)\n const wallTimes =\n runs.length > 0\n ? runs.map((run) => run.wallMs)\n : traces.map((trace) => trace.durationMs).filter(isFiniteNumber)\n const metrics: ReleaseConfidenceMetrics = {\n scenarioCount,\n searchRuns,\n holdoutRuns,\n unscoredRuns,\n unclassifiedTerminalRuns,\n terminalFailureRuns,\n reliabilityRate,\n passRate: passOutcomeRate(passOutcomes),\n realnessGatedRuns: runs.length - honestRuns.length,\n meanScore: meanOrNull(scoreUniverse),\n searchMeanScore,\n holdoutMeanScore,\n overfitGap: diffOrNull(searchMeanScore, holdoutMeanScore),\n meanCostUsd,\n p95WallMs: percentileOrNull(wallTimes, 0.95),\n failedRows: failed.length,\n failuresWithAsi: failed.filter((row) => row.hasAsi).length,\n singleShotTraces: traces.filter((t) => t.turnCount === 1).length,\n multiShotTraces: traces.filter((t) => (t.turnCount ?? 0) > 1).length,\n splitCounts,\n domainCounts: countDomains(scenarios),\n failureClassCounts: countFailureClasses(runs, traces, thresholds.failureScoreThreshold),\n responsibleSurfaceCounts: countResponsibleSurfaces(traces),\n }\n\n const issues: ReleaseConfidenceIssue[] = []\n checkCorpus(input, thresholds, metrics, issues)\n checkQuality(thresholds, metrics, issues)\n checkReliability(metrics, issues)\n checkGeneralization(input.gateDecision ?? null, thresholds, metrics, issues)\n checkDiagnostics(thresholds, metrics, issues)\n checkEfficiency(thresholds, metrics, issues)\n\n const axes = buildAxes(metrics, thresholds, issues)\n const status = issues.some((i) => i.severity === 'critical')\n ? 'fail'\n : issues.length > 0\n ? 'warn'\n : 'pass'\n\n return {\n target: input.target,\n candidateId,\n baselineId: input.baselineId ?? null,\n status,\n promote: status === 'pass' && (input.gateDecision ? input.gateDecision.promote : true),\n axes,\n issues,\n metrics,\n dataset: input.dataset ?? null,\n gateDecision: input.gateDecision ?? null,\n summary: renderSummary(input.target, status, metrics, issues),\n }\n}\n\nexport function assertReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard {\n const scorecard = evaluateReleaseConfidence(input)\n if (scorecard.status === 'fail') {\n throw new VerificationError(scorecard.summary)\n }\n return scorecard\n}\n\nfunction filterCandidate(\n runs: readonly RunRecord[],\n candidateId: string | null,\n baselineId?: string,\n): RunRecord[] {\n if (candidateId) return runs.filter((r) => r.candidateId === candidateId)\n if (baselineId) return runs.filter((r) => r.candidateId !== baselineId)\n return [...runs]\n}\n\nfunction filterTraceCandidate(\n traces: readonly ReleaseTraceEvidence[],\n candidateId: string | null,\n baselineId?: string,\n): ReleaseTraceEvidence[] {\n if (candidateId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId === candidateId)\n if (baselineId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId !== baselineId)\n return [...traces]\n}\n\nfunction validateReleaseTraceEvidence(\n trace: ReleaseTraceEvidence,\n index: number,\n): ReleaseTraceEvidence {\n const value = trace as unknown as Record<string, unknown>\n if (Object.hasOwn(value, 'failureMode')) {\n throw new ValidationError(\n `traces[${index}].failureMode is not supported; use canonical failureClass`,\n )\n }\n if (trace.failureClass !== undefined && !FAILURE_CLASSES.includes(trace.failureClass)) {\n throw new ValidationError(\n `traces[${index}].failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n )\n }\n return trace\n}\n\nfunction checkCorpus(\n input: ReleaseConfidenceInput,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireCorpus && !input.dataset && (input.scenarios?.length ?? 0) === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_corpus',\n detail: 'No Dataset manifest or scenarios supplied.',\n })\n }\n if (metrics.scenarioCount < thresholds.minScenarioCount) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'few_scenarios',\n detail: `${metrics.scenarioCount} scenario(s) < min ${thresholds.minScenarioCount}.`,\n })\n }\n if (thresholds.requireHoldout && metrics.splitCounts.holdout === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_holdout_split',\n detail: 'Corpus has no holdout scenarios.',\n })\n }\n}\n\nfunction checkQuality(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.searchRuns < thresholds.minSearchRuns) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'few_search_runs',\n detail: `${metrics.searchRuns} search run(s) < min ${thresholds.minSearchRuns}.`,\n })\n }\n if (metrics.unscoredRuns > 0) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'unscored_runs',\n detail: `${metrics.unscoredRuns} supplied run(s) have no task result.`,\n })\n }\n if (metrics.passRate === null || metrics.meanScore === null) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'missing_quality_scores',\n detail: 'No task-quality scores are available for pass-rate and mean-score checks.',\n })\n }\n if (metrics.passRate !== null && metrics.passRate < thresholds.minPassRate) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_pass_rate',\n detail: `passRate ${fmt(metrics.passRate)} < ${fmt(thresholds.minPassRate)}.`,\n })\n }\n if (metrics.meanScore !== null && metrics.meanScore < thresholds.minMeanScore) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_mean_score',\n detail: `meanScore ${fmt(metrics.meanScore)} < ${fmt(thresholds.minMeanScore)}.`,\n })\n }\n}\n\nfunction checkReliability(\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.reliabilityRate === null) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'missing_reliability_evidence',\n detail:\n metrics.unclassifiedTerminalRuns > 0\n ? `${metrics.unclassifiedTerminalRuns} supplied run(s) have no classified terminal result.`\n : 'No classified terminal results are available.',\n })\n }\n if (metrics.terminalFailureRuns > 0) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'terminal_run_failures',\n detail: `${metrics.terminalFailureRuns} run(s) ended failed, cancelled, or incomplete.`,\n })\n }\n}\n\nfunction checkGeneralization(\n gateDecision: GateDecision | null,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireHoldout && metrics.holdoutRuns < thresholds.minHoldoutRuns) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'few_holdout_runs',\n detail: `${metrics.holdoutRuns} holdout run(s) < min ${thresholds.minHoldoutRuns}.`,\n })\n }\n if (metrics.overfitGap !== null && metrics.overfitGap > thresholds.maxOverfitGap) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'overfit_gap',\n detail: `search-holdout gap ${fmt(metrics.overfitGap)} > ${fmt(thresholds.maxOverfitGap)}.`,\n })\n }\n if (gateDecision && !gateDecision.promote) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: `gate_${gateDecision.rejectionCode ?? 'reject'}`,\n detail: gateDecision.reason,\n })\n }\n}\n\nfunction checkDiagnostics(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (!thresholds.requireAsiForFailures) return\n if (metrics.failedRows > metrics.failuresWithAsi) {\n issues.push({\n axis: 'diagnostics',\n severity: 'critical',\n code: 'missing_failure_asi',\n detail: `${metrics.failedRows - metrics.failuresWithAsi} failed row(s) have no actionable side information.`,\n })\n }\n}\n\nfunction checkEfficiency(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (Number.isFinite(thresholds.maxMeanCostUsd) && metrics.meanCostUsd === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_cost',\n detail: 'A finite cost limit was configured but no cost evidence is available.',\n })\n } else if (metrics.meanCostUsd !== null && metrics.meanCostUsd > thresholds.maxMeanCostUsd) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'cost_budget',\n detail: `meanCostUsd ${fmt(metrics.meanCostUsd)} > ${fmt(thresholds.maxMeanCostUsd)}.`,\n })\n }\n if (Number.isFinite(thresholds.maxP95WallMs) && metrics.p95WallMs === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_latency',\n detail: 'A finite latency limit was configured but no latency evidence is available.',\n })\n } else if (metrics.p95WallMs !== null && metrics.p95WallMs > thresholds.maxP95WallMs) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'latency_budget',\n detail: `p95WallMs ${fmt(metrics.p95WallMs)} > ${fmt(thresholds.maxP95WallMs)}.`,\n })\n }\n}\n\nfunction buildAxes(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n issues: ReleaseConfidenceIssue[],\n): ReleaseConfidenceAxis[] {\n return [\n axis(\n 'corpus',\n issues,\n bounded(metrics.scenarioCount / Math.max(1, thresholds.minScenarioCount)),\n `${metrics.scenarioCount} scenarios; holdout=${metrics.splitCounts.holdout}`,\n ),\n axis(\n 'quality',\n issues,\n metrics.passRate === null || metrics.meanScore === null\n ? null\n : Math.min(metrics.passRate, metrics.meanScore),\n `passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`,\n ),\n axis(\n 'reliability',\n issues,\n metrics.reliabilityRate,\n `successRate=${fmt(metrics.reliabilityRate)} terminalFailures=${metrics.terminalFailureRuns} unclassified=${metrics.unclassifiedTerminalRuns}`,\n ),\n axis(\n 'generalization',\n issues,\n gapScore(metrics.overfitGap, thresholds.maxOverfitGap),\n `holdoutRuns=${metrics.holdoutRuns} overfitGap=${fmt(metrics.overfitGap)}`,\n ),\n axis(\n 'diagnostics',\n issues,\n metrics.failedRows === 0 ? 1 : metrics.failuresWithAsi / metrics.failedRows,\n `failuresWithAsi=${metrics.failuresWithAsi}/${metrics.failedRows}`,\n ),\n axis(\n 'efficiency',\n issues,\n efficiencyScore(metrics, thresholds),\n `meanCostUsd=${fmt(metrics.meanCostUsd)} p95WallMs=${fmt(metrics.p95WallMs)}`,\n ),\n ]\n}\n\nfunction axis(\n name: ReleaseConfidenceAxisName,\n issues: ReleaseConfidenceIssue[],\n score: number | null,\n detail: string,\n): ReleaseConfidenceAxis {\n const own = issues.filter((i) => i.axis === name)\n const status = own.some((i) => i.severity === 'critical')\n ? 'fail'\n : own.length > 0\n ? 'warn'\n : 'pass'\n return { name, status, score: score === null ? null : bounded(score), detail }\n}\n\nfunction countScenarioSplits(scenarios: readonly DatasetScenario[]): Record<DatasetSplit, number> {\n const counts: Record<DatasetSplit, number> = { train: 0, dev: 0, test: 0, holdout: 0 }\n for (const scenario of scenarios) counts[scenario.split ?? 'train']++\n return counts\n}\n\nfunction countDomains(scenarios: readonly DatasetScenario[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const scenario of scenarios) {\n const domain = scenario.tags?.domain ?? scenario.tags?.category ?? 'uncategorized'\n out[domain] = (out[domain] ?? 0) + 1\n }\n return out\n}\n\nfunction countFailureClasses(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Partial<Record<FailureClass, number>> {\n const out: Partial<Record<FailureClass, number>> = {}\n for (const run of runs) {\n // Ungated: a failure-mode census counts what the runs REPORTED, which is\n // the only way a gamed run's inflated score is visible at all. It is not a\n // promotion number — `passRate` is, and that one excludes gated runs and\n // publishes the excluded count as `metrics.realnessGatedRuns`.\n if (runPassOutcome(run, threshold) === false) {\n const failureClass =\n run.failureClass !== undefined && run.failureClass !== 'success'\n ? run.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n const failureClass =\n trace.failureClass !== undefined && trace.failureClass !== 'success'\n ? trace.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction countResponsibleSurfaces(traces: readonly ReleaseTraceEvidence[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const trace of traces) {\n for (const asi of trace.asi ?? []) {\n const surface = asi.responsibleSurface ?? 'unknown'\n out[surface] = (out[surface] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction failedRows(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Array<{ hasAsi: boolean }> {\n const out: Array<{ hasAsi: boolean }> = []\n for (const run of runs) {\n if (runPassOutcome(run, threshold) === false) {\n const asiMetric = run.outcome.raw.asi\n out.push({ hasAsi: typeof asiMetric === 'number' && asiMetric > 0 })\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n out.push({ hasAsi: (trace.asi?.length ?? 0) > 0 })\n }\n }\n return out\n}\n\nfunction passOutcomeRate(outcomes: readonly (boolean | null)[]): number | null {\n const classified = outcomes.filter((outcome): outcome is boolean => outcome !== null)\n if (classified.length === 0) return null\n return classified.filter(Boolean).length / classified.length\n}\n\nfunction runPassOutcome(run: RunRecord, threshold: number): boolean | null {\n if (hasExplicitTaskFailure(run)) return false\n const score = runSplitScore(run)\n return score === undefined ? null : score >= threshold\n}\n\nfunction hasExplicitTaskFailure(run: RunRecord): boolean {\n return run.failureClass !== undefined && run.failureClass !== 'success'\n}\n\nfunction tracePassOutcome(trace: ReleaseTraceEvidence, threshold: number): boolean | null {\n if (trace.failureClass !== undefined && trace.failureClass !== 'success') return false\n if (trace.ok === false) return false\n if (isFiniteNumber(trace.score)) return trace.score >= threshold\n return trace.ok === true ? true : null\n}\n\nfunction isFailedTerminalOutcome(\n outcome: RunRecord['terminalOutcome'],\n): outcome is 'failed' | 'cancelled' | 'incomplete' {\n return outcome === 'failed' || outcome === 'cancelled' || outcome === 'incomplete'\n}\n\nfunction terminalSuccess(outcome: RunRecord['terminalOutcome']): boolean | undefined {\n if (outcome === 'succeeded') return true\n if (isFailedTerminalOutcome(outcome)) return false\n return undefined\n}\n\nfunction scoresFor(runs: readonly RunRecord[], split: RunSplitTag): number[] {\n return runs\n .filter((run) => run.splitTag === split)\n .map(runSplitScore)\n .filter(isFiniteNumber)\n}\n\n/**\n * RAW and split-exact (`observedSplitScore`): this feeds the per-split means,\n * the overfit gap, and the pass threshold — descriptions of what the runs\n * reported. A gamed run inflating them is visible next to `realnessGatedRuns`,\n * and the promotion number (`passRate`) excludes gated runs entirely.\n */\nfunction runSplitScore(run: RunRecord): number | undefined {\n const score = observedSplitScore(run, run.splitTag === 'holdout' ? 'holdout' : 'search')\n return isFiniteNumber(score) ? score : undefined\n}\n\nfunction meanOrNull(xs: readonly number[]): number | null {\n if (xs.length === 0) return null\n return xs.reduce((sum, x) => sum + x, 0) / xs.length\n}\n\nfunction percentileOrNull(xs: readonly number[], p: number): number | null {\n if (xs.length === 0) return null\n const sorted = [...xs].sort((a, b) => a - b)\n return sorted[Math.min(sorted.length - 1, Math.max(0, Math.ceil(p * sorted.length) - 1))]!\n}\n\nfunction isFiniteNumber(value: unknown): value is number {\n return typeof value === 'number' && Number.isFinite(value)\n}\n\nfunction diffOrNull(a: number | null, b: number | null): number | null {\n if (a === null || b === null) return null\n return a - b\n}\n\nfunction gapScore(gap: number | null, maxGap: number): number | null {\n if (gap === null) return null\n if (maxGap <= 0) return gap <= 0 ? 1 : 0\n return bounded(1 - Math.max(0, gap) / maxGap)\n}\n\nfunction efficiencyScore(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n): number | null {\n const cost = Number.isFinite(thresholds.maxMeanCostUsd)\n ? metrics.meanCostUsd === null\n ? null\n : bounded(thresholds.maxMeanCostUsd / Math.max(metrics.meanCostUsd, 1e-12))\n : 1\n const latency = Number.isFinite(thresholds.maxP95WallMs)\n ? metrics.p95WallMs === null\n ? null\n : bounded(thresholds.maxP95WallMs / Math.max(metrics.p95WallMs, 1e-12))\n : 1\n if (cost === null || latency === null) return null\n return Math.min(cost, latency)\n}\n\nfunction bounded(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n\nfunction renderSummary(\n target: string,\n status: ReleaseConfidenceStatus,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): string {\n const prefix = `release confidence ${status}: ${target}`\n const metricText = `scenarios=${metrics.scenarioCount} searchRuns=${metrics.searchRuns} holdoutRuns=${metrics.holdoutRuns} passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`\n if (issues.length === 0) return `${prefix}; ${metricText}`\n return `${prefix}; ${metricText}; issues=${issues.map((i) => i.code).join(',')}`\n}\n\nfunction fmt(x: number | null): string {\n if (x === null) return 'n/a'\n return x.toFixed(4)\n}\n","/**\n * Bootstrap-CI promotion gate.\n *\n * In any iterative-improvement loop (GEPA, prompt evolution, dataset\n * curation), the question is \"did this generation actually improve, or are\n * we celebrating noise?\". With small N and noisy outcomes, point-estimate\n * deltas lie. Bootstrap confidence intervals tell the operator whether the\n * delta is real before code or prompts get promoted.\n *\n * This module is pure functions — no I/O, no model calls. Easy to unit-test\n * and to compose into any verdict gate.\n *\n * Default gate:\n * - Bootstrap mean baseline vs candidate (1k resamples).\n * - Compute the delta distribution; pass if the lower CI bound > 0.\n * - Tunable confidence (default 95%) and resample count.\n *\n * Verdict semantics intentionally match the existing `experiments.jsonl`\n * vocabulary:\n * - ADVANCE: candidate's CI lower bound > baseline mean (real win)\n * - KEEP: overlap, but candidate point estimate >= baseline (neutral)\n * - REVERT: candidate's CI upper bound < baseline mean (real regression)\n * - INCONCLUSIVE: not enough samples or CI straddles zero with no signal\n */\n\nexport type Verdict = 'ADVANCE' | 'KEEP' | 'REVERT' | 'INCONCLUSIVE'\n\nexport interface BootstrapResult {\n baselineMean: number\n candidateMean: number\n /** candidateMean - baselineMean, point estimate. */\n delta: number\n /** Lower bound of the (1 - alpha) CI on the delta. */\n ciLower: number\n /** Upper bound of the (1 - alpha) CI on the delta. */\n ciUpper: number\n /** Number of bootstrap resamples used. */\n iterations: number\n alpha: number\n verdict: Verdict\n}\n\nexport interface BootstrapOptions {\n /** Confidence level alpha (default 0.05 → 95% CI). */\n alpha?: number\n /** Number of resamples (default 1000). */\n iterations?: number\n /**\n * Minimum total samples (baseline + candidate) below which we always\n * return INCONCLUSIVE — bootstrap with too few samples is meaningless.\n * Default 6 (combined).\n */\n minTotalSamples?: number\n /** RNG seed for reproducibility. Default: Math.random. */\n seed?: number\n}\n\n/**\n * Compute the bootstrap CI on (candidateMean - baselineMean) and a verdict.\n *\n * Uses simple percentile bootstrap on the difference of resampled means.\n * That's the standard non-parametric primitive — no distributional\n * assumptions, robust to skew, easy to reason about.\n */\nexport function bootstrapCi(\n baseline: number[],\n candidate: number[],\n options: BootstrapOptions = {},\n): BootstrapResult {\n const alpha = options.alpha ?? 0.05\n const iterations = options.iterations ?? 1000\n const minTotal = options.minTotalSamples ?? 6\n const rng = mulberry32(options.seed ?? hashSeed(baseline, candidate))\n\n const baselineMean = mean(baseline)\n const candidateMean = mean(candidate)\n const delta = candidateMean - baselineMean\n\n if (\n baseline.length + candidate.length < minTotal ||\n baseline.length === 0 ||\n candidate.length === 0\n ) {\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower: -Infinity,\n ciUpper: Infinity,\n iterations: 0,\n alpha,\n verdict: 'INCONCLUSIVE',\n }\n }\n\n const deltas: number[] = new Array(iterations)\n for (let i = 0; i < iterations; i++) {\n const bResample = resample(baseline, rng)\n const cResample = resample(candidate, rng)\n deltas[i] = mean(cResample) - mean(bResample)\n }\n deltas.sort((a, b) => a - b)\n const lowerIdx = Math.floor((alpha / 2) * iterations)\n const upperIdx = Math.floor((1 - alpha / 2) * iterations) - 1\n const ciLower = deltas[Math.max(0, lowerIdx)]!\n const ciUpper = deltas[Math.min(iterations - 1, upperIdx)]!\n\n let verdict: Verdict\n if (ciLower > 0) verdict = 'ADVANCE'\n else if (ciUpper < 0) verdict = 'REVERT'\n else if (delta >= 0) verdict = 'KEEP'\n else verdict = 'INCONCLUSIVE'\n\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower,\n ciUpper,\n iterations,\n alpha,\n verdict,\n }\n}\n\nfunction mean(xs: number[]): number {\n if (xs.length === 0) return 0\n let s = 0\n for (const x of xs) s += x\n return s / xs.length\n}\n\nfunction resample(xs: number[], rng: () => number): number[] {\n const out = new Array(xs.length)\n for (let i = 0; i < xs.length; i++) out[i] = xs[Math.floor(rng() * xs.length)]\n return out\n}\n\n/** Mulberry32 — fast deterministic PRNG. Stable across runs given the same seed. */\nfunction mulberry32(seed: number): () => number {\n let t = seed >>> 0\n return () => {\n t += 0x6d2b79f5\n let r = t\n r = Math.imul(r ^ (r >>> 15), r | 1)\n r ^= r + Math.imul(r ^ (r >>> 7), r | 61)\n return ((r ^ (r >>> 14)) >>> 0) / 4294967296\n }\n}\n\n/** Stable seed derived from the inputs — same data → same CI bounds. */\nfunction hashSeed(a: number[], b: number[]): number {\n let h = 2166136261\n for (const x of [...a, ...b]) {\n const view = new Float64Array([x])\n const bytes = new Uint8Array(view.buffer)\n for (const byte of bytes) {\n h ^= byte\n h = Math.imul(h, 16777619)\n }\n }\n return h >>> 0\n}\n\n/**\n * Judge-replay promotion gate.\n *\n * The cheap inner-loop judge that drives an evolution run is by definition\n * fast and noisy. When you're about to promote a winning variant to the\n * canonical default, you want a STRONGER judge (a more expensive model, a\n * human grader, a separately-trained reward model) to confirm the win\n * generalises beyond the inner loop.\n *\n * This helper takes raw winner + baseline outputs, scores both through the\n * stronger judge, and applies `bootstrapCi`. ADVANCE means the stronger\n * judge agrees the winner is real with the configured confidence. Doesn't\n * matter what shape your \"output\" is — pass a string, an object, anything\n * the judge can read.\n */\nexport interface JudgeReplayGateArgs<TOutput> {\n baselineOutputs: TOutput[]\n candidateOutputs: TOutput[]\n /** Stronger judge — async to allow LLM calls. Return a 0..N scalar score. */\n judge: (output: TOutput) => Promise<number> | number\n alpha?: number\n iterations?: number\n /** RNG seed for reproducibility. */\n seed?: number\n /** Maximum concurrent judge calls. Default 4. */\n judgeConcurrency?: number\n}\n\n/**\n * Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.\n */\nexport async function judgeReplayGate<TOutput>(\n args: JudgeReplayGateArgs<TOutput>,\n): Promise<BootstrapResult & { baselineSamples: number; candidateSamples: number }> {\n const concurrency = args.judgeConcurrency ?? 4\n const baselineScores = await scoreAll(args.baselineOutputs, args.judge, concurrency)\n const candidateScores = await scoreAll(args.candidateOutputs, args.judge, concurrency)\n const ci = bootstrapCi(baselineScores, candidateScores, {\n ...(args.alpha !== undefined ? { alpha: args.alpha } : {}),\n ...(args.iterations !== undefined ? { iterations: args.iterations } : {}),\n ...(args.seed !== undefined ? { seed: args.seed } : {}),\n })\n return {\n ...ci,\n baselineSamples: baselineScores.length,\n candidateSamples: candidateScores.length,\n }\n}\n\nasync function scoreAll<TOutput>(\n outputs: TOutput[],\n judge: (output: TOutput) => Promise<number> | number,\n concurrency: number,\n): Promise<number[]> {\n const results: number[] = new Array(outputs.length)\n let next = 0\n async function worker(): Promise<void> {\n while (true) {\n const i = next++\n if (i >= outputs.length) return\n const v = await judge(outputs[i]!)\n results[i] = Number.isFinite(v) ? v : 0\n }\n }\n await Promise.all(Array.from({ length: Math.max(1, concurrency) }, () => worker()))\n return results\n}\n","import type { ReleaseConfidenceScorecard } from './release-confidence'\nimport type { RunRecord } from './run-record'\nimport { summaryTable } from './summary-report'\n\nexport interface RenderReleaseReportOptions {\n title?: string\n runs?: readonly RunRecord[]\n comparator?: string\n traceAnalystFindings?: readonly string[]\n nextActions?: readonly string[]\n}\n\nexport function renderReleaseReport(\n scorecard: ReleaseConfidenceScorecard,\n options: RenderReleaseReportOptions = {},\n): string {\n const title = options.title ?? `Release Report: ${scorecard.target}`\n const lines: string[] = []\n lines.push(`# ${title}`)\n lines.push('')\n lines.push(`Status: **${scorecard.status.toUpperCase()}**`)\n lines.push(`Promote: **${scorecard.promote ? 'yes' : 'no'}**`)\n if (scorecard.candidateId) lines.push(`Candidate: \\`${scorecard.candidateId}\\``)\n if (scorecard.baselineId) lines.push(`Baseline: \\`${scorecard.baselineId}\\``)\n lines.push('')\n lines.push(scorecard.summary)\n lines.push('')\n\n lines.push('## Metrics')\n lines.push('')\n lines.push('| Metric | Value |')\n lines.push('|---|---:|')\n lines.push(`| Scenarios | ${scorecard.metrics.scenarioCount} |`)\n lines.push(`| Search runs | ${scorecard.metrics.searchRuns} |`)\n lines.push(`| Holdout runs | ${scorecard.metrics.holdoutRuns} |`)\n lines.push(`| Unscored runs | ${scorecard.metrics.unscoredRuns} |`)\n lines.push(`| Terminal failures | ${scorecard.metrics.terminalFailureRuns} |`)\n lines.push(`| Pass rate | ${pct(scorecard.metrics.passRate)} |`)\n lines.push(`| Mean score | ${num(scorecard.metrics.meanScore)} |`)\n lines.push(`| Search mean | ${num(scorecard.metrics.searchMeanScore)} |`)\n lines.push(`| Holdout mean | ${num(scorecard.metrics.holdoutMeanScore)} |`)\n lines.push(`| Overfit gap | ${num(scorecard.metrics.overfitGap)} |`)\n lines.push(`| Mean cost | $${num(scorecard.metrics.meanCostUsd)} |`)\n lines.push(`| p95 wall time | ${duration(scorecard.metrics.p95WallMs)} |`)\n lines.push('')\n\n if (scorecard.issues.length > 0) {\n lines.push('## Issues')\n lines.push('')\n for (const issue of scorecard.issues) {\n lines.push(`- **${issue.severity}** \\`${issue.code}\\` (${issue.axis}): ${issue.detail}`)\n }\n lines.push('')\n }\n\n const surfaces = entries(scorecard.metrics.responsibleSurfaceCounts)\n if (surfaces.length > 0) {\n lines.push('## Responsible Surfaces')\n lines.push('')\n for (const [surface, count] of surfaces) lines.push(`- ${surface}: ${count}`)\n lines.push('')\n }\n\n const failures = entries(scorecard.metrics.failureClassCounts)\n if (failures.length > 0) {\n lines.push('## Failure Classes')\n lines.push('')\n for (const [mode, count] of failures) lines.push(`- ${mode}: ${count}`)\n lines.push('')\n }\n\n if (options.runs && options.runs.length > 0) {\n lines.push('## Run Summary')\n lines.push('')\n lines.push(\n summaryTable([...options.runs], {\n comparator: options.comparator ?? scorecard.baselineId ?? undefined,\n split: 'holdout',\n }).markdown,\n )\n lines.push('')\n }\n\n if (options.traceAnalystFindings && options.traceAnalystFindings.length > 0) {\n lines.push('## TraceAnalyst Findings')\n lines.push('')\n for (const finding of options.traceAnalystFindings) lines.push(`- ${finding}`)\n lines.push('')\n }\n\n const nextActions = options.nextActions ?? defaultNextActions(scorecard)\n if (nextActions.length > 0) {\n lines.push('## Next Actions')\n lines.push('')\n for (const action of nextActions) lines.push(`- ${action}`)\n lines.push('')\n }\n\n return `${lines.join('\\n').trimEnd()}\\n`\n}\n\nfunction defaultNextActions(scorecard: ReleaseConfidenceScorecard): string[] {\n if (scorecard.promote) return ['Promote the candidate and keep canaries enabled.']\n return scorecard.issues\n .filter((issue) => issue.severity === 'critical')\n .map((issue) => `Resolve ${issue.code}: ${issue.detail}`)\n}\n\nfunction entries(values: Record<string, number>): Array<[string, number]> {\n return Object.entries(values)\n .filter(([, count]) => count > 0)\n .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))\n}\n\nfunction pct(value: number | null): string {\n return value === null ? 'n/a' : `${(value * 100).toFixed(1)}%`\n}\n\nfunction num(value: number | null): string {\n return value === null ? 'n/a' : value.toFixed(3)\n}\n\nfunction duration(value: number | null): string {\n return value === null ? 'n/a' : `${Math.round(value)} ms`\n}\n"],"mappings":";;;;;;AAoKA,MAAM,qBAA4D;CAChE,eAAe;CACf,kBAAkB;CAClB,eAAe;CACf,gBAAgB;CAChB,gBAAgB;CAChB,aAAa;CACb,cAAc;CACd,eAAe;CACf,gBAAgB,OAAO;CACvB,cAAc,OAAO;CACrB,uBAAuB;CACvB,uBAAuB;AACzB;AAEA,SAAgB,0BACd,OAC4B;CAC5B,MAAM,aAAa;EAAE,GAAG;EAAoB,GAAG,MAAM;CAAW;CAChE,MAAM,cAAc,MAAM,eAAe;CACzC,MAAM,OAAO,iBACV,MAAM,QAAQ,CAAC,EAAA,CAAG,IAAI,iBAAiB,GACxC,aACA,MAAM,UACR;CACA,MAAM,SAAS,sBACZ,MAAM,UAAU,CAAC,EAAA,CAAG,IAAI,4BAA4B,GACrD,aACA,MAAM,UACR;CACA,MAAM,YAAY,MAAM,aAAa,CAAC;CACtC,MAAM,gBAAgB,MAAM,SAAS,iBAAiB,UAAU;CAChE,MAAM,cAAc,MAAM,SAAS,eAAe,oBAAoB,SAAS;CAC/E,MAAM,eAAe,UAAU,MAAM,QAAQ;CAC7C,MAAM,gBAAgB,UAAU,MAAM,SAAS;CAC/C,MAAM,YAAY,KAAK,IAAI,aAAa,CAAC,CAAC,OAAO,cAAc;CAC/D,MAAM,cAAc,OAAO,KAAK,MAAM,EAAE,KAAK,CAAC,CAAC,OAAO,cAAc;CACpE,MAAM,gBAAgB,KAAK,SAAS,IAAI,YAAY;CACpD,MAAM,cAAc,KAAK,QAAQ,QAAQ,cAAc,GAAG,MAAM,KAAA,CAAS;CACzE,MAAM,eAAe,KAAK,QACvB,QACC,cAAc,GAAG,MAAM,KAAA,KACvB,CAAC,uBAAuB,GAAG,KAC3B,CAAC,wBAAwB,IAAI,eAAe,CAChD,CAAC,CAAC;CAOF,MAAM,aAAa,KAAK,QAAQ,QAAQ,CAAC,gBAAgB,GAAG,CAAC;CAC7D,MAAM,eACJ,KAAK,SAAS,IACV,WAAW,KAAK,QAAQ,eAAe,KAAK,WAAW,qBAAqB,CAAC,IAC7E,OAAO,KAAK,UAAU,iBAAiB,OAAO,WAAW,qBAAqB,CAAC;CACrF,MAAM,kBACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,gBAAgB,IAAI,eAAe,CAAC,IACtD,OAAO,KAAK,UAAU,MAAM,EAAE;CACpC,MAAM,2BAA2B,gBAAgB,QAAQ,YAAY,YAAY,KAAA,CAAS,CAAC,CAAC;CAC5F,MAAM,sBAAsB,gBAAgB,QAAQ,YAAY,YAAY,KAAK,CAAC,CAAC;CACnF,MAAM,kBACJ,gBAAgB,WAAW,KAAK,2BAA2B,IACvD,QACC,gBAAgB,SAAS,uBAAuB,gBAAgB;CACvE,MAAM,aAAa,YAAY,QAAQ,MAAM,EAAE,aAAa,QAAQ,CAAC,CAAC;CACtE,MAAM,cAAc,YAAY,QAAQ,MAAM,EAAE,aAAa,SAAS,CAAC,CAAC;CACxE,MAAM,SAAS,WAAW,MAAM,QAAQ,WAAW,qBAAqB;CACxE,MAAM,kBAAkB,WAAW,YAAY;CAC/C,MAAM,mBAAmB,WAAW,aAAa;CACjD,MAAM,WAAW,KAAK,SAAS,QAC7B,IAAI,eAAe,SAAS,eAAe,CAAC,IAAI,CAAC,IAAI,eAAe,GAAG,CACzE;CACA,MAAM,aAAa,OAAO,KAAK,UAAU,MAAM,OAAO,CAAC,CAAC,OAAO,cAAc;CAC7E,MAAM,cACJ,KAAK,SAAS,IACV,SAAS,WAAW,KAAK,SACvB,WAAW,QAAQ,IACnB,OACF,WAAW,UAAU;CAC3B,MAAM,YACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,IAAI,MAAM,IAC5B,OAAO,KAAK,UAAU,MAAM,UAAU,CAAC,CAAC,OAAO,cAAc;CACnE,MAAM,UAAoC;EACxC;EACA;EACA;EACA;EACA;EACA;EACA;EACA,UAAU,gBAAgB,YAAY;EACtC,mBAAmB,KAAK,SAAS,WAAW;EAC5C,WAAW,WAAW,aAAa;EACnC;EACA;EACA,YAAY,WAAW,iBAAiB,gBAAgB;EACxD;EACA,WAAW,iBAAiB,WAAW,GAAI;EAC3C,YAAY,OAAO;EACnB,iBAAiB,OAAO,QAAQ,QAAQ,IAAI,MAAM,CAAC,CAAC;EACpD,kBAAkB,OAAO,QAAQ,MAAM,EAAE,cAAc,CAAC,CAAC,CAAC;EAC1D,iBAAiB,OAAO,QAAQ,OAAO,EAAE,aAAa,KAAK,CAAC,CAAC,CAAC;EAC9D;EACA,cAAc,aAAa,SAAS;EACpC,oBAAoB,oBAAoB,MAAM,QAAQ,WAAW,qBAAqB;EACtF,0BAA0B,yBAAyB,MAAM;CAC3D;CAEA,MAAM,SAAmC,CAAC;CAC1C,YAAY,OAAO,YAAY,SAAS,MAAM;CAC9C,aAAa,YAAY,SAAS,MAAM;CACxC,iBAAiB,SAAS,MAAM;CAChC,oBAAoB,MAAM,gBAAgB,MAAM,YAAY,SAAS,MAAM;CAC3E,iBAAiB,YAAY,SAAS,MAAM;CAC5C,gBAAgB,YAAY,SAAS,MAAM;CAE3C,MAAM,OAAO,UAAU,SAAS,YAAY,MAAM;CAClD,MAAM,SAAS,OAAO,MAAM,MAAM,EAAE,aAAa,UAAU,IACvD,SACA,OAAO,SAAS,IACd,SACA;CAEN,OAAO;EACL,QAAQ,MAAM;EACd;EACA,YAAY,MAAM,cAAc;EAChC;EACA,SAAS,WAAW,WAAW,MAAM,eAAe,MAAM,aAAa,UAAU;EACjF;EACA;EACA;EACA,SAAS,MAAM,WAAW;EAC1B,cAAc,MAAM,gBAAgB;EACpC,SAAS,cAAc,MAAM,QAAQ,QAAQ,SAAS,MAAM;CAC9D;AACF;AAEA,SAAgB,wBAAwB,OAA2D;CACjG,MAAM,YAAY,0BAA0B,KAAK;CACjD,IAAI,UAAU,WAAW,QACvB,MAAM,IAAI,kBAAkB,UAAU,OAAO;CAE/C,OAAO;AACT;AAEA,SAAS,gBACP,MACA,aACA,YACa;CACb,IAAI,aAAa,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,WAAW;CACxE,IAAI,YAAY,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,UAAU;CACtE,OAAO,CAAC,GAAG,IAAI;AACjB;AAEA,SAAS,qBACP,QACA,aACA,YACwB;CACxB,IAAI,aACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,WAAW;CAC1F,IAAI,YACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,UAAU;CACzF,OAAO,CAAC,GAAG,MAAM;AACnB;AAEA,SAAS,6BACP,OACA,OACsB;CACtB,MAAM,QAAQ;CACd,IAAI,OAAO,OAAO,OAAO,aAAa,GACpC,MAAM,IAAI,gBACR,UAAU,MAAM,2DAClB;CAEF,IAAI,MAAM,iBAAiB,KAAA,KAAa,CAAC,gBAAgB,SAAS,MAAM,YAAY,GAClF,MAAM,IAAI,gBACR,UAAU,MAAM,gCAAgC,gBAAgB,KAAK,IAAI,GAC3E;CAEF,OAAO;AACT;AAEA,SAAS,YACP,OACA,YACA,SACA,QACM;CACN,IAAI,WAAW,iBAAiB,CAAC,MAAM,YAAY,MAAM,WAAW,UAAU,OAAO,GACnF,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,gBAAgB,WAAW,kBACrC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,cAAc,qBAAqB,WAAW,iBAAiB;CACpF,CAAC;CAEH,IAAI,WAAW,kBAAkB,QAAQ,YAAY,YAAY,GAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;AAEL;AAEA,SAAS,aACP,YACA,SACA,QACM;CACN,IAAI,QAAQ,aAAa,WAAW,eAClC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,WAAW,uBAAuB,WAAW,cAAc;CAChF,CAAC;CAEH,IAAI,QAAQ,eAAe,GACzB,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa;CAClC,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,cAAc,MACrD,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,WAAW,WAAW,aAC7D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,YAAY,IAAI,QAAQ,QAAQ,EAAE,KAAK,IAAI,WAAW,WAAW,EAAE;CAC7E,CAAC;CAEH,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,iBACP,SACA,QACM;CACN,IAAI,QAAQ,oBAAoB,MAC9B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QACE,QAAQ,2BAA2B,IAC/B,GAAG,QAAQ,yBAAyB,wDACpC;CACR,CAAC;CAEH,IAAI,QAAQ,sBAAsB,GAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,oBAAoB;CACzC,CAAC;AAEL;AAEA,SAAS,oBACP,cACA,YACA,SACA,QACM;CACN,IAAI,WAAW,kBAAkB,QAAQ,cAAc,WAAW,gBAChE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,YAAY,wBAAwB,WAAW,eAAe;CACnF,CAAC;CAEH,IAAI,QAAQ,eAAe,QAAQ,QAAQ,aAAa,WAAW,eACjE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,sBAAsB,IAAI,QAAQ,UAAU,EAAE,KAAK,IAAI,WAAW,aAAa,EAAE;CAC3F,CAAC;CAEH,IAAI,gBAAgB,CAAC,aAAa,SAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM,QAAQ,aAAa,iBAAiB;EAC5C,QAAQ,aAAa;CACvB,CAAC;AAEL;AAEA,SAAS,iBACP,YACA,SACA,QACM;CACN,IAAI,CAAC,WAAW,uBAAuB;CACvC,IAAI,QAAQ,aAAa,QAAQ,iBAC/B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa,QAAQ,gBAAgB;CAC1D,CAAC;AAEL;AAEA,SAAS,gBACP,YACA,SACA,QACM;CACN,IAAI,OAAO,SAAS,WAAW,cAAc,KAAK,QAAQ,gBAAgB,MACxE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,gBAAgB,QAAQ,QAAQ,cAAc,WAAW,gBAC1E,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,eAAe,IAAI,QAAQ,WAAW,EAAE,KAAK,IAAI,WAAW,cAAc,EAAE;CACtF,CAAC;CAEH,IAAI,OAAO,SAAS,WAAW,YAAY,KAAK,QAAQ,cAAc,MACpE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cACtE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,UACP,SACA,YACA,QACyB;CACzB,OAAO;EACL,KACE,UACA,QACA,QAAQ,QAAQ,gBAAgB,KAAK,IAAI,GAAG,WAAW,gBAAgB,CAAC,GACxE,GAAG,QAAQ,cAAc,sBAAsB,QAAQ,YAAY,SACrE;EACA,KACE,WACA,QACA,QAAQ,aAAa,QAAQ,QAAQ,cAAc,OAC/C,OACA,KAAK,IAAI,QAAQ,UAAU,QAAQ,SAAS,GAChD,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS,GACtE;EACA,KACE,eACA,QACA,QAAQ,iBACR,eAAe,IAAI,QAAQ,eAAe,EAAE,oBAAoB,QAAQ,oBAAoB,gBAAgB,QAAQ,0BACtH;EACA,KACE,kBACA,QACA,SAAS,QAAQ,YAAY,WAAW,aAAa,GACrD,eAAe,QAAQ,YAAY,cAAc,IAAI,QAAQ,UAAU,GACzE;EACA,KACE,eACA,QACA,QAAQ,eAAe,IAAI,IAAI,QAAQ,kBAAkB,QAAQ,YACjE,mBAAmB,QAAQ,gBAAgB,GAAG,QAAQ,YACxD;EACA,KACE,cACA,QACA,gBAAgB,SAAS,UAAU,GACnC,eAAe,IAAI,QAAQ,WAAW,EAAE,aAAa,IAAI,QAAQ,SAAS,GAC5E;CACF;AACF;AAEA,SAAS,KACP,MACA,QACA,OACA,QACuB;CACvB,MAAM,MAAM,OAAO,QAAQ,MAAM,EAAE,SAAS,IAAI;CAMhD,OAAO;EAAE;EAAM,QALA,IAAI,MAAM,MAAM,EAAE,aAAa,UAAU,IACpD,SACA,IAAI,SAAS,IACX,SACA;EACiB,OAAO,UAAU,OAAO,OAAO,QAAQ,KAAK;EAAG;CAAO;AAC/E;AAEA,SAAS,oBAAoB,WAAqE;CAChG,MAAM,SAAuC;EAAE,OAAO;EAAG,KAAK;EAAG,MAAM;EAAG,SAAS;CAAE;CACrF,KAAK,MAAM,YAAY,WAAW,OAAO,SAAS,SAAS,QAAQ;CACnE,OAAO;AACT;AAEA,SAAS,aAAa,WAA+D;CACnF,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,YAAY,WAAW;EAChC,MAAM,SAAS,SAAS,MAAM,UAAU,SAAS,MAAM,YAAY;EACnE,IAAI,WAAW,IAAI,WAAW,KAAK;CACrC;CACA,OAAO;AACT;AAEA,SAAS,oBACP,MACA,QACA,WACuC;CACvC,MAAM,MAA6C,CAAC;CACpD,KAAK,MAAM,OAAO,MAKhB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,eACJ,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,YACnD,IAAI,eACJ;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OAAO;EAChD,MAAM,eACJ,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,YACvD,MAAM,eACN;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,QAAiE;CACjG,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,SAAS,QAClB,KAAK,MAAM,OAAO,MAAM,OAAO,CAAC,GAAG;EACjC,MAAM,UAAU,IAAI,sBAAsB;EAC1C,IAAI,YAAY,IAAI,YAAY,KAAK;CACvC;CAEF,OAAO;AACT;AAEA,SAAS,WACP,MACA,QACA,WAC4B;CAC5B,MAAM,MAAkC,CAAC;CACzC,KAAK,MAAM,OAAO,MAChB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,YAAY,IAAI,QAAQ,IAAI;EAClC,IAAI,KAAK,EAAE,QAAQ,OAAO,cAAc,YAAY,YAAY,EAAE,CAAC;CACrE;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OACzC,IAAI,KAAK,EAAE,SAAS,MAAM,KAAK,UAAU,KAAK,EAAE,CAAC;CAGrD,OAAO;AACT;AAEA,SAAS,gBAAgB,UAAsD;CAC7E,MAAM,aAAa,SAAS,QAAQ,YAAgC,YAAY,IAAI;CACpF,IAAI,WAAW,WAAW,GAAG,OAAO;CACpC,OAAO,WAAW,OAAO,OAAO,CAAC,CAAC,SAAS,WAAW;AACxD;AAEA,SAAS,eAAe,KAAgB,WAAmC;CACzE,IAAI,uBAAuB,GAAG,GAAG,OAAO;CACxC,MAAM,QAAQ,cAAc,GAAG;CAC/B,OAAO,UAAU,KAAA,IAAY,OAAO,SAAS;AAC/C;AAEA,SAAS,uBAAuB,KAAyB;CACvD,OAAO,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB;AAChE;AAEA,SAAS,iBAAiB,OAA6B,WAAmC;CACxF,IAAI,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,WAAW,OAAO;CACjF,IAAI,MAAM,OAAO,OAAO,OAAO;CAC/B,IAAI,eAAe,MAAM,KAAK,GAAG,OAAO,MAAM,SAAS;CACvD,OAAO,MAAM,OAAO,OAAO,OAAO;AACpC;AAEA,SAAS,wBACP,SACkD;CAClD,OAAO,YAAY,YAAY,YAAY,eAAe,YAAY;AACxE;AAEA,SAAS,gBAAgB,SAA4D;CACnF,IAAI,YAAY,aAAa,OAAO;CACpC,IAAI,wBAAwB,OAAO,GAAG,OAAO;AAE/C;AAEA,SAAS,UAAU,MAA4B,OAA8B;CAC3E,OAAO,KACJ,QAAQ,QAAQ,IAAI,aAAa,KAAK,CAAC,CACvC,IAAI,aAAa,CAAC,CAClB,OAAO,cAAc;AAC1B;;;;;;;AAQA,SAAS,cAAc,KAAoC;CACzD,MAAM,QAAQ,mBAAmB,KAAK,IAAI,aAAa,YAAY,YAAY,QAAQ;CACvF,OAAO,eAAe,KAAK,IAAI,QAAQ,KAAA;AACzC;AAEA,SAAS,WAAW,IAAsC;CACxD,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,OAAO,GAAG,QAAQ,KAAK,MAAM,MAAM,GAAG,CAAC,IAAI,GAAG;AAChD;AAEA,SAAS,iBAAiB,IAAuB,GAA0B;CACzE,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,MAAM,SAAS,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,OAAO,OAAO,KAAK,IAAI,OAAO,SAAS,GAAG,KAAK,IAAI,GAAG,KAAK,KAAK,IAAI,OAAO,MAAM,IAAI,CAAC,CAAC;AACzF;AAEA,SAAS,eAAe,OAAiC;CACvD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK;AAC3D;AAEA,SAAS,WAAW,GAAkB,GAAiC;CACrE,IAAI,MAAM,QAAQ,MAAM,MAAM,OAAO;CACrC,OAAO,IAAI;AACb;AAEA,SAAS,SAAS,KAAoB,QAA+B;CACnE,IAAI,QAAQ,MAAM,OAAO;CACzB,IAAI,UAAU,GAAG,OAAO,OAAO,IAAI,IAAI;CACvC,OAAO,QAAQ,IAAI,KAAK,IAAI,GAAG,GAAG,IAAI,MAAM;AAC9C;AAEA,SAAS,gBACP,SACA,YACe;CACf,MAAM,OAAO,OAAO,SAAS,WAAW,cAAc,IAClD,QAAQ,gBAAgB,OACtB,OACA,QAAQ,WAAW,iBAAiB,KAAK,IAAI,QAAQ,aAAa,KAAK,CAAC,IAC1E;CACJ,MAAM,UAAU,OAAO,SAAS,WAAW,YAAY,IACnD,QAAQ,cAAc,OACpB,OACA,QAAQ,WAAW,eAAe,KAAK,IAAI,QAAQ,WAAW,KAAK,CAAC,IACtE;CACJ,IAAI,SAAS,QAAQ,YAAY,MAAM,OAAO;CAC9C,OAAO,KAAK,IAAI,MAAM,OAAO;AAC/B;AAEA,SAAS,QAAQ,GAAmB;CAClC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;AAEA,SAAS,cACP,QACA,QACA,SACA,QACQ;CACR,MAAM,SAAS,sBAAsB,OAAO,IAAI;CAChD,MAAM,aAAa,aAAa,QAAQ,cAAc,cAAc,QAAQ,WAAW,eAAe,QAAQ,YAAY,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS;CAC9L,IAAI,OAAO,WAAW,GAAG,OAAO,GAAG,OAAO,IAAI;CAC9C,OAAO,GAAG,OAAO,IAAI,WAAW,WAAW,OAAO,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,GAAG;AAC/E;AAEA,SAAS,IAAI,GAA0B;CACrC,IAAI,MAAM,MAAM,OAAO;CACvB,OAAO,EAAE,QAAQ,CAAC;AACpB;;;;;;;;;;AC9tBA,SAAgB,YACd,UACA,WACA,UAA4B,CAAC,GACZ;CACjB,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,aAAa,QAAQ,cAAc;CACzC,MAAM,WAAW,QAAQ,mBAAmB;CAC5C,MAAM,MAAM,WAAW,QAAQ,QAAQ,SAAS,UAAU,SAAS,CAAC;CAEpE,MAAM,eAAe,KAAK,QAAQ;CAClC,MAAM,gBAAgB,KAAK,SAAS;CACpC,MAAM,QAAQ,gBAAgB;CAE9B,IACE,SAAS,SAAS,UAAU,SAAS,YACrC,SAAS,WAAW,KACpB,UAAU,WAAW,GAErB,OAAO;EACL;EACA;EACA;EACA,SAAS;EACT,SAAS;EACT,YAAY;EACZ;EACA,SAAS;CACX;CAGF,MAAM,SAAmB,IAAI,MAAM,UAAU;CAC7C,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,YAAY,SAAS,UAAU,GAAG;EAExC,OAAO,KAAK,KADM,SAAS,WAAW,GACb,CAAC,IAAI,KAAK,SAAS;CAC9C;CACA,OAAO,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3B,MAAM,WAAW,KAAK,MAAO,QAAQ,IAAK,UAAU;CACpD,MAAM,WAAW,KAAK,OAAO,IAAI,QAAQ,KAAK,UAAU,IAAI;CAC5D,MAAM,UAAU,OAAO,KAAK,IAAI,GAAG,QAAQ;CAC3C,MAAM,UAAU,OAAO,KAAK,IAAI,aAAa,GAAG,QAAQ;CAExD,IAAI;CACJ,IAAI,UAAU,GAAG,UAAU;MACtB,IAAI,UAAU,GAAG,UAAU;MAC3B,IAAI,SAAS,GAAG,UAAU;MAC1B,UAAU;CAEf,OAAO;EACL;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,IAAI,KAAK;CACzB,OAAO,IAAI,GAAG;AAChB;AAEA,SAAS,SAAS,IAAc,KAA6B;CAC3D,MAAM,MAAM,IAAI,MAAM,GAAG,MAAM;CAC/B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,QAAQ,KAAK,IAAI,KAAK,GAAG,KAAK,MAAM,IAAI,IAAI,GAAG,MAAM;CAC5E,OAAO;AACT;;AAGA,SAAS,WAAW,MAA4B;CAC9C,IAAI,IAAI,SAAS;CACjB,aAAa;EACX,KAAK;EACL,IAAI,IAAI;EACR,IAAI,KAAK,KAAK,IAAK,MAAM,IAAK,IAAI,CAAC;EACnC,KAAK,IAAI,KAAK,KAAK,IAAK,MAAM,GAAI,IAAI,EAAE;EACxC,SAAS,IAAK,MAAM,QAAS,KAAK;CACpC;AACF;;AAGA,SAAS,SAAS,GAAa,GAAqB;CAClD,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,CAAC,GAAG,GAAG,GAAG,CAAC,GAAG;EAC5B,MAAM,OAAO,IAAI,aAAa,CAAC,CAAC,CAAC;EACjC,MAAM,QAAQ,IAAI,WAAW,KAAK,MAAM;EACxC,KAAK,MAAM,QAAQ,OAAO;GACxB,KAAK;GACL,IAAI,KAAK,KAAK,GAAG,QAAQ;EAC3B;CACF;CACA,OAAO,MAAM;AACf;;;;AAiCA,eAAsB,gBACpB,MACkF;CAClF,MAAM,cAAc,KAAK,oBAAoB;CAC7C,MAAM,iBAAiB,MAAM,SAAS,KAAK,iBAAiB,KAAK,OAAO,WAAW;CACnF,MAAM,kBAAkB,MAAM,SAAS,KAAK,kBAAkB,KAAK,OAAO,WAAW;CAMrF,OAAO;EACL,GANS,YAAY,gBAAgB,iBAAiB;GACtD,GAAI,KAAK,UAAU,KAAA,IAAY,EAAE,OAAO,KAAK,MAAM,IAAI,CAAC;GACxD,GAAI,KAAK,eAAe,KAAA,IAAY,EAAE,YAAY,KAAK,WAAW,IAAI,CAAC;GACvE,GAAI,KAAK,SAAS,KAAA,IAAY,EAAE,MAAM,KAAK,KAAK,IAAI,CAAC;EACvD,CAEM;EACJ,iBAAiB,eAAe;EAChC,kBAAkB,gBAAgB;CACpC;AACF;AAEA,eAAe,SACb,SACA,OACA,aACmB;CACnB,MAAM,UAAoB,IAAI,MAAM,QAAQ,MAAM;CAClD,IAAI,OAAO;CACX,eAAe,SAAwB;EACrC,OAAO,MAAM;GACX,MAAM,IAAI;GACV,IAAI,KAAK,QAAQ,QAAQ;GACzB,MAAM,IAAI,MAAM,MAAM,QAAQ,EAAG;GACjC,QAAQ,KAAK,OAAO,SAAS,CAAC,IAAI,IAAI;EACxC;CACF;CACA,MAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,GAAG,WAAW,EAAE,SAAS,OAAO,CAAC,CAAC;CAClF,OAAO;AACT;;;AC1NA,SAAgB,oBACd,WACA,UAAsC,CAAC,GAC/B;CACR,MAAM,QAAQ,QAAQ,SAAS,mBAAmB,UAAU;CAC5D,MAAM,QAAkB,CAAC;CACzB,MAAM,KAAK,KAAK,OAAO;CACvB,MAAM,KAAK,EAAE;CACb,MAAM,KAAK,aAAa,UAAU,OAAO,YAAY,EAAE,GAAG;CAC1D,MAAM,KAAK,cAAc,UAAU,UAAU,QAAQ,KAAK,GAAG;CAC7D,IAAI,UAAU,aAAa,MAAM,KAAK,gBAAgB,UAAU,YAAY,GAAG;CAC/E,IAAI,UAAU,YAAY,MAAM,KAAK,eAAe,UAAU,WAAW,GAAG;CAC5E,MAAM,KAAK,EAAE;CACb,MAAM,KAAK,UAAU,OAAO;CAC5B,MAAM,KAAK,EAAE;CAEb,MAAM,KAAK,YAAY;CACvB,MAAM,KAAK,EAAE;CACb,MAAM,KAAK,oBAAoB;CAC/B,MAAM,KAAK,YAAY;CACvB,MAAM,KAAK,iBAAiB,UAAU,QAAQ,cAAc,GAAG;CAC/D,MAAM,KAAK,mBAAmB,UAAU,QAAQ,WAAW,GAAG;CAC9D,MAAM,KAAK,oBAAoB,UAAU,QAAQ,YAAY,GAAG;CAChE,MAAM,KAAK,qBAAqB,UAAU,QAAQ,aAAa,GAAG;CAClE,MAAM,KAAK,yBAAyB,UAAU,QAAQ,oBAAoB,GAAG;CAC7E,MAAM,KAAK,iBAAiB,IAAI,UAAU,QAAQ,QAAQ,EAAE,GAAG;CAC/D,MAAM,KAAK,kBAAkB,IAAI,UAAU,QAAQ,SAAS,EAAE,GAAG;CACjE,MAAM,KAAK,mBAAmB,IAAI,UAAU,QAAQ,eAAe,EAAE,GAAG;CACxE,MAAM,KAAK,oBAAoB,IAAI,UAAU,QAAQ,gBAAgB,EAAE,GAAG;CAC1E,MAAM,KAAK,mBAAmB,IAAI,UAAU,QAAQ,UAAU,EAAE,GAAG;CACnE,MAAM,KAAK,kBAAkB,IAAI,UAAU,QAAQ,WAAW,EAAE,GAAG;CACnE,MAAM,KAAK,qBAAqB,SAAS,UAAU,QAAQ,SAAS,EAAE,GAAG;CACzE,MAAM,KAAK,EAAE;CAEb,IAAI,UAAU,OAAO,SAAS,GAAG;EAC/B,MAAM,KAAK,WAAW;EACtB,MAAM,KAAK,EAAE;EACb,KAAK,MAAM,SAAS,UAAU,QAC5B,MAAM,KAAK,OAAO,MAAM,SAAS,OAAO,MAAM,KAAK,MAAM,MAAM,KAAK,KAAK,MAAM,QAAQ;EAEzF,MAAM,KAAK,EAAE;CACf;CAEA,MAAM,WAAW,QAAQ,UAAU,QAAQ,wBAAwB;CACnE,IAAI,SAAS,SAAS,GAAG;EACvB,MAAM,KAAK,yBAAyB;EACpC,MAAM,KAAK,EAAE;EACb,KAAK,MAAM,CAAC,SAAS,UAAU,UAAU,MAAM,KAAK,KAAK,QAAQ,IAAI,OAAO;EAC5E,MAAM,KAAK,EAAE;CACf;CAEA,MAAM,WAAW,QAAQ,UAAU,QAAQ,kBAAkB;CAC7D,IAAI,SAAS,SAAS,GAAG;EACvB,MAAM,KAAK,oBAAoB;EAC/B,MAAM,KAAK,EAAE;EACb,KAAK,MAAM,CAAC,MAAM,UAAU,UAAU,MAAM,KAAK,KAAK,KAAK,IAAI,OAAO;EACtE,MAAM,KAAK,EAAE;CACf;CAEA,IAAI,QAAQ,QAAQ,QAAQ,KAAK,SAAS,GAAG;EAC3C,MAAM,KAAK,gBAAgB;EAC3B,MAAM,KAAK,EAAE;EACb,MAAM,KACJ,aAAa,CAAC,GAAG,QAAQ,IAAI,GAAG;GAC9B,YAAY,QAAQ,cAAc,UAAU,cAAc,KAAA;GAC1D,OAAO;EACT,CAAC,CAAC,CAAC,QACL;EACA,MAAM,KAAK,EAAE;CACf;CAEA,IAAI,QAAQ,wBAAwB,QAAQ,qBAAqB,SAAS,GAAG;EAC3E,MAAM,KAAK,0BAA0B;EACrC,MAAM,KAAK,EAAE;EACb,KAAK,MAAM,WAAW,QAAQ,sBAAsB,MAAM,KAAK,KAAK,SAAS;EAC7E,MAAM,KAAK,EAAE;CACf;CAEA,MAAM,cAAc,QAAQ,eAAe,mBAAmB,SAAS;CACvE,IAAI,YAAY,SAAS,GAAG;EAC1B,MAAM,KAAK,iBAAiB;EAC5B,MAAM,KAAK,EAAE;EACb,KAAK,MAAM,UAAU,aAAa,MAAM,KAAK,KAAK,QAAQ;EAC1D,MAAM,KAAK,EAAE;CACf;CAEA,OAAO,GAAG,MAAM,KAAK,IAAI,CAAC,CAAC,QAAQ,EAAE;AACvC;AAEA,SAAS,mBAAmB,WAAiD;CAC3E,IAAI,UAAU,SAAS,OAAO,CAAC,kDAAkD;CACjF,OAAO,UAAU,OACd,QAAQ,UAAU,MAAM,aAAa,UAAU,CAAC,CAChD,KAAK,UAAU,WAAW,MAAM,KAAK,IAAI,MAAM,QAAQ;AAC5D;AAEA,SAAS,QAAQ,QAAyD;CACxE,OAAO,OAAO,QAAQ,MAAM,CAAC,CAC1B,QAAQ,GAAG,WAAW,QAAQ,CAAC,CAAC,CAChC,MAAM,GAAG,MAAM,EAAE,KAAK,EAAE,MAAM,EAAE,EAAE,CAAC,cAAc,EAAE,EAAE,CAAC;AAC3D;AAEA,SAAS,IAAI,OAA8B;CACzC,OAAO,UAAU,OAAO,QAAQ,IAAI,QAAQ,IAAA,CAAK,QAAQ,CAAC,EAAE;AAC9D;AAEA,SAAS,IAAI,OAA8B;CACzC,OAAO,UAAU,OAAO,QAAQ,MAAM,QAAQ,CAAC;AACjD;AAEA,SAAS,SAAS,OAA8B;CAC9C,OAAO,UAAU,OAAO,QAAQ,GAAG,KAAK,MAAM,KAAK,EAAE;AACvD"}
@@ -1,7 +1,7 @@
1
1
  import { o as FailureClass } from "./schema-BtVldJ3T.js";
2
2
  import { a as RunRecord, s as RunSplitTag } from "./run-record-DcObtIGh.js";
3
3
  import { a as DatasetScenario, o as DatasetSplit, r as DatasetManifest } from "./dataset-BvtnC8Dc.js";
4
- import { b as GateDecision } from "./summary-report-CFnQgNfg.js";
4
+ import { b as GateDecision } from "./summary-report-DyOhItws.js";
5
5
  //#region src/release-confidence.d.ts
6
6
  /** Severity of an actionable finding attached to a run/trace. */
7
7
  type AsiSeverity = 'info' | 'warning' | 'error' | 'critical';
@@ -241,4 +241,4 @@ interface RenderReleaseReportOptions {
241
241
  declare function renderReleaseReport(scorecard: ReleaseConfidenceScorecard, options?: RenderReleaseReportOptions): string;
242
242
  //#endregion
243
243
  export { ReleaseConfidenceStatus as _, JudgeReplayGateArgs as a, assertReleaseConfidence as b, judgeReplayGate as c, ReleaseConfidenceAxis as d, ReleaseConfidenceAxisName as f, ReleaseConfidenceScorecard as g, ReleaseConfidenceMetrics as h, BootstrapResult as i, ActionableSideInfo as l, ReleaseConfidenceIssue as m, renderReleaseReport as n, Verdict as o, ReleaseConfidenceInput as p, BootstrapOptions as r, bootstrapCi as s, RenderReleaseReportOptions as t, AsiSeverity as u, ReleaseConfidenceThresholds as v, evaluateReleaseConfidence as x, ReleaseTraceEvidence as y };
244
- //# sourceMappingURL=release-report-CjHWa8Ia.d.ts.map
244
+ //# sourceMappingURL=release-report-DKBtegGt.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"release-report-CjHWa8Ia.d.ts","names":[],"sources":["../src/release-confidence.ts","../src/promotion-gate.ts","../src/release-report.ts"],"mappings":";;;;;;KAsBY;;UAGK;;EAEf;;EAEA;EACA,WAAW;;EAEX;;EAEA;;EAEA;;EAEA;EACA,WAAW;;KAGD;KACA;UAQK;EACf;EACA;EACA,QAAQ;EACR;EACA;EACA;EACA;EACA;;EAEA,eAAe;EACf,MAAM;EACN,WAAW;;UAGI;;EAEf;EACA;EACA;EACA;;EAEA;EACA;EACA;;EAEA;EACA;EACA;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;EACA,UAAU;EACV,qBAAqB;EACrB,gBAAgB;EAChB,kBAAkB;EAClB,eAAe;EACf,aAAa;;UAGE;EACf,MAAM;EACN,QAAQ;EACR;EACA;;UAGe;EACf,MAAM;EACN;EACA;EACA;;UAGe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,aAAa,OAAO;EACpB,cAAc;EACd,oBAAoB,QAAQ,OAAO;EACnC,0BAA0B;;;;;;;EAO1B;;UAGe;EACf;EACA;EACA;EACA,QAAQ;EACR;EACA,MAAM;EACN,QAAQ;EACR,SAAS;EACT,SAAS;EACT,cAAc;EACd;;iBAkBc,0BACd,OAAO,yBACN;iBA4Ha,wBAAwB,OAAO,yBAAyB;;;;;;;;;;;;;;;;;;;;;;;;;;;KCxR5D;UAEK;EACf;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA,SAAS;;UAGM;;EAEf;;EAEA;;;;;;EAMA;;EAEA;;;;;;;;;iBAUc,YACd,oBACA,qBACA,UAAS,mBACR;;;;;;;;;;;;;;;;UA+Gc,oBAAoB;EACnC,iBAAiB;EACjB,kBAAkB;;EAElB,QAAQ,QAAQ,YAAY;EAC5B;EACA;;EAEA;;EAEA;;;;;iBAMoB,gBAAgB,SACpC,MAAM,oBAAoB,WACzB,QAAQ;EAAoB;EAAyB;;;;UCjMvC;EACf;EACA,gBAAgB;EAChB;EACA;EACA;;iBAGc,oBACd,WAAW,4BACX,UAAS"}
1
+ {"version":3,"file":"release-report-DKBtegGt.d.ts","names":[],"sources":["../src/release-confidence.ts","../src/promotion-gate.ts","../src/release-report.ts"],"mappings":";;;;;;KAsBY;;UAGK;;EAEf;;EAEA;EACA,WAAW;;EAEX;;EAEA;;EAEA;;EAEA;EACA,WAAW;;KAGD;KACA;UAQK;EACf;EACA;EACA,QAAQ;EACR;EACA;EACA;EACA;EACA;;EAEA,eAAe;EACf,MAAM;EACN,WAAW;;UAGI;;EAEf;EACA;EACA;EACA;;EAEA;EACA;EACA;;EAEA;EACA;EACA;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;EACA,UAAU;EACV,qBAAqB;EACrB,gBAAgB;EAChB,kBAAkB;EAClB,eAAe;EACf,aAAa;;UAGE;EACf,MAAM;EACN,QAAQ;EACR;EACA;;UAGe;EACf,MAAM;EACN;EACA;EACA;;UAGe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,aAAa,OAAO;EACpB,cAAc;EACd,oBAAoB,QAAQ,OAAO;EACnC,0BAA0B;;;;;;;EAO1B;;UAGe;EACf;EACA;EACA;EACA,QAAQ;EACR;EACA,MAAM;EACN,QAAQ;EACR,SAAS;EACT,SAAS;EACT,cAAc;EACd;;iBAkBc,0BACd,OAAO,yBACN;iBA4Ha,wBAAwB,OAAO,yBAAyB;;;;;;;;;;;;;;;;;;;;;;;;;;;KCxR5D;UAEK;EACf;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA,SAAS;;UAGM;;EAEf;;EAEA;;;;;;EAMA;;EAEA;;;;;;;;;iBAUc,YACd,oBACA,qBACA,UAAS,mBACR;;;;;;;;;;;;;;;;UA+Gc,oBAAoB;EACnC,iBAAiB;EACjB,kBAAkB;;EAElB,QAAQ,QAAQ,YAAY;EAC5B;EACA;;EAEA;;EAEA;;;;;iBAMoB,gBAAgB,SACpC,MAAM,oBAAoB,WACzB,QAAQ;EAAoB;EAAyB;;;;UCjMvC;EACf;EACA,gBAAgB;EAChB;EACA;EACA;;iBAGc,oBACd,WAAW,4BACX,UAAS"}
@@ -1,6 +1,6 @@
1
- import { I as pairedBootstrap, Z as wilcoxonSignedRank, d as PairedBootstrapOptions, f as PairedBootstrapResult, y as benjaminiHochberg } from "./statistics-DbvkkDPa.js";
2
- import { _ as paretoChart, a as ParetoPoint, c as ResearchReportCandidate, d as ResearchReportOptions, f as ResearchReportRecommendation, g as gainHistogram, h as SummaryTableRow, i as ParetoFigureSpec, l as ResearchReportDecision, m as SummaryTableOptions, n as GainDistributionFigureSpec, o as RESEARCH_REPORT_HARD_PAIR_FLOOR, p as SummaryTable, r as GainDistributionOptions, s as ResearchReport, t as GainDistributionBin, u as ResearchReportMethodology, v as researchReport, y as summaryTable } from "./summary-report-CFnQgNfg.js";
3
- import { _ as ReleaseConfidenceStatus, a as JudgeReplayGateArgs, b as assertReleaseConfidence, c as judgeReplayGate, d as ReleaseConfidenceAxis, f as ReleaseConfidenceAxisName, g as ReleaseConfidenceScorecard, h as ReleaseConfidenceMetrics, i as BootstrapResult, m as ReleaseConfidenceIssue, n as renderReleaseReport, o as Verdict, p as ReleaseConfidenceInput, r as BootstrapOptions, s as bootstrapCi, t as RenderReleaseReportOptions, v as ReleaseConfidenceThresholds, x as evaluateReleaseConfidence, y as ReleaseTraceEvidence } from "./release-report-CjHWa8Ia.js";
1
+ import { A as benjaminiHochberg, _ as PairedBootstrapResult, ct as wilcoxonSignedRank, g as PairedBootstrapOptions, q as pairedBootstrap } from "./statistics-D_4Snl-5.js";
2
+ import { _ as paretoChart, a as ParetoPoint, c as ResearchReportCandidate, d as ResearchReportOptions, f as ResearchReportRecommendation, g as gainHistogram, h as SummaryTableRow, i as ParetoFigureSpec, l as ResearchReportDecision, m as SummaryTableOptions, n as GainDistributionFigureSpec, o as RESEARCH_REPORT_HARD_PAIR_FLOOR, p as SummaryTable, r as GainDistributionOptions, s as ResearchReport, t as GainDistributionBin, u as ResearchReportMethodology, v as researchReport, y as summaryTable } from "./summary-report-DyOhItws.js";
3
+ import { _ as ReleaseConfidenceStatus, a as JudgeReplayGateArgs, b as assertReleaseConfidence, c as judgeReplayGate, d as ReleaseConfidenceAxis, f as ReleaseConfidenceAxisName, g as ReleaseConfidenceScorecard, h as ReleaseConfidenceMetrics, i as BootstrapResult, m as ReleaseConfidenceIssue, n as renderReleaseReport, o as Verdict, p as ReleaseConfidenceInput, r as BootstrapOptions, s as bootstrapCi, t as RenderReleaseReportOptions, v as ReleaseConfidenceThresholds, x as evaluateReleaseConfidence, y as ReleaseTraceEvidence } from "./release-report-DKBtegGt.js";
4
4
  import { a as PairedEvalueStep, c as pairedEvalueSequence, i as PairedEvalueSequence, n as InterimReleaseConfidenceInput, o as SequentialDecision, r as PairedEvalueOptions, s as evaluateInterimReleaseConfidence, t as InterimReleaseConfidence } from "./sequential-CYwq6Ff_.js";
5
5
  import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "./rubric-predictive-validity-C1dCLcvb.js";
6
6
  export { type BootstrapOptions, type BootstrapResult, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeReplayGateArgs, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type ParetoFigureSpec, type ParetoPoint, RESEARCH_REPORT_HARD_PAIR_FLOOR, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type RubricOutcomePair, type RubricPredictiveValidityInput, type RubricPredictiveValidityReport, type RubricRanking, type SequentialDecision, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type Verdict, assertReleaseConfidence, benjaminiHochberg, bootstrapCi, evaluateInterimReleaseConfidence, evaluateReleaseConfidence, gainHistogram, judgeReplayGate, pairedBootstrap, pairedEvalueSequence, paretoChart, renderReleaseReport, researchReport, rubricPredictiveValidity, summaryTable, wilcoxonSignedRank };
package/dist/reporting.js CHANGED
@@ -1,6 +1,6 @@
1
- import { N as wilcoxonSignedRank, t as benjaminiHochberg, v as pairedBootstrap } from "./statistics-DWM_AyLe.js";
2
- import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-wuilQkvK.js";
3
- import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-Ci17nIdU.js";
1
+ import { C as pairedBootstrap, R as wilcoxonSignedRank, o as benjaminiHochberg } from "./statistics-RwRNu2__.js";
2
+ import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-BVZBmRZp.js";
3
+ import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-BxtossFi.js";
4
4
  import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
5
- import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-QG7ydk0s.js";
5
+ import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-D6Q6n9oq.js";
6
6
  export { RESEARCH_REPORT_HARD_PAIR_FLOOR, assertReleaseConfidence, benjaminiHochberg, bootstrapCi, evaluateInterimReleaseConfidence, evaluateReleaseConfidence, gainHistogram, judgeReplayGate, pairedBootstrap, pairedEvalueSequence, paretoChart, renderReleaseReport, researchReport, rubricPredictiveValidity, summaryTable, wilcoxonSignedRank };
@@ -5,7 +5,7 @@ import { s as TraceStore } from "./store-CT9YIIve.js";
5
5
  import { i as TraceEmitter, t as RunCompleteHook } from "./emitter-DGQGoLyj.js";
6
6
  import { a as RunIntegrityReport, n as RunIntegrityExpectations } from "./integrity-rmVhXWA7.js";
7
7
  import { o as LlmClientOptions, u as LlmRouteRequirements } from "./llm-client-BiK4HW0u.js";
8
- import { b as GateDecision, d as ResearchReportOptions, s as ResearchReport } from "./summary-report-CFnQgNfg.js";
8
+ import { b as GateDecision, d as ResearchReportOptions, s as ResearchReport } from "./summary-report-DyOhItws.js";
9
9
  //#region src/eval-campaign.d.ts
10
10
  interface CampaignVariant<V> {
11
11
  id: string;
@@ -312,4 +312,4 @@ declare class NoopResearcher implements Researcher {
312
312
  }
313
313
  //#endregion
314
314
  export { EvalCampaignResult as _, FailureMode as a, SteeringChange as c, CampaignRunContext as d, CampaignRunOutcome as f, EvalCampaignOptions as g, CampaignVariant as h, ExperimentResult as i, CampaignFactoryParams as l, CampaignScenario as m, CallbackResearcherOptions as n, NoopResearcher as o, CampaignRunner as p, ExperimentPlan as r, Researcher as s, CallbackResearcher as t, CampaignIntegrityPolicy as u, FailedRun as v, runEvalCampaign as y };
315
- //# sourceMappingURL=researcher-CbSKhK8z.d.ts.map
315
+ //# sourceMappingURL=researcher-BtD5U1Up.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"researcher-CbSKhK8z.d.ts","names":[],"sources":["../src/eval-campaign.ts","../src/researcher.ts"],"mappings":";;;;;;;;;UAyEiB,gBAAgB;EAC/B;EACA,SAAS;;UAGM;EACf;;EAEA,OAAO;;UAGQ,mBAAmB;;EAElC;;EAEA;EACA,SAAS;EACT;EACA;EACA,cAAc;EACd;EACA,UAAU;;;;;;;EAOV,SAAS;EACT,OAAO;EACP,SAAS;;;;;;EAMT,SAAS;;UAGD;;EAER;;EAEA;;EAEA;;EAEA,gBAAgB;EAChB,YAAY;;EAEZ;;EAEA;;EAEA;;EAEA,MAAM;;EAEN,gBAAgB;;;;;;EAMhB,cAAc;;;;;;EAMd,eAAe,mBAAmB;;;KAIxB,qBAAqB,2BAA2B;KAEhD,eAAe,MAAM,KAAK,mBAAmB,OAAO,QAAQ;KAE5D;UAEK,oBAAoB;;;;;EAKnC;EACA,UAAU,gBAAgB;EAC1B,WAAW;;EAEX;;EAEA,WAAW;;EAEX;;;;;;EAMA,SAAS;;;;;;EAMT,oBAAoB;;;;;;EAMpB,eAAe,QAAQ,0BAA0B;;;;;;;EAOjD,kBAAkB,QAAQ,0BAA0B;;;;;EAKpD;;;;;EAKA,gBAAgB;;;;;;EAMhB,YAAY;;EAEZ,qBAAqB;;;;;EAKrB,QAAQ,eAAe;;;;;EAKvB;IAAW;MAAwB,KACjC;;;;;EAOF;;EAEA;;;;EAIA;;EAEA,SAAS,QAAQ;;;;;;;EAOjB,eACI,mBACA,0BAEE,QAAQ;IACN,SAAS;IACT,cAAc;QAGd,mBACA,wBACA,QAAQ,mBAAmB;;UAGpB;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;;EAEA;EACA;;EAEA,MAAM;;EAEN,kBAAkB;EAClB,YAAY;;EAEZ,SAAS;EACT;EACA;;iBAgBoB,gBAAgB,GACpC,MAAM,oBAAoB,KACzB,QAAQ;;;;UC5QM;;;EAGf;;EAEA;EACA;;;IAGE;;IAEA;;;;UAKa;EACf;;;;EAIA;;;EAGA;;EAEA;;;UAIe;EACf;EACA;EACA,SAAS;;;EAGT;;EAEA;IAAU;IAAkB;;;;UAIb;EACf,MAAM;EACN,MAAM;EACN,cAAc;;;;;;;;;;;;;;;;;;;UAoBC;EACf,gBAAgB,MAAM,cAAc,QAAQ;EAC5C,cAAc,UAAU,gBAAgB,QAAQ;EAChD,YAAY,SAAS,kBAAkB,UAAU,iBAAiB,QAAQ;EAC1E,eAAe,MAAM,iBAAiB,QAAQ;;UAG/B;EACf,iBAAiB;EACjB,eAAe;EACf,aAAa;EACb,gBAAgB;;;;;;cAOL,8BAA8B;mBACZ;EAA7B,YAA6B,WAAW;EAExC,gBAAgB,MAAM,cAAc,QAAQ;EAI5C,cAAc,UAAU,gBAAgB,QAAQ;EAIhD,YAAY,SAAS,kBAAkB,UAAU,iBAAiB,QAAQ;EAI1E,eAAe,MAAM,iBAAiB,QAAQ;;;;;;;;;cAYnC,0BAA0B;mBACpB;EAEjB,YAAY;EAIN,gBAAgB,OAAO,cAAc,QAAQ;EAI7C,cAAc,WAAW,gBAAgB,QAAQ;EAIjD,YACJ,UAAU,kBACV,WAAW,iBACV,QAAQ;EAIL,eAAe,OAAO,iBAAiB,QAAQ"}
1
+ {"version":3,"file":"researcher-BtD5U1Up.d.ts","names":[],"sources":["../src/eval-campaign.ts","../src/researcher.ts"],"mappings":";;;;;;;;;UAyEiB,gBAAgB;EAC/B;EACA,SAAS;;UAGM;EACf;;EAEA,OAAO;;UAGQ,mBAAmB;;EAElC;;EAEA;EACA,SAAS;EACT;EACA;EACA,cAAc;EACd;EACA,UAAU;;;;;;;EAOV,SAAS;EACT,OAAO;EACP,SAAS;;;;;;EAMT,SAAS;;UAGD;;EAER;;EAEA;;EAEA;;EAEA,gBAAgB;EAChB,YAAY;;EAEZ;;EAEA;;EAEA;;EAEA,MAAM;;EAEN,gBAAgB;;;;;;EAMhB,cAAc;;;;;;EAMd,eAAe,mBAAmB;;;KAIxB,qBAAqB,2BAA2B;KAEhD,eAAe,MAAM,KAAK,mBAAmB,OAAO,QAAQ;KAE5D;UAEK,oBAAoB;;;;;EAKnC;EACA,UAAU,gBAAgB;EAC1B,WAAW;;EAEX;;EAEA,WAAW;;EAEX;;;;;;EAMA,SAAS;;;;;;EAMT,oBAAoB;;;;;;EAMpB,eAAe,QAAQ,0BAA0B;;;;;;;EAOjD,kBAAkB,QAAQ,0BAA0B;;;;;EAKpD;;;;;EAKA,gBAAgB;;;;;;EAMhB,YAAY;;EAEZ,qBAAqB;;;;;EAKrB,QAAQ,eAAe;;;;;EAKvB;IAAW;MAAwB,KACjC;;;;;EAOF;;EAEA;;;;EAIA;;EAEA,SAAS,QAAQ;;;;;;;EAOjB,eACI,mBACA,0BAEE,QAAQ;IACN,SAAS;IACT,cAAc;QAGd,mBACA,wBACA,QAAQ,mBAAmB;;UAGpB;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;;EAEA;EACA;;EAEA,MAAM;;EAEN,kBAAkB;EAClB,YAAY;;EAEZ,SAAS;EACT;EACA;;iBAgBoB,gBAAgB,GACpC,MAAM,oBAAoB,KACzB,QAAQ;;;;UC5QM;;;EAGf;;EAEA;EACA;;;IAGE;;IAEA;;;;UAKa;EACf;;;;EAIA;;;EAGA;;EAEA;;;UAIe;EACf;EACA;EACA,SAAS;;;EAGT;;EAEA;IAAU;IAAkB;;;;UAIb;EACf,MAAM;EACN,MAAM;EACN,cAAc;;;;;;;;;;;;;;;;;;;UAoBC;EACf,gBAAgB,MAAM,cAAc,QAAQ;EAC5C,cAAc,UAAU,gBAAgB,QAAQ;EAChD,YAAY,SAAS,kBAAkB,UAAU,iBAAiB,QAAQ;EAC1E,eAAe,MAAM,iBAAiB,QAAQ;;UAG/B;EACf,iBAAiB;EACjB,eAAe;EACf,aAAa;EACb,gBAAgB;;;;;;cAOL,8BAA8B;mBACZ;EAA7B,YAA6B,WAAW;EAExC,gBAAgB,MAAM,cAAc,QAAQ;EAI5C,cAAc,UAAU,gBAAgB,QAAQ;EAIhD,YAAY,SAAS,kBAAkB,UAAU,iBAAiB,QAAQ;EAI1E,eAAe,MAAM,iBAAiB,QAAQ;;;;;;;;;cAYnC,0BAA0B;mBACpB;EAEjB,YAAY;EAIN,gBAAgB,OAAO,cAAc,QAAQ;EAI7C,cAAc,WAAW,gBAAgB,QAAQ;EAIjD,YACJ,UAAU,kBACV,WAAW,iBACV,QAAQ;EAIL,eAAe,OAAO,iBAAiB,QAAQ"}