@tangle-network/agent-eval 0.161.0 → 0.163.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +51 -0
- package/dist/{active-curriculum-CD5TU2yW.js → active-curriculum-OjIWgrUJ.js} +2 -2
- package/dist/{active-curriculum-CD5TU2yW.js.map → active-curriculum-OjIWgrUJ.js.map} +1 -1
- package/dist/analyst/index.d.ts +2 -2
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +3 -3
- package/dist/{baseline-BhPRQBVn.js → baseline-BC-eBZ7U.js} +2 -2
- package/dist/{baseline-BhPRQBVn.js.map → baseline-BC-eBZ7U.js.map} +1 -1
- package/dist/{benchmark-command-BVtaq_ve.js → benchmark-command-CF-4GEWZ.js} +9 -8
- package/dist/benchmark-command-CF-4GEWZ.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +22 -8
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BSmOwskD.js → campaign-DQZmc2Dq.js} +12 -12
- package/dist/{campaign-BSmOwskD.js.map → campaign-DQZmc2Dq.js.map} +1 -1
- package/dist/{chat-json-call-6g5sJobJ.js → chat-json-call-5Jxna-aV.js} +2 -2
- package/dist/{chat-json-call-6g5sJobJ.js.map → chat-json-call-5Jxna-aV.js.map} +1 -1
- package/dist/cli.js +3 -3
- package/dist/contract/index.js +7 -7
- package/dist/{counterfactual-D_VWavVm.js → counterfactual-Bjq1mlUu.js} +2 -2
- package/dist/{counterfactual-D_VWavVm.js.map → counterfactual-Bjq1mlUu.js.map} +1 -1
- package/dist/{define-agent-eval-h-s-sI-v.js → define-agent-eval-D08pWIJb.js} +13 -13
- package/dist/{define-agent-eval-h-s-sI-v.js.map → define-agent-eval-D08pWIJb.js.map} +1 -1
- package/dist/{descriptive-jDOuI6mz.js → descriptive-1V17A-qa.js} +2 -2
- package/dist/{descriptive-jDOuI6mz.js.map → descriptive-1V17A-qa.js.map} +1 -1
- package/dist/{dspy-rlm-engine-DptEII26.js → dspy-rlm-engine-DhA9qKIm.js} +2 -2
- package/dist/{dspy-rlm-engine-DptEII26.js.map → dspy-rlm-engine-DhA9qKIm.js.map} +1 -1
- package/dist/emitter-D_jYSGRd.d.ts.map +1 -1
- package/dist/{emitter-BpYFQPj4.js → emitter-DeQHiDMm.js} +13 -6
- package/dist/emitter-DeQHiDMm.js.map +1 -0
- package/dist/{eval-campaign-BsXWL2-2.js → eval-campaign-BfohKmzx.js} +3 -3
- package/dist/{eval-campaign-BsXWL2-2.js.map → eval-campaign-BfohKmzx.js.map} +1 -1
- package/dist/experiment/index.js +8 -8
- package/dist/{experiment-tracker-Ym6rEQT1.js → experiment-tracker-BKEumQug.js} +2 -2
- package/dist/{experiment-tracker-Ym6rEQT1.js.map → experiment-tracker-BKEumQug.js.map} +1 -1
- package/dist/{external-optimizer-process-WosTBChy.js → external-optimizer-process-BFmh36vW.js} +2 -2
- package/dist/{external-optimizer-process-WosTBChy.js.map → external-optimizer-process-BFmh36vW.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-BIWbHpgD.js → external-optimizer-subprocess-CqLMW3nh.js} +2 -2
- package/dist/{external-optimizer-subprocess-BIWbHpgD.js.map → external-optimizer-subprocess-CqLMW3nh.js.map} +1 -1
- package/dist/fuzz.d.ts +1 -2
- package/dist/fuzz.d.ts.map +1 -1
- package/dist/fuzz.js +4 -10
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +3 -3
- package/dist/index.js +26 -26
- package/dist/index.js.map +1 -1
- package/dist/{internal-BDHPCnjk.js → internal-BMFSR8Ns.js} +3 -19
- package/dist/internal-BMFSR8Ns.js.map +1 -0
- package/dist/{judge-calibration-zZjLz8hr.js → judge-calibration-BnpVKtnb.js} +3 -13
- package/dist/judge-calibration-BnpVKtnb.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -1
- package/dist/{llm-client-hgDieDNN.js → llm-client-BFMRpmqb.js} +8 -11
- package/dist/llm-client-BFMRpmqb.js.map +1 -0
- package/dist/{llm-judge-BhasIPFT.js → llm-judge-Du7WQPh7.js} +7 -7
- package/dist/{llm-judge-BhasIPFT.js.map → llm-judge-Du7WQPh7.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -1
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +14 -22
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-D-XRF_fy.js → paired-arms-D4aeIHUy.js} +3 -3
- package/dist/{paired-arms-D-XRF_fy.js.map → paired-arms-D4aeIHUy.js.map} +1 -1
- package/dist/{paired-tests-BHIhYVdu.js → paired-tests-C8iCsioC.js} +3 -3
- package/dist/{paired-tests-BHIhYVdu.js.map → paired-tests-C8iCsioC.js.map} +1 -1
- package/dist/pipelines/index.d.ts +3 -2
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -19
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{power-and-mde-CHIrXJll.js → power-and-mde-B8F2RdcD.js} +3 -3
- package/dist/{power-and-mde-CHIrXJll.js.map → power-and-mde-B8F2RdcD.js.map} +1 -1
- package/dist/{power-preflight-DEw-uC7q.js → power-preflight-CFXm0Vjo.js} +3 -3
- package/dist/{power-preflight-DEw-uC7q.js.map → power-preflight-CFXm0Vjo.js.map} +1 -1
- package/dist/{produced-state-DZ89riy5.js → produced-state-Be0BK3RN.js} +3 -3
- package/dist/{produced-state-DZ89riy5.js.map → produced-state-Be0BK3RN.js.map} +1 -1
- package/dist/{promotion-policy-xzA40Evo.js → promotion-policy-LY9mVQ7W.js} +3 -3
- package/dist/{promotion-policy-xzA40Evo.js.map → promotion-policy-LY9mVQ7W.js.map} +1 -1
- package/dist/{query-DxPYqpmT.d.ts → query-CwnHlu5p.d.ts} +19 -2
- package/dist/query-CwnHlu5p.d.ts.map +1 -0
- package/dist/{query-CHmMP42p.js → query-_5g6re3_.js} +44 -2
- package/dist/query-_5g6re3_.js.map +1 -0
- package/dist/random-Dn5fPWkt.js +21 -0
- package/dist/random-Dn5fPWkt.js.map +1 -0
- package/dist/record-id-DUgsK5qp.js +17 -0
- package/dist/record-id-DUgsK5qp.js.map +1 -0
- package/dist/{release-confidence-DKfD2RYU.js → release-confidence-nGDJiiwc.js} +2 -2
- package/dist/{release-confidence-DKfD2RYU.js.map → release-confidence-nGDJiiwc.js.map} +1 -1
- package/dist/reporting.js +6 -6
- package/dist/{reward-hacking-t4lB1yt8.js → reward-hacking-O5zKDANP.js} +2 -2
- package/dist/{reward-hacking-t4lB1yt8.js.map → reward-hacking-O5zKDANP.js.map} +1 -1
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +9 -15
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-CK8SCOg-.js → rubric-predictive-validity-2D5Gw9z9.js} +3 -3
- package/dist/{rubric-predictive-validity-CK8SCOg-.js.map → rubric-predictive-validity-2D5Gw9z9.js.map} +1 -1
- package/dist/{semantic-concept-judge-BSkKKHeq.js → semantic-concept-judge-BsDMOwJr.js} +2 -2
- package/dist/{semantic-concept-judge-BSkKKHeq.js.map → semantic-concept-judge-BsDMOwJr.js.map} +1 -1
- package/dist/{sequential-rYW-Ophm.js → sequential-BLMbdrD7.js} +3 -3
- package/dist/{sequential-rYW-Ophm.js.map → sequential-BLMbdrD7.js.map} +1 -1
- package/dist/{server-BtFd4uzB.js → server-BjYiJHoJ.js} +2 -2
- package/dist/{server-BtFd4uzB.js.map → server-BjYiJHoJ.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DbaekMcn.js → skillopt-optimization-method-UArRo-nr.js} +6 -6
- package/dist/{skillopt-optimization-method-DbaekMcn.js.map → skillopt-optimization-method-UArRo-nr.js.map} +1 -1
- package/dist/{store-tool-spans-Dlh9vkFK.js → store-tool-spans-BVga3c37.js} +21 -17
- package/dist/store-tool-spans-BVga3c37.js.map +1 -0
- package/dist/store-tool-spans-DPUG7UUY.d.ts.map +1 -1
- package/dist/{student-t-CvBq2mve.js → student-t-BA-Uy51p.js} +2 -2
- package/dist/{student-t-CvBq2mve.js.map → student-t-BA-Uy51p.js.map} +1 -1
- package/dist/{summary-report-BI5hUtvK.js → summary-report-BXeQ5Ues.js} +6 -6
- package/dist/{summary-report-BI5hUtvK.js.map → summary-report-BXeQ5Ues.js.map} +1 -1
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{tool-waste-BqzmVdJk.js → tool-waste-8BQiUc8K.js} +4 -4
- package/dist/{tool-waste-BqzmVdJk.js.map → tool-waste-8BQiUc8K.js.map} +1 -1
- package/dist/{tool-waste-Dro0gJi3.d.ts → tool-waste-CKc7bYIg.d.ts} +2 -2
- package/dist/{tool-waste-Dro0gJi3.d.ts.map → tool-waste-CKc7bYIg.d.ts.map} +1 -1
- package/dist/trace-repair/index.js +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.d.ts.map +1 -1
- package/dist/traces.js +8 -17
- package/dist/traces.js.map +1 -1
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +3 -13
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/{types-BPb2Kf_C2.d.ts → types-BPb2Kf_C.d.ts} +1 -1
- package/dist/types-BPb2Kf_C.d.ts.map +1 -0
- package/dist/types-Bfk0uxRj.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/package.json +2 -2
- package/dist/benchmark-command-BVtaq_ve.js.map +0 -1
- package/dist/emitter-BpYFQPj4.js.map +0 -1
- package/dist/internal-BDHPCnjk.js.map +0 -1
- package/dist/judge-calibration-zZjLz8hr.js.map +0 -1
- package/dist/llm-client-hgDieDNN.js.map +0 -1
- package/dist/query-CHmMP42p.js.map +0 -1
- package/dist/query-DxPYqpmT.d.ts.map +0 -1
- package/dist/store-tool-spans-Dlh9vkFK.js.map +0 -1
- package/dist/types-BPb2Kf_C2.d.ts.map +0 -1
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"release-confidence-DKfD2RYU.js","names":[],"sources":["../src/promotion-gate.ts","../src/release-confidence.ts"],"sourcesContent":["/**\n * Bootstrap-CI promotion gate.\n *\n * In any iterative-improvement loop (GEPA, prompt evolution, dataset\n * curation), the question is \"did this generation actually improve, or are\n * we celebrating noise?\". With small N and noisy outcomes, point-estimate\n * deltas lie. Bootstrap confidence intervals tell the operator whether the\n * delta is real before code or prompts get promoted.\n *\n * This module is pure functions — no I/O, no model calls. Easy to unit-test\n * and to compose into any verdict gate.\n *\n * Default gate:\n * - Bootstrap mean baseline vs candidate (1k resamples).\n * - Compute the delta distribution; pass if the lower CI bound > 0.\n * - Tunable confidence (default 95%) and resample count.\n *\n * Verdict semantics intentionally match the existing `experiments.jsonl`\n * vocabulary:\n * - ADVANCE: candidate's CI lower bound > baseline mean (real win)\n * - KEEP: overlap, but candidate point estimate >= baseline (neutral)\n * - REVERT: candidate's CI upper bound < baseline mean (real regression)\n * - INCONCLUSIVE: not enough samples or CI straddles zero with no signal\n */\n\nimport { mulberry32 } from './statistics'\n\nexport type Verdict = 'ADVANCE' | 'KEEP' | 'REVERT' | 'INCONCLUSIVE'\n\nexport interface BootstrapResult {\n baselineMean: number\n candidateMean: number\n /** candidateMean - baselineMean, point estimate. */\n delta: number\n /** Lower bound of the (1 - alpha) CI on the delta. */\n ciLower: number\n /** Upper bound of the (1 - alpha) CI on the delta. */\n ciUpper: number\n /** Number of bootstrap resamples used. */\n iterations: number\n alpha: number\n verdict: Verdict\n}\n\nexport interface BootstrapOptions {\n /** Confidence level alpha (default 0.05 → 95% CI). */\n alpha?: number\n /** Number of resamples (default 1000). */\n iterations?: number\n /**\n * Minimum total samples (baseline + candidate) below which we always\n * return INCONCLUSIVE — bootstrap with too few samples is meaningless.\n * Default 6 (combined).\n */\n minTotalSamples?: number\n /** RNG seed for reproducibility. Absent, the seed is derived from the\n * compared arms, so the same inputs reproduce the same verdict. */\n seed?: number\n}\n\n/**\n * Compute the bootstrap CI on (candidateMean - baselineMean) and a verdict.\n *\n * Uses simple percentile bootstrap on the difference of resampled means.\n * That's the standard non-parametric primitive — no distributional\n * assumptions, robust to skew, easy to reason about.\n */\nexport function bootstrapCi(\n baseline: number[],\n candidate: number[],\n options: BootstrapOptions = {},\n): BootstrapResult {\n const alpha = options.alpha ?? 0.05\n const iterations = options.iterations ?? 1000\n const minTotal = options.minTotalSamples ?? 6\n const rng = mulberry32(options.seed ?? hashSeed(baseline, candidate))\n\n const baselineMean = mean(baseline)\n const candidateMean = mean(candidate)\n const delta = candidateMean - baselineMean\n\n if (\n baseline.length + candidate.length < minTotal ||\n baseline.length === 0 ||\n candidate.length === 0\n ) {\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower: -Infinity,\n ciUpper: Infinity,\n iterations: 0,\n alpha,\n verdict: 'INCONCLUSIVE',\n }\n }\n\n const deltas: number[] = new Array(iterations)\n for (let i = 0; i < iterations; i++) {\n const bResample = resample(baseline, rng)\n const cResample = resample(candidate, rng)\n deltas[i] = mean(cResample) - mean(bResample)\n }\n deltas.sort((a, b) => a - b)\n const lowerIdx = Math.floor((alpha / 2) * iterations)\n const upperIdx = Math.floor((1 - alpha / 2) * iterations) - 1\n const ciLower = deltas[Math.max(0, lowerIdx)]!\n const ciUpper = deltas[Math.min(iterations - 1, upperIdx)]!\n\n let verdict: Verdict\n // A ZERO-WIDTH interval is an absence of evidence, not a certainty. Constant\n // arms make every resample identical, so pass/fail data promotes on nothing:\n // [0,0,0] vs [1,1,1] is six samples, clears the `minTotalSamples` floor, and\n // yields [1, 1] — an \"ADVANCE\" carrying no information about how far the\n // estimate could be wrong. Same rule as `pairedDeltaTest`, one estimator over.\n if (!Number.isFinite(ciLower) || !Number.isFinite(ciUpper) || ciLower === ciUpper) {\n verdict = 'INCONCLUSIVE'\n } else if (ciLower > 0) verdict = 'ADVANCE'\n else if (ciUpper < 0) verdict = 'REVERT'\n else if (delta >= 0) verdict = 'KEEP'\n else verdict = 'INCONCLUSIVE'\n\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower,\n ciUpper,\n iterations,\n alpha,\n verdict,\n }\n}\n\nfunction mean(xs: number[]): number {\n if (xs.length === 0) return 0\n let s = 0\n for (const x of xs) s += x\n return s / xs.length\n}\n\nfunction resample(xs: number[], rng: () => number): number[] {\n const out = new Array(xs.length)\n for (let i = 0; i < xs.length; i++) out[i] = xs[Math.floor(rng() * xs.length)]\n return out\n}\n\n/** Stable seed derived from the inputs — same data → same CI bounds. */\nfunction hashSeed(a: number[], b: number[]): number {\n let h = 2166136261\n for (const x of [...a, ...b]) {\n const view = new Float64Array([x])\n const bytes = new Uint8Array(view.buffer)\n for (const byte of bytes) {\n h ^= byte\n h = Math.imul(h, 16777619)\n }\n }\n return h >>> 0\n}\n\n/**\n * Judge-replay promotion gate.\n *\n * The cheap inner-loop judge that drives an evolution run is by definition\n * fast and noisy. When you're about to promote a winning variant to the\n * canonical default, you want a STRONGER judge (a more expensive model, a\n * human grader, a separately-trained reward model) to confirm the win\n * generalises beyond the inner loop.\n *\n * This helper takes raw winner + baseline outputs, scores both through the\n * stronger judge, and applies `bootstrapCi`. ADVANCE means the stronger\n * judge agrees the winner is real with the configured confidence. Doesn't\n * matter what shape your \"output\" is — pass a string, an object, anything\n * the judge can read.\n */\nexport interface JudgeReplayGateArgs<TOutput> {\n baselineOutputs: TOutput[]\n candidateOutputs: TOutput[]\n /** Stronger judge — async to allow LLM calls. Return a 0..N scalar score. */\n judge: (output: TOutput) => Promise<number> | number\n alpha?: number\n iterations?: number\n /** RNG seed for reproducibility. */\n seed?: number\n /** Maximum concurrent judge calls. Default 4. */\n judgeConcurrency?: number\n}\n\n/**\n * Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.\n */\nexport async function judgeReplayGate<TOutput>(\n args: JudgeReplayGateArgs<TOutput>,\n): Promise<BootstrapResult & { baselineSamples: number; candidateSamples: number }> {\n const concurrency = args.judgeConcurrency ?? 4\n const baselineScores = await scoreAll(args.baselineOutputs, args.judge, concurrency)\n const candidateScores = await scoreAll(args.candidateOutputs, args.judge, concurrency)\n const ci = bootstrapCi(baselineScores, candidateScores, {\n ...(args.alpha !== undefined ? { alpha: args.alpha } : {}),\n ...(args.iterations !== undefined ? { iterations: args.iterations } : {}),\n ...(args.seed !== undefined ? { seed: args.seed } : {}),\n })\n return {\n ...ci,\n baselineSamples: baselineScores.length,\n candidateSamples: candidateScores.length,\n }\n}\n\nasync function scoreAll<TOutput>(\n outputs: TOutput[],\n judge: (output: TOutput) => Promise<number> | number,\n concurrency: number,\n): Promise<number[]> {\n const results: number[] = new Array(outputs.length)\n let next = 0\n async function worker(): Promise<void> {\n while (true) {\n const i = next++\n if (i >= outputs.length) return\n const v = await judge(outputs[i]!)\n results[i] = Number.isFinite(v) ? v : 0\n }\n }\n await Promise.all(Array.from({ length: Math.max(1, concurrency) }, () => worker()))\n return results\n}\n","/**\n * Release confidence gate.\n *\n * This is the production-facing composition layer over the lower-level\n * primitives:\n * - Dataset manifests prove corpus/version coverage.\n * - RunRecord rows prove reproducible search/holdout outcomes.\n * - Multi-shot trace evidence carries turn counts and ASI diagnostics.\n * - HeldOutGate decisions remain the paired promotion authority.\n *\n * The gate is intentionally pure and conservative. Missing declared evidence\n * fails closed instead of being treated as a neutral zero.\n */\n\nimport type { DatasetManifest, DatasetScenario, DatasetSplit } from './dataset'\nimport { ValidationError, VerificationError } from './errors'\nimport type { GateDecision } from './held-out-gate'\nimport { isRealnessGated, observedSplitScore } from './rollout/reward'\nimport { type RunRecord, type RunSplitTag, validateRunRecord } from './run-record'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Severity of an actionable finding attached to a run/trace. */\nexport type AsiSeverity = 'info' | 'warning' | 'error' | 'critical'\n\n/** Actionable side-info — a diagnosed finding the loop can act on. */\nexport interface ActionableSideInfo {\n /** Stable expectation/check id when available. */\n expectationId?: string\n /** Human-readable diagnosis of what happened. */\n message: string\n severity?: AsiSeverity\n /** Concrete trace excerpt, file path, tool call, screenshot id, etc. */\n evidence?: string\n /** Prompt/tool/context surface likely responsible. */\n responsibleSurface?: string\n /** Suggested fix in natural language. */\n suggestion?: string\n /** Whether this expectation was satisfied. Defaults to false for ASI rows. */\n matched?: boolean\n metadata?: Record<string, unknown>\n}\n\nexport type ReleaseConfidenceStatus = 'pass' | 'warn' | 'fail'\nexport type ReleaseConfidenceAxisName =\n | 'corpus'\n | 'quality'\n | 'reliability'\n | 'generalization'\n | 'diagnostics'\n | 'efficiency'\n\nexport interface ReleaseTraceEvidence {\n scenarioId: string\n candidateId?: string\n split?: RunSplitTag\n score?: number\n ok?: boolean\n turnCount?: number\n costUsd?: number\n durationMs?: number\n /** Canonical task-failure class. Free-form detail belongs in ASI. */\n failureClass?: FailureClass\n asi?: ActionableSideInfo[]\n metadata?: Record<string, unknown>\n}\n\nexport interface ReleaseConfidenceThresholds {\n /** Require a Dataset manifest or explicit scenarios. Default true. */\n requireCorpus?: boolean\n minScenarioCount?: number\n minSearchRuns?: number\n minHoldoutRuns?: number\n /** Require at least one holdout scenario/run. Default true. */\n requireHoldout?: boolean\n minPassRate?: number\n minMeanScore?: number\n /** Search mean may exceed holdout mean by at most this much. */\n maxOverfitGap?: number\n maxMeanCostUsd?: number\n maxP95WallMs?: number\n /** Low-score/failed rows must carry ASI. Default true. */\n requireAsiForFailures?: boolean\n /** Score below this is considered a failure for ASI coverage. Default 0.5. */\n failureScoreThreshold?: number\n}\n\nexport interface ReleaseConfidenceInput {\n target: string\n candidateId?: string\n baselineId?: string\n dataset?: DatasetManifest\n scenarios?: readonly DatasetScenario[]\n runs?: readonly RunRecord[]\n traces?: readonly ReleaseTraceEvidence[]\n gateDecision?: GateDecision | null\n thresholds?: ReleaseConfidenceThresholds\n}\n\nexport interface ReleaseConfidenceAxis {\n name: ReleaseConfidenceAxisName\n status: ReleaseConfidenceStatus\n score: number | null\n detail: string\n}\n\nexport interface ReleaseConfidenceIssue {\n axis: ReleaseConfidenceAxisName\n severity: 'critical' | 'warning'\n code: string\n detail: string\n}\n\nexport interface ReleaseConfidenceMetrics {\n scenarioCount: number\n /** Search rows with a finite search score. */\n searchRuns: number\n /** Holdout rows with a finite holdout score. */\n holdoutRuns: number\n /** Runs with neither a split-matched score nor an explicit task failure. */\n unscoredRuns: number\n /** Run rows, or trace rows when no runs exist, with no classified terminal result. */\n unclassifiedTerminalRuns: number\n /** Run rows, or trace rows when no runs exist, that ended unsuccessfully. */\n terminalFailureRuns: number\n /** Success fraction when every run or fallback trace row has a classified result. */\n reliabilityRate: number | null\n passRate: number | null\n meanScore: number | null\n searchMeanScore: number | null\n holdoutMeanScore: number | null\n overfitGap: number | null\n meanCostUsd: number | null\n p95WallMs: number | null\n failedRows: number\n failuresWithAsi: number\n singleShotTraces: number\n multiShotTraces: number\n splitCounts: Record<DatasetSplit, number>\n domainCounts: Record<string, number>\n failureClassCounts: Partial<Record<FailureClass, number>>\n responsibleSurfaceCounts: Record<string, number>\n /**\n * Runs excluded from `passRate` because the authenticity gate flagged them as\n * gamed. Surfaced, never silent: a release whose pass rate is computed over a\n * shrunken denominator has to say by how much, or the exclusion is just a\n * different way of hiding the same runs.\n */\n realnessGatedRuns: number\n}\n\nexport interface ReleaseConfidenceScorecard {\n target: string\n candidateId: string | null\n baselineId: string | null\n status: ReleaseConfidenceStatus\n promote: boolean\n axes: ReleaseConfidenceAxis[]\n issues: ReleaseConfidenceIssue[]\n metrics: ReleaseConfidenceMetrics\n dataset: DatasetManifest | null\n gateDecision: GateDecision | null\n summary: string\n}\n\nconst DEFAULT_THRESHOLDS: Required<ReleaseConfidenceThresholds> = {\n requireCorpus: true,\n minScenarioCount: 1,\n minSearchRuns: 1,\n minHoldoutRuns: 1,\n requireHoldout: true,\n minPassRate: 0.8,\n minMeanScore: 0.7,\n maxOverfitGap: 0.15,\n maxMeanCostUsd: Number.POSITIVE_INFINITY,\n maxP95WallMs: Number.POSITIVE_INFINITY,\n requireAsiForFailures: true,\n failureScoreThreshold: 0.5,\n}\n\nexport function evaluateReleaseConfidence(\n input: ReleaseConfidenceInput,\n): ReleaseConfidenceScorecard {\n const thresholds = { ...DEFAULT_THRESHOLDS, ...input.thresholds }\n const candidateId = input.candidateId ?? null\n const runs = filterCandidate(\n (input.runs ?? []).map(validateRunRecord),\n candidateId,\n input.baselineId,\n )\n const traces = filterTraceCandidate(\n (input.traces ?? []).map(validateReleaseTraceEvidence),\n candidateId,\n input.baselineId,\n )\n const scenarios = input.scenarios ?? []\n const scenarioCount = input.dataset?.scenarioCount ?? scenarios.length\n const splitCounts = input.dataset?.splitCounts ?? countScenarioSplits(scenarios)\n const searchScores = scoresFor(runs, 'search')\n const holdoutScores = scoresFor(runs, 'holdout')\n const runScores = runs.map(runSplitScore).filter(isFiniteNumber)\n const traceScores = traces.map((t) => t.score).filter(isFiniteNumber)\n const scoreUniverse = runs.length > 0 ? runScores : traceScores\n const qualityRuns = runs.filter((run) => runSplitScore(run) !== undefined)\n const unscoredRuns = runs.filter(\n (run) =>\n runSplitScore(run) === undefined &&\n !hasExplicitTaskFailure(run) &&\n !isFailedTerminalOutcome(run.terminalOutcome),\n ).length\n // Realness-gated runs are EXCLUDED from the pass rate — numerator AND\n // denominator — and the count of what was dropped ships beside the rate as\n // `metrics.realnessGatedRuns`. Counting a gamed run as a pass made faking a\n // success the cheapest way to improve a release scorecard; scoring it 0\n // instead would be the other error, silently deflating the rate with a run\n // the gate says carries no usable verdict at all.\n const honestRuns = runs.filter((run) => !isRealnessGated(run))\n const passOutcomes =\n runs.length > 0\n ? honestRuns.map((run) => runPassOutcome(run, thresholds.failureScoreThreshold))\n : traces.map((trace) => tracePassOutcome(trace, thresholds.failureScoreThreshold))\n const reliabilityRows =\n runs.length > 0\n ? runs.map((run) => terminalSuccess(run.terminalOutcome))\n : traces.map((trace) => trace.ok)\n const unclassifiedTerminalRuns = reliabilityRows.filter((outcome) => outcome === undefined).length\n const terminalFailureRuns = reliabilityRows.filter((outcome) => outcome === false).length\n const reliabilityRate =\n reliabilityRows.length === 0 || unclassifiedTerminalRuns > 0\n ? null\n : (reliabilityRows.length - terminalFailureRuns) / reliabilityRows.length\n const searchRuns = qualityRuns.filter((r) => r.splitTag === 'search').length\n const holdoutRuns = qualityRuns.filter((r) => r.splitTag === 'holdout').length\n const failed = failedRows(runs, traces, thresholds.failureScoreThreshold)\n const searchMeanScore = meanOrNull(searchScores)\n const holdoutMeanScore = meanOrNull(holdoutScores)\n const runCosts = runs.flatMap((run) =>\n run.costProvenance.kind === 'uncaptured' ? [] : [run.costProvenance.usd],\n )\n const traceCosts = traces.map((trace) => trace.costUsd).filter(isFiniteNumber)\n const meanCostUsd =\n runs.length > 0\n ? runCosts.length === runs.length\n ? meanOrNull(runCosts)\n : null\n : meanOrNull(traceCosts)\n const wallTimes =\n runs.length > 0\n ? runs.map((run) => run.wallMs)\n : traces.map((trace) => trace.durationMs).filter(isFiniteNumber)\n const metrics: ReleaseConfidenceMetrics = {\n scenarioCount,\n searchRuns,\n holdoutRuns,\n unscoredRuns,\n unclassifiedTerminalRuns,\n terminalFailureRuns,\n reliabilityRate,\n passRate: passOutcomeRate(passOutcomes),\n realnessGatedRuns: runs.length - honestRuns.length,\n meanScore: meanOrNull(scoreUniverse),\n searchMeanScore,\n holdoutMeanScore,\n overfitGap: diffOrNull(searchMeanScore, holdoutMeanScore),\n meanCostUsd,\n p95WallMs: percentileOrNull(wallTimes, 0.95),\n failedRows: failed.length,\n failuresWithAsi: failed.filter((row) => row.hasAsi).length,\n singleShotTraces: traces.filter((t) => t.turnCount === 1).length,\n multiShotTraces: traces.filter((t) => (t.turnCount ?? 0) > 1).length,\n splitCounts,\n domainCounts: countDomains(scenarios),\n failureClassCounts: countFailureClasses(runs, traces, thresholds.failureScoreThreshold),\n responsibleSurfaceCounts: countResponsibleSurfaces(traces),\n }\n\n const issues: ReleaseConfidenceIssue[] = []\n checkCorpus(input, thresholds, metrics, issues)\n checkQuality(thresholds, metrics, issues)\n checkReliability(metrics, issues)\n checkGeneralization(input.gateDecision ?? null, thresholds, metrics, issues)\n checkDiagnostics(thresholds, metrics, issues)\n checkEfficiency(thresholds, metrics, issues)\n\n const axes = buildAxes(metrics, thresholds, issues)\n const status = issues.some((i) => i.severity === 'critical')\n ? 'fail'\n : issues.length > 0\n ? 'warn'\n : 'pass'\n\n return {\n target: input.target,\n candidateId,\n baselineId: input.baselineId ?? null,\n status,\n promote: status === 'pass' && (input.gateDecision ? input.gateDecision.promote : true),\n axes,\n issues,\n metrics,\n dataset: input.dataset ?? null,\n gateDecision: input.gateDecision ?? null,\n summary: renderSummary(input.target, status, metrics, issues),\n }\n}\n\nexport function assertReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard {\n const scorecard = evaluateReleaseConfidence(input)\n if (scorecard.status === 'fail') {\n throw new VerificationError(scorecard.summary)\n }\n return scorecard\n}\n\nfunction filterCandidate(\n runs: readonly RunRecord[],\n candidateId: string | null,\n baselineId?: string,\n): RunRecord[] {\n if (candidateId) return runs.filter((r) => r.candidateId === candidateId)\n if (baselineId) return runs.filter((r) => r.candidateId !== baselineId)\n return [...runs]\n}\n\nfunction filterTraceCandidate(\n traces: readonly ReleaseTraceEvidence[],\n candidateId: string | null,\n baselineId?: string,\n): ReleaseTraceEvidence[] {\n if (candidateId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId === candidateId)\n if (baselineId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId !== baselineId)\n return [...traces]\n}\n\nfunction validateReleaseTraceEvidence(\n trace: ReleaseTraceEvidence,\n index: number,\n): ReleaseTraceEvidence {\n const value = trace as unknown as Record<string, unknown>\n if (Object.hasOwn(value, 'failureMode')) {\n throw new ValidationError(\n `traces[${index}].failureMode is not supported; use canonical failureClass`,\n )\n }\n if (trace.failureClass !== undefined && !FAILURE_CLASSES.includes(trace.failureClass)) {\n throw new ValidationError(\n `traces[${index}].failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n )\n }\n return trace\n}\n\nfunction checkCorpus(\n input: ReleaseConfidenceInput,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireCorpus && !input.dataset && (input.scenarios?.length ?? 0) === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_corpus',\n detail: 'No Dataset manifest or scenarios supplied.',\n })\n }\n if (metrics.scenarioCount < thresholds.minScenarioCount) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'few_scenarios',\n detail: `${metrics.scenarioCount} scenario(s) < min ${thresholds.minScenarioCount}.`,\n })\n }\n if (thresholds.requireHoldout && metrics.splitCounts.holdout === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_holdout_split',\n detail: 'Corpus has no holdout scenarios.',\n })\n }\n}\n\nfunction checkQuality(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.searchRuns < thresholds.minSearchRuns) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'few_search_runs',\n detail: `${metrics.searchRuns} search run(s) < min ${thresholds.minSearchRuns}.`,\n })\n }\n if (metrics.unscoredRuns > 0) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'unscored_runs',\n detail: `${metrics.unscoredRuns} supplied run(s) have no task result.`,\n })\n }\n if (metrics.passRate === null || metrics.meanScore === null) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'missing_quality_scores',\n detail: 'No task-quality scores are available for pass-rate and mean-score checks.',\n })\n }\n if (metrics.passRate !== null && metrics.passRate < thresholds.minPassRate) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_pass_rate',\n detail: `passRate ${fmt(metrics.passRate)} < ${fmt(thresholds.minPassRate)}.`,\n })\n }\n if (metrics.meanScore !== null && metrics.meanScore < thresholds.minMeanScore) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_mean_score',\n detail: `meanScore ${fmt(metrics.meanScore)} < ${fmt(thresholds.minMeanScore)}.`,\n })\n }\n}\n\nfunction checkReliability(\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.reliabilityRate === null) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'missing_reliability_evidence',\n detail:\n metrics.unclassifiedTerminalRuns > 0\n ? `${metrics.unclassifiedTerminalRuns} supplied run(s) have no classified terminal result.`\n : 'No classified terminal results are available.',\n })\n }\n if (metrics.terminalFailureRuns > 0) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'terminal_run_failures',\n detail: `${metrics.terminalFailureRuns} run(s) ended failed, cancelled, or incomplete.`,\n })\n }\n}\n\nfunction checkGeneralization(\n gateDecision: GateDecision | null,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireHoldout && metrics.holdoutRuns < thresholds.minHoldoutRuns) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'few_holdout_runs',\n detail: `${metrics.holdoutRuns} holdout run(s) < min ${thresholds.minHoldoutRuns}.`,\n })\n }\n if (metrics.overfitGap !== null && metrics.overfitGap > thresholds.maxOverfitGap) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'overfit_gap',\n detail: `search-holdout gap ${fmt(metrics.overfitGap)} > ${fmt(thresholds.maxOverfitGap)}.`,\n })\n }\n if (gateDecision && !gateDecision.promote) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: `gate_${gateDecision.rejectionCode ?? 'reject'}`,\n detail: gateDecision.reason,\n })\n }\n}\n\nfunction checkDiagnostics(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (!thresholds.requireAsiForFailures) return\n if (metrics.failedRows > metrics.failuresWithAsi) {\n issues.push({\n axis: 'diagnostics',\n severity: 'critical',\n code: 'missing_failure_asi',\n detail: `${metrics.failedRows - metrics.failuresWithAsi} failed row(s) have no actionable side information.`,\n })\n }\n}\n\nfunction checkEfficiency(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (Number.isFinite(thresholds.maxMeanCostUsd) && metrics.meanCostUsd === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_cost',\n detail: 'A finite cost limit was configured but no cost evidence is available.',\n })\n } else if (metrics.meanCostUsd !== null && metrics.meanCostUsd > thresholds.maxMeanCostUsd) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'cost_budget',\n detail: `meanCostUsd ${fmt(metrics.meanCostUsd)} > ${fmt(thresholds.maxMeanCostUsd)}.`,\n })\n }\n if (Number.isFinite(thresholds.maxP95WallMs) && metrics.p95WallMs === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_latency',\n detail: 'A finite latency limit was configured but no latency evidence is available.',\n })\n } else if (metrics.p95WallMs !== null && metrics.p95WallMs > thresholds.maxP95WallMs) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'latency_budget',\n detail: `p95WallMs ${fmt(metrics.p95WallMs)} > ${fmt(thresholds.maxP95WallMs)}.`,\n })\n }\n}\n\nfunction buildAxes(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n issues: ReleaseConfidenceIssue[],\n): ReleaseConfidenceAxis[] {\n return [\n axis(\n 'corpus',\n issues,\n bounded(metrics.scenarioCount / Math.max(1, thresholds.minScenarioCount)),\n `${metrics.scenarioCount} scenarios; holdout=${metrics.splitCounts.holdout}`,\n ),\n axis(\n 'quality',\n issues,\n metrics.passRate === null || metrics.meanScore === null\n ? null\n : Math.min(metrics.passRate, metrics.meanScore),\n `passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`,\n ),\n axis(\n 'reliability',\n issues,\n metrics.reliabilityRate,\n `successRate=${fmt(metrics.reliabilityRate)} terminalFailures=${metrics.terminalFailureRuns} unclassified=${metrics.unclassifiedTerminalRuns}`,\n ),\n axis(\n 'generalization',\n issues,\n gapScore(metrics.overfitGap, thresholds.maxOverfitGap),\n `holdoutRuns=${metrics.holdoutRuns} overfitGap=${fmt(metrics.overfitGap)}`,\n ),\n axis(\n 'diagnostics',\n issues,\n metrics.failedRows === 0 ? 1 : metrics.failuresWithAsi / metrics.failedRows,\n `failuresWithAsi=${metrics.failuresWithAsi}/${metrics.failedRows}`,\n ),\n axis(\n 'efficiency',\n issues,\n efficiencyScore(metrics, thresholds),\n `meanCostUsd=${fmt(metrics.meanCostUsd)} p95WallMs=${fmt(metrics.p95WallMs)}`,\n ),\n ]\n}\n\nfunction axis(\n name: ReleaseConfidenceAxisName,\n issues: ReleaseConfidenceIssue[],\n score: number | null,\n detail: string,\n): ReleaseConfidenceAxis {\n const own = issues.filter((i) => i.axis === name)\n const status = own.some((i) => i.severity === 'critical')\n ? 'fail'\n : own.length > 0\n ? 'warn'\n : 'pass'\n return { name, status, score: score === null ? null : bounded(score), detail }\n}\n\nfunction countScenarioSplits(scenarios: readonly DatasetScenario[]): Record<DatasetSplit, number> {\n const counts: Record<DatasetSplit, number> = { train: 0, dev: 0, test: 0, holdout: 0 }\n for (const scenario of scenarios) counts[scenario.split ?? 'train']++\n return counts\n}\n\nfunction countDomains(scenarios: readonly DatasetScenario[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const scenario of scenarios) {\n const domain = scenario.tags?.domain ?? scenario.tags?.category ?? 'uncategorized'\n out[domain] = (out[domain] ?? 0) + 1\n }\n return out\n}\n\nfunction countFailureClasses(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Partial<Record<FailureClass, number>> {\n const out: Partial<Record<FailureClass, number>> = {}\n for (const run of runs) {\n // Ungated: a failure-mode census counts what the runs REPORTED, which is\n // the only way a gamed run's inflated score is visible at all. It is not a\n // promotion number — `passRate` is, and that one excludes gated runs and\n // publishes the excluded count as `metrics.realnessGatedRuns`.\n if (runPassOutcome(run, threshold) === false) {\n const failureClass =\n run.failureClass !== undefined && run.failureClass !== 'success'\n ? run.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n const failureClass =\n trace.failureClass !== undefined && trace.failureClass !== 'success'\n ? trace.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction countResponsibleSurfaces(traces: readonly ReleaseTraceEvidence[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const trace of traces) {\n for (const asi of trace.asi ?? []) {\n const surface = asi.responsibleSurface ?? 'unknown'\n out[surface] = (out[surface] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction failedRows(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Array<{ hasAsi: boolean }> {\n const out: Array<{ hasAsi: boolean }> = []\n for (const run of runs) {\n if (runPassOutcome(run, threshold) === false) {\n const asiMetric = run.outcome.raw.asi\n out.push({ hasAsi: typeof asiMetric === 'number' && asiMetric > 0 })\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n out.push({ hasAsi: (trace.asi?.length ?? 0) > 0 })\n }\n }\n return out\n}\n\nfunction passOutcomeRate(outcomes: readonly (boolean | null)[]): number | null {\n const classified = outcomes.filter((outcome): outcome is boolean => outcome !== null)\n if (classified.length === 0) return null\n return classified.filter(Boolean).length / classified.length\n}\n\nfunction runPassOutcome(run: RunRecord, threshold: number): boolean | null {\n if (hasExplicitTaskFailure(run)) return false\n const score = runSplitScore(run)\n return score === undefined ? null : score >= threshold\n}\n\nfunction hasExplicitTaskFailure(run: RunRecord): boolean {\n return run.failureClass !== undefined && run.failureClass !== 'success'\n}\n\nfunction tracePassOutcome(trace: ReleaseTraceEvidence, threshold: number): boolean | null {\n if (trace.failureClass !== undefined && trace.failureClass !== 'success') return false\n if (trace.ok === false) return false\n if (isFiniteNumber(trace.score)) return trace.score >= threshold\n return trace.ok === true ? true : null\n}\n\nfunction isFailedTerminalOutcome(\n outcome: RunRecord['terminalOutcome'],\n): outcome is 'failed' | 'cancelled' | 'incomplete' {\n return outcome === 'failed' || outcome === 'cancelled' || outcome === 'incomplete'\n}\n\nfunction terminalSuccess(outcome: RunRecord['terminalOutcome']): boolean | undefined {\n if (outcome === 'succeeded') return true\n if (isFailedTerminalOutcome(outcome)) return false\n return undefined\n}\n\nfunction scoresFor(runs: readonly RunRecord[], split: RunSplitTag): number[] {\n return runs\n .filter((run) => run.splitTag === split)\n .map(runSplitScore)\n .filter(isFiniteNumber)\n}\n\n/**\n * RAW and split-exact (`observedSplitScore`): this feeds the per-split means,\n * the overfit gap, and the pass threshold — descriptions of what the runs\n * reported. A gamed run inflating them is visible next to `realnessGatedRuns`,\n * and the promotion number (`passRate`) excludes gated runs entirely.\n */\nfunction runSplitScore(run: RunRecord): number | undefined {\n const score = observedSplitScore(run, run.splitTag === 'holdout' ? 'holdout' : 'search')\n return isFiniteNumber(score) ? score : undefined\n}\n\nfunction meanOrNull(xs: readonly number[]): number | null {\n if (xs.length === 0) return null\n return xs.reduce((sum, x) => sum + x, 0) / xs.length\n}\n\nfunction percentileOrNull(xs: readonly number[], p: number): number | null {\n if (xs.length === 0) return null\n const sorted = [...xs].sort((a, b) => a - b)\n return sorted[Math.min(sorted.length - 1, Math.max(0, Math.ceil(p * sorted.length) - 1))]!\n}\n\nfunction isFiniteNumber(value: unknown): value is number {\n return typeof value === 'number' && Number.isFinite(value)\n}\n\nfunction diffOrNull(a: number | null, b: number | null): number | null {\n if (a === null || b === null) return null\n return a - b\n}\n\nfunction gapScore(gap: number | null, maxGap: number): number | null {\n if (gap === null) return null\n if (maxGap <= 0) return gap <= 0 ? 1 : 0\n return bounded(1 - Math.max(0, gap) / maxGap)\n}\n\nfunction efficiencyScore(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n): number | null {\n const cost = Number.isFinite(thresholds.maxMeanCostUsd)\n ? metrics.meanCostUsd === null\n ? null\n : bounded(thresholds.maxMeanCostUsd / Math.max(metrics.meanCostUsd, 1e-12))\n : 1\n const latency = Number.isFinite(thresholds.maxP95WallMs)\n ? metrics.p95WallMs === null\n ? null\n : bounded(thresholds.maxP95WallMs / Math.max(metrics.p95WallMs, 1e-12))\n : 1\n if (cost === null || latency === null) return null\n return Math.min(cost, latency)\n}\n\nfunction bounded(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n\nfunction renderSummary(\n target: string,\n status: ReleaseConfidenceStatus,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): string {\n const prefix = `release confidence ${status}: ${target}`\n const metricText = `scenarios=${metrics.scenarioCount} searchRuns=${metrics.searchRuns} holdoutRuns=${metrics.holdoutRuns} passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`\n if (issues.length === 0) return `${prefix}; ${metricText}`\n return `${prefix}; ${metricText}; issues=${issues.map((i) => i.code).join(',')}`\n}\n\nfunction fmt(x: number | null): string {\n if (x === null) return 'n/a'\n return x.toFixed(4)\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmEA,SAAgB,YACd,UACA,WACA,UAA4B,CAAC,GACZ;CACjB,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,aAAa,QAAQ,cAAc;CACzC,MAAM,WAAW,QAAQ,mBAAmB;CAC5C,MAAM,MAAM,WAAW,QAAQ,QAAQ,SAAS,UAAU,SAAS,CAAC;CAEpE,MAAM,eAAe,KAAK,QAAQ;CAClC,MAAM,gBAAgB,KAAK,SAAS;CACpC,MAAM,QAAQ,gBAAgB;CAE9B,IACE,SAAS,SAAS,UAAU,SAAS,YACrC,SAAS,WAAW,KACpB,UAAU,WAAW,GAErB,OAAO;EACL;EACA;EACA;EACA,SAAS;EACT,SAAS;EACT,YAAY;EACZ;EACA,SAAS;CACX;CAGF,MAAM,SAAmB,IAAI,MAAM,UAAU;CAC7C,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,YAAY,SAAS,UAAU,GAAG;EAExC,OAAO,KAAK,KADM,SAAS,WAAW,GACb,CAAC,IAAI,KAAK,SAAS;CAC9C;CACA,OAAO,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3B,MAAM,WAAW,KAAK,MAAO,QAAQ,IAAK,UAAU;CACpD,MAAM,WAAW,KAAK,OAAO,IAAI,QAAQ,KAAK,UAAU,IAAI;CAC5D,MAAM,UAAU,OAAO,KAAK,IAAI,GAAG,QAAQ;CAC3C,MAAM,UAAU,OAAO,KAAK,IAAI,aAAa,GAAG,QAAQ;CAExD,IAAI;CAMJ,IAAI,CAAC,OAAO,SAAS,OAAO,KAAK,CAAC,OAAO,SAAS,OAAO,KAAK,YAAY,SACxE,UAAU;MACL,IAAI,UAAU,GAAG,UAAU;MAC7B,IAAI,UAAU,GAAG,UAAU;MAC3B,IAAI,SAAS,GAAG,UAAU;MAC1B,UAAU;CAEf,OAAO;EACL;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,IAAI,KAAK;CACzB,OAAO,IAAI,GAAG;AAChB;AAEA,SAAS,SAAS,IAAc,KAA6B;CAC3D,MAAM,MAAM,IAAI,MAAM,GAAG,MAAM;CAC/B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,QAAQ,KAAK,IAAI,KAAK,GAAG,KAAK,MAAM,IAAI,IAAI,GAAG,MAAM;CAC5E,OAAO;AACT;;AAGA,SAAS,SAAS,GAAa,GAAqB;CAClD,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,CAAC,GAAG,GAAG,GAAG,CAAC,GAAG;EAC5B,MAAM,OAAO,IAAI,aAAa,CAAC,CAAC,CAAC;EACjC,MAAM,QAAQ,IAAI,WAAW,KAAK,MAAM;EACxC,KAAK,MAAM,QAAQ,OAAO;GACxB,KAAK;GACL,IAAI,KAAK,KAAK,GAAG,QAAQ;EAC3B;CACF;CACA,OAAO,MAAM;AACf;;;;AAiCA,eAAsB,gBACpB,MACkF;CAClF,MAAM,cAAc,KAAK,oBAAoB;CAC7C,MAAM,iBAAiB,MAAM,SAAS,KAAK,iBAAiB,KAAK,OAAO,WAAW;CACnF,MAAM,kBAAkB,MAAM,SAAS,KAAK,kBAAkB,KAAK,OAAO,WAAW;CAMrF,OAAO;EACL,GANS,YAAY,gBAAgB,iBAAiB;GACtD,GAAI,KAAK,UAAU,KAAA,IAAY,EAAE,OAAO,KAAK,MAAM,IAAI,CAAC;GACxD,GAAI,KAAK,eAAe,KAAA,IAAY,EAAE,YAAY,KAAK,WAAW,IAAI,CAAC;GACvE,GAAI,KAAK,SAAS,KAAA,IAAY,EAAE,MAAM,KAAK,KAAK,IAAI,CAAC;EACvD,CAEM;EACJ,iBAAiB,eAAe;EAChC,kBAAkB,gBAAgB;CACpC;AACF;AAEA,eAAe,SACb,SACA,OACA,aACmB;CACnB,MAAM,UAAoB,IAAI,MAAM,QAAQ,MAAM;CAClD,IAAI,OAAO;CACX,eAAe,SAAwB;EACrC,OAAO,MAAM;GACX,MAAM,IAAI;GACV,IAAI,KAAK,QAAQ,QAAQ;GACzB,MAAM,IAAI,MAAM,MAAM,QAAQ,EAAG;GACjC,QAAQ,KAAK,OAAO,SAAS,CAAC,IAAI,IAAI;EACxC;CACF;CACA,MAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,GAAG,WAAW,EAAE,SAAS,OAAO,CAAC,CAAC;CAClF,OAAO;AACT;;;AChEA,MAAM,qBAA4D;CAChE,eAAe;CACf,kBAAkB;CAClB,eAAe;CACf,gBAAgB;CAChB,gBAAgB;CAChB,aAAa;CACb,cAAc;CACd,eAAe;CACf,gBAAgB,OAAO;CACvB,cAAc,OAAO;CACrB,uBAAuB;CACvB,uBAAuB;AACzB;AAEA,SAAgB,0BACd,OAC4B;CAC5B,MAAM,aAAa;EAAE,GAAG;EAAoB,GAAG,MAAM;CAAW;CAChE,MAAM,cAAc,MAAM,eAAe;CACzC,MAAM,OAAO,iBACV,MAAM,QAAQ,CAAC,EAAA,CAAG,IAAI,iBAAiB,GACxC,aACA,MAAM,UACR;CACA,MAAM,SAAS,sBACZ,MAAM,UAAU,CAAC,EAAA,CAAG,IAAI,4BAA4B,GACrD,aACA,MAAM,UACR;CACA,MAAM,YAAY,MAAM,aAAa,CAAC;CACtC,MAAM,gBAAgB,MAAM,SAAS,iBAAiB,UAAU;CAChE,MAAM,cAAc,MAAM,SAAS,eAAe,oBAAoB,SAAS;CAC/E,MAAM,eAAe,UAAU,MAAM,QAAQ;CAC7C,MAAM,gBAAgB,UAAU,MAAM,SAAS;CAC/C,MAAM,YAAY,KAAK,IAAI,aAAa,CAAC,CAAC,OAAO,cAAc;CAC/D,MAAM,cAAc,OAAO,KAAK,MAAM,EAAE,KAAK,CAAC,CAAC,OAAO,cAAc;CACpE,MAAM,gBAAgB,KAAK,SAAS,IAAI,YAAY;CACpD,MAAM,cAAc,KAAK,QAAQ,QAAQ,cAAc,GAAG,MAAM,KAAA,CAAS;CACzE,MAAM,eAAe,KAAK,QACvB,QACC,cAAc,GAAG,MAAM,KAAA,KACvB,CAAC,uBAAuB,GAAG,KAC3B,CAAC,wBAAwB,IAAI,eAAe,CAChD,CAAC,CAAC;CAOF,MAAM,aAAa,KAAK,QAAQ,QAAQ,CAAC,gBAAgB,GAAG,CAAC;CAC7D,MAAM,eACJ,KAAK,SAAS,IACV,WAAW,KAAK,QAAQ,eAAe,KAAK,WAAW,qBAAqB,CAAC,IAC7E,OAAO,KAAK,UAAU,iBAAiB,OAAO,WAAW,qBAAqB,CAAC;CACrF,MAAM,kBACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,gBAAgB,IAAI,eAAe,CAAC,IACtD,OAAO,KAAK,UAAU,MAAM,EAAE;CACpC,MAAM,2BAA2B,gBAAgB,QAAQ,YAAY,YAAY,KAAA,CAAS,CAAC,CAAC;CAC5F,MAAM,sBAAsB,gBAAgB,QAAQ,YAAY,YAAY,KAAK,CAAC,CAAC;CACnF,MAAM,kBACJ,gBAAgB,WAAW,KAAK,2BAA2B,IACvD,QACC,gBAAgB,SAAS,uBAAuB,gBAAgB;CACvE,MAAM,aAAa,YAAY,QAAQ,MAAM,EAAE,aAAa,QAAQ,CAAC,CAAC;CACtE,MAAM,cAAc,YAAY,QAAQ,MAAM,EAAE,aAAa,SAAS,CAAC,CAAC;CACxE,MAAM,SAAS,WAAW,MAAM,QAAQ,WAAW,qBAAqB;CACxE,MAAM,kBAAkB,WAAW,YAAY;CAC/C,MAAM,mBAAmB,WAAW,aAAa;CACjD,MAAM,WAAW,KAAK,SAAS,QAC7B,IAAI,eAAe,SAAS,eAAe,CAAC,IAAI,CAAC,IAAI,eAAe,GAAG,CACzE;CACA,MAAM,aAAa,OAAO,KAAK,UAAU,MAAM,OAAO,CAAC,CAAC,OAAO,cAAc;CAC7E,MAAM,cACJ,KAAK,SAAS,IACV,SAAS,WAAW,KAAK,SACvB,WAAW,QAAQ,IACnB,OACF,WAAW,UAAU;CAC3B,MAAM,YACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,IAAI,MAAM,IAC5B,OAAO,KAAK,UAAU,MAAM,UAAU,CAAC,CAAC,OAAO,cAAc;CACnE,MAAM,UAAoC;EACxC;EACA;EACA;EACA;EACA;EACA;EACA;EACA,UAAU,gBAAgB,YAAY;EACtC,mBAAmB,KAAK,SAAS,WAAW;EAC5C,WAAW,WAAW,aAAa;EACnC;EACA;EACA,YAAY,WAAW,iBAAiB,gBAAgB;EACxD;EACA,WAAW,iBAAiB,WAAW,GAAI;EAC3C,YAAY,OAAO;EACnB,iBAAiB,OAAO,QAAQ,QAAQ,IAAI,MAAM,CAAC,CAAC;EACpD,kBAAkB,OAAO,QAAQ,MAAM,EAAE,cAAc,CAAC,CAAC,CAAC;EAC1D,iBAAiB,OAAO,QAAQ,OAAO,EAAE,aAAa,KAAK,CAAC,CAAC,CAAC;EAC9D;EACA,cAAc,aAAa,SAAS;EACpC,oBAAoB,oBAAoB,MAAM,QAAQ,WAAW,qBAAqB;EACtF,0BAA0B,yBAAyB,MAAM;CAC3D;CAEA,MAAM,SAAmC,CAAC;CAC1C,YAAY,OAAO,YAAY,SAAS,MAAM;CAC9C,aAAa,YAAY,SAAS,MAAM;CACxC,iBAAiB,SAAS,MAAM;CAChC,oBAAoB,MAAM,gBAAgB,MAAM,YAAY,SAAS,MAAM;CAC3E,iBAAiB,YAAY,SAAS,MAAM;CAC5C,gBAAgB,YAAY,SAAS,MAAM;CAE3C,MAAM,OAAO,UAAU,SAAS,YAAY,MAAM;CAClD,MAAM,SAAS,OAAO,MAAM,MAAM,EAAE,aAAa,UAAU,IACvD,SACA,OAAO,SAAS,IACd,SACA;CAEN,OAAO;EACL,QAAQ,MAAM;EACd;EACA,YAAY,MAAM,cAAc;EAChC;EACA,SAAS,WAAW,WAAW,MAAM,eAAe,MAAM,aAAa,UAAU;EACjF;EACA;EACA;EACA,SAAS,MAAM,WAAW;EAC1B,cAAc,MAAM,gBAAgB;EACpC,SAAS,cAAc,MAAM,QAAQ,QAAQ,SAAS,MAAM;CAC9D;AACF;AAEA,SAAgB,wBAAwB,OAA2D;CACjG,MAAM,YAAY,0BAA0B,KAAK;CACjD,IAAI,UAAU,WAAW,QACvB,MAAM,IAAI,kBAAkB,UAAU,OAAO;CAE/C,OAAO;AACT;AAEA,SAAS,gBACP,MACA,aACA,YACa;CACb,IAAI,aAAa,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,WAAW;CACxE,IAAI,YAAY,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,UAAU;CACtE,OAAO,CAAC,GAAG,IAAI;AACjB;AAEA,SAAS,qBACP,QACA,aACA,YACwB;CACxB,IAAI,aACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,WAAW;CAC1F,IAAI,YACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,UAAU;CACzF,OAAO,CAAC,GAAG,MAAM;AACnB;AAEA,SAAS,6BACP,OACA,OACsB;CACtB,MAAM,QAAQ;CACd,IAAI,OAAO,OAAO,OAAO,aAAa,GACpC,MAAM,IAAI,gBACR,UAAU,MAAM,2DAClB;CAEF,IAAI,MAAM,iBAAiB,KAAA,KAAa,CAAC,gBAAgB,SAAS,MAAM,YAAY,GAClF,MAAM,IAAI,gBACR,UAAU,MAAM,gCAAgC,gBAAgB,KAAK,IAAI,GAC3E;CAEF,OAAO;AACT;AAEA,SAAS,YACP,OACA,YACA,SACA,QACM;CACN,IAAI,WAAW,iBAAiB,CAAC,MAAM,YAAY,MAAM,WAAW,UAAU,OAAO,GACnF,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,gBAAgB,WAAW,kBACrC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,cAAc,qBAAqB,WAAW,iBAAiB;CACpF,CAAC;CAEH,IAAI,WAAW,kBAAkB,QAAQ,YAAY,YAAY,GAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;AAEL;AAEA,SAAS,aACP,YACA,SACA,QACM;CACN,IAAI,QAAQ,aAAa,WAAW,eAClC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,WAAW,uBAAuB,WAAW,cAAc;CAChF,CAAC;CAEH,IAAI,QAAQ,eAAe,GACzB,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa;CAClC,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,cAAc,MACrD,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,WAAW,WAAW,aAC7D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,YAAY,IAAI,QAAQ,QAAQ,EAAE,KAAK,IAAI,WAAW,WAAW,EAAE;CAC7E,CAAC;CAEH,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,iBACP,SACA,QACM;CACN,IAAI,QAAQ,oBAAoB,MAC9B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QACE,QAAQ,2BAA2B,IAC/B,GAAG,QAAQ,yBAAyB,wDACpC;CACR,CAAC;CAEH,IAAI,QAAQ,sBAAsB,GAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,oBAAoB;CACzC,CAAC;AAEL;AAEA,SAAS,oBACP,cACA,YACA,SACA,QACM;CACN,IAAI,WAAW,kBAAkB,QAAQ,cAAc,WAAW,gBAChE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,YAAY,wBAAwB,WAAW,eAAe;CACnF,CAAC;CAEH,IAAI,QAAQ,eAAe,QAAQ,QAAQ,aAAa,WAAW,eACjE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,sBAAsB,IAAI,QAAQ,UAAU,EAAE,KAAK,IAAI,WAAW,aAAa,EAAE;CAC3F,CAAC;CAEH,IAAI,gBAAgB,CAAC,aAAa,SAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM,QAAQ,aAAa,iBAAiB;EAC5C,QAAQ,aAAa;CACvB,CAAC;AAEL;AAEA,SAAS,iBACP,YACA,SACA,QACM;CACN,IAAI,CAAC,WAAW,uBAAuB;CACvC,IAAI,QAAQ,aAAa,QAAQ,iBAC/B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa,QAAQ,gBAAgB;CAC1D,CAAC;AAEL;AAEA,SAAS,gBACP,YACA,SACA,QACM;CACN,IAAI,OAAO,SAAS,WAAW,cAAc,KAAK,QAAQ,gBAAgB,MACxE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,gBAAgB,QAAQ,QAAQ,cAAc,WAAW,gBAC1E,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,eAAe,IAAI,QAAQ,WAAW,EAAE,KAAK,IAAI,WAAW,cAAc,EAAE;CACtF,CAAC;CAEH,IAAI,OAAO,SAAS,WAAW,YAAY,KAAK,QAAQ,cAAc,MACpE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cACtE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,UACP,SACA,YACA,QACyB;CACzB,OAAO;EACL,KACE,UACA,QACA,QAAQ,QAAQ,gBAAgB,KAAK,IAAI,GAAG,WAAW,gBAAgB,CAAC,GACxE,GAAG,QAAQ,cAAc,sBAAsB,QAAQ,YAAY,SACrE;EACA,KACE,WACA,QACA,QAAQ,aAAa,QAAQ,QAAQ,cAAc,OAC/C,OACA,KAAK,IAAI,QAAQ,UAAU,QAAQ,SAAS,GAChD,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS,GACtE;EACA,KACE,eACA,QACA,QAAQ,iBACR,eAAe,IAAI,QAAQ,eAAe,EAAE,oBAAoB,QAAQ,oBAAoB,gBAAgB,QAAQ,0BACtH;EACA,KACE,kBACA,QACA,SAAS,QAAQ,YAAY,WAAW,aAAa,GACrD,eAAe,QAAQ,YAAY,cAAc,IAAI,QAAQ,UAAU,GACzE;EACA,KACE,eACA,QACA,QAAQ,eAAe,IAAI,IAAI,QAAQ,kBAAkB,QAAQ,YACjE,mBAAmB,QAAQ,gBAAgB,GAAG,QAAQ,YACxD;EACA,KACE,cACA,QACA,gBAAgB,SAAS,UAAU,GACnC,eAAe,IAAI,QAAQ,WAAW,EAAE,aAAa,IAAI,QAAQ,SAAS,GAC5E;CACF;AACF;AAEA,SAAS,KACP,MACA,QACA,OACA,QACuB;CACvB,MAAM,MAAM,OAAO,QAAQ,MAAM,EAAE,SAAS,IAAI;CAMhD,OAAO;EAAE;EAAM,QALA,IAAI,MAAM,MAAM,EAAE,aAAa,UAAU,IACpD,SACA,IAAI,SAAS,IACX,SACA;EACiB,OAAO,UAAU,OAAO,OAAO,QAAQ,KAAK;EAAG;CAAO;AAC/E;AAEA,SAAS,oBAAoB,WAAqE;CAChG,MAAM,SAAuC;EAAE,OAAO;EAAG,KAAK;EAAG,MAAM;EAAG,SAAS;CAAE;CACrF,KAAK,MAAM,YAAY,WAAW,OAAO,SAAS,SAAS,QAAQ;CACnE,OAAO;AACT;AAEA,SAAS,aAAa,WAA+D;CACnF,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,YAAY,WAAW;EAChC,MAAM,SAAS,SAAS,MAAM,UAAU,SAAS,MAAM,YAAY;EACnE,IAAI,WAAW,IAAI,WAAW,KAAK;CACrC;CACA,OAAO;AACT;AAEA,SAAS,oBACP,MACA,QACA,WACuC;CACvC,MAAM,MAA6C,CAAC;CACpD,KAAK,MAAM,OAAO,MAKhB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,eACJ,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,YACnD,IAAI,eACJ;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OAAO;EAChD,MAAM,eACJ,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,YACvD,MAAM,eACN;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,QAAiE;CACjG,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,SAAS,QAClB,KAAK,MAAM,OAAO,MAAM,OAAO,CAAC,GAAG;EACjC,MAAM,UAAU,IAAI,sBAAsB;EAC1C,IAAI,YAAY,IAAI,YAAY,KAAK;CACvC;CAEF,OAAO;AACT;AAEA,SAAS,WACP,MACA,QACA,WAC4B;CAC5B,MAAM,MAAkC,CAAC;CACzC,KAAK,MAAM,OAAO,MAChB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,YAAY,IAAI,QAAQ,IAAI;EAClC,IAAI,KAAK,EAAE,QAAQ,OAAO,cAAc,YAAY,YAAY,EAAE,CAAC;CACrE;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OACzC,IAAI,KAAK,EAAE,SAAS,MAAM,KAAK,UAAU,KAAK,EAAE,CAAC;CAGrD,OAAO;AACT;AAEA,SAAS,gBAAgB,UAAsD;CAC7E,MAAM,aAAa,SAAS,QAAQ,YAAgC,YAAY,IAAI;CACpF,IAAI,WAAW,WAAW,GAAG,OAAO;CACpC,OAAO,WAAW,OAAO,OAAO,CAAC,CAAC,SAAS,WAAW;AACxD;AAEA,SAAS,eAAe,KAAgB,WAAmC;CACzE,IAAI,uBAAuB,GAAG,GAAG,OAAO;CACxC,MAAM,QAAQ,cAAc,GAAG;CAC/B,OAAO,UAAU,KAAA,IAAY,OAAO,SAAS;AAC/C;AAEA,SAAS,uBAAuB,KAAyB;CACvD,OAAO,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB;AAChE;AAEA,SAAS,iBAAiB,OAA6B,WAAmC;CACxF,IAAI,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,WAAW,OAAO;CACjF,IAAI,MAAM,OAAO,OAAO,OAAO;CAC/B,IAAI,eAAe,MAAM,KAAK,GAAG,OAAO,MAAM,SAAS;CACvD,OAAO,MAAM,OAAO,OAAO,OAAO;AACpC;AAEA,SAAS,wBACP,SACkD;CAClD,OAAO,YAAY,YAAY,YAAY,eAAe,YAAY;AACxE;AAEA,SAAS,gBAAgB,SAA4D;CACnF,IAAI,YAAY,aAAa,OAAO;CACpC,IAAI,wBAAwB,OAAO,GAAG,OAAO;AAE/C;AAEA,SAAS,UAAU,MAA4B,OAA8B;CAC3E,OAAO,KACJ,QAAQ,QAAQ,IAAI,aAAa,KAAK,CAAC,CACvC,IAAI,aAAa,CAAC,CAClB,OAAO,cAAc;AAC1B;;;;;;;AAQA,SAAS,cAAc,KAAoC;CACzD,MAAM,QAAQ,mBAAmB,KAAK,IAAI,aAAa,YAAY,YAAY,QAAQ;CACvF,OAAO,eAAe,KAAK,IAAI,QAAQ,KAAA;AACzC;AAEA,SAAS,WAAW,IAAsC;CACxD,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,OAAO,GAAG,QAAQ,KAAK,MAAM,MAAM,GAAG,CAAC,IAAI,GAAG;AAChD;AAEA,SAAS,iBAAiB,IAAuB,GAA0B;CACzE,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,MAAM,SAAS,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,OAAO,OAAO,KAAK,IAAI,OAAO,SAAS,GAAG,KAAK,IAAI,GAAG,KAAK,KAAK,IAAI,OAAO,MAAM,IAAI,CAAC,CAAC;AACzF;AAEA,SAAS,eAAe,OAAiC;CACvD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK;AAC3D;AAEA,SAAS,WAAW,GAAkB,GAAiC;CACrE,IAAI,MAAM,QAAQ,MAAM,MAAM,OAAO;CACrC,OAAO,IAAI;AACb;AAEA,SAAS,SAAS,KAAoB,QAA+B;CACnE,IAAI,QAAQ,MAAM,OAAO;CACzB,IAAI,UAAU,GAAG,OAAO,OAAO,IAAI,IAAI;CACvC,OAAO,QAAQ,IAAI,KAAK,IAAI,GAAG,GAAG,IAAI,MAAM;AAC9C;AAEA,SAAS,gBACP,SACA,YACe;CACf,MAAM,OAAO,OAAO,SAAS,WAAW,cAAc,IAClD,QAAQ,gBAAgB,OACtB,OACA,QAAQ,WAAW,iBAAiB,KAAK,IAAI,QAAQ,aAAa,KAAK,CAAC,IAC1E;CACJ,MAAM,UAAU,OAAO,SAAS,WAAW,YAAY,IACnD,QAAQ,cAAc,OACpB,OACA,QAAQ,WAAW,eAAe,KAAK,IAAI,QAAQ,WAAW,KAAK,CAAC,IACtE;CACJ,IAAI,SAAS,QAAQ,YAAY,MAAM,OAAO;CAC9C,OAAO,KAAK,IAAI,MAAM,OAAO;AAC/B;AAEA,SAAS,QAAQ,GAAmB;CAClC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;AAEA,SAAS,cACP,QACA,QACA,SACA,QACQ;CACR,MAAM,SAAS,sBAAsB,OAAO,IAAI;CAChD,MAAM,aAAa,aAAa,QAAQ,cAAc,cAAc,QAAQ,WAAW,eAAe,QAAQ,YAAY,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS;CAC9L,IAAI,OAAO,WAAW,GAAG,OAAO,GAAG,OAAO,IAAI;CAC9C,OAAO,GAAG,OAAO,IAAI,WAAW,WAAW,OAAO,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,GAAG;AAC/E;AAEA,SAAS,IAAI,GAA0B;CACrC,IAAI,MAAM,MAAM,OAAO;CACvB,OAAO,EAAE,QAAQ,CAAC;AACpB"}
|
|
1
|
+
{"version":3,"file":"release-confidence-nGDJiiwc.js","names":[],"sources":["../src/promotion-gate.ts","../src/release-confidence.ts"],"sourcesContent":["/**\n * Bootstrap-CI promotion gate.\n *\n * In any iterative-improvement loop (GEPA, prompt evolution, dataset\n * curation), the question is \"did this generation actually improve, or are\n * we celebrating noise?\". With small N and noisy outcomes, point-estimate\n * deltas lie. Bootstrap confidence intervals tell the operator whether the\n * delta is real before code or prompts get promoted.\n *\n * This module is pure functions — no I/O, no model calls. Easy to unit-test\n * and to compose into any verdict gate.\n *\n * Default gate:\n * - Bootstrap mean baseline vs candidate (1k resamples).\n * - Compute the delta distribution; pass if the lower CI bound > 0.\n * - Tunable confidence (default 95%) and resample count.\n *\n * Verdict semantics intentionally match the existing `experiments.jsonl`\n * vocabulary:\n * - ADVANCE: candidate's CI lower bound > baseline mean (real win)\n * - KEEP: overlap, but candidate point estimate >= baseline (neutral)\n * - REVERT: candidate's CI upper bound < baseline mean (real regression)\n * - INCONCLUSIVE: not enough samples or CI straddles zero with no signal\n */\n\nimport { mulberry32 } from './statistics'\n\nexport type Verdict = 'ADVANCE' | 'KEEP' | 'REVERT' | 'INCONCLUSIVE'\n\nexport interface BootstrapResult {\n baselineMean: number\n candidateMean: number\n /** candidateMean - baselineMean, point estimate. */\n delta: number\n /** Lower bound of the (1 - alpha) CI on the delta. */\n ciLower: number\n /** Upper bound of the (1 - alpha) CI on the delta. */\n ciUpper: number\n /** Number of bootstrap resamples used. */\n iterations: number\n alpha: number\n verdict: Verdict\n}\n\nexport interface BootstrapOptions {\n /** Confidence level alpha (default 0.05 → 95% CI). */\n alpha?: number\n /** Number of resamples (default 1000). */\n iterations?: number\n /**\n * Minimum total samples (baseline + candidate) below which we always\n * return INCONCLUSIVE — bootstrap with too few samples is meaningless.\n * Default 6 (combined).\n */\n minTotalSamples?: number\n /** RNG seed for reproducibility. Absent, the seed is derived from the\n * compared arms, so the same inputs reproduce the same verdict. */\n seed?: number\n}\n\n/**\n * Compute the bootstrap CI on (candidateMean - baselineMean) and a verdict.\n *\n * Uses simple percentile bootstrap on the difference of resampled means.\n * That's the standard non-parametric primitive — no distributional\n * assumptions, robust to skew, easy to reason about.\n */\nexport function bootstrapCi(\n baseline: number[],\n candidate: number[],\n options: BootstrapOptions = {},\n): BootstrapResult {\n const alpha = options.alpha ?? 0.05\n const iterations = options.iterations ?? 1000\n const minTotal = options.minTotalSamples ?? 6\n const rng = mulberry32(options.seed ?? hashSeed(baseline, candidate))\n\n const baselineMean = mean(baseline)\n const candidateMean = mean(candidate)\n const delta = candidateMean - baselineMean\n\n if (\n baseline.length + candidate.length < minTotal ||\n baseline.length === 0 ||\n candidate.length === 0\n ) {\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower: -Infinity,\n ciUpper: Infinity,\n iterations: 0,\n alpha,\n verdict: 'INCONCLUSIVE',\n }\n }\n\n const deltas: number[] = new Array(iterations)\n for (let i = 0; i < iterations; i++) {\n const bResample = resample(baseline, rng)\n const cResample = resample(candidate, rng)\n deltas[i] = mean(cResample) - mean(bResample)\n }\n deltas.sort((a, b) => a - b)\n const lowerIdx = Math.floor((alpha / 2) * iterations)\n const upperIdx = Math.floor((1 - alpha / 2) * iterations) - 1\n const ciLower = deltas[Math.max(0, lowerIdx)]!\n const ciUpper = deltas[Math.min(iterations - 1, upperIdx)]!\n\n let verdict: Verdict\n // A ZERO-WIDTH interval is an absence of evidence, not a certainty. Constant\n // arms make every resample identical, so pass/fail data promotes on nothing:\n // [0,0,0] vs [1,1,1] is six samples, clears the `minTotalSamples` floor, and\n // yields [1, 1] — an \"ADVANCE\" carrying no information about how far the\n // estimate could be wrong. Same rule as `pairedDeltaTest`, one estimator over.\n if (!Number.isFinite(ciLower) || !Number.isFinite(ciUpper) || ciLower === ciUpper) {\n verdict = 'INCONCLUSIVE'\n } else if (ciLower > 0) verdict = 'ADVANCE'\n else if (ciUpper < 0) verdict = 'REVERT'\n else if (delta >= 0) verdict = 'KEEP'\n else verdict = 'INCONCLUSIVE'\n\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower,\n ciUpper,\n iterations,\n alpha,\n verdict,\n }\n}\n\nfunction mean(xs: number[]): number {\n if (xs.length === 0) return 0\n let s = 0\n for (const x of xs) s += x\n return s / xs.length\n}\n\nfunction resample(xs: number[], rng: () => number): number[] {\n const out = new Array(xs.length)\n for (let i = 0; i < xs.length; i++) out[i] = xs[Math.floor(rng() * xs.length)]\n return out\n}\n\n/** Stable seed derived from the inputs — same data → same CI bounds. */\nfunction hashSeed(a: number[], b: number[]): number {\n let h = 2166136261\n for (const x of [...a, ...b]) {\n const view = new Float64Array([x])\n const bytes = new Uint8Array(view.buffer)\n for (const byte of bytes) {\n h ^= byte\n h = Math.imul(h, 16777619)\n }\n }\n return h >>> 0\n}\n\n/**\n * Judge-replay promotion gate.\n *\n * The cheap inner-loop judge that drives an evolution run is by definition\n * fast and noisy. When you're about to promote a winning variant to the\n * canonical default, you want a STRONGER judge (a more expensive model, a\n * human grader, a separately-trained reward model) to confirm the win\n * generalises beyond the inner loop.\n *\n * This helper takes raw winner + baseline outputs, scores both through the\n * stronger judge, and applies `bootstrapCi`. ADVANCE means the stronger\n * judge agrees the winner is real with the configured confidence. Doesn't\n * matter what shape your \"output\" is — pass a string, an object, anything\n * the judge can read.\n */\nexport interface JudgeReplayGateArgs<TOutput> {\n baselineOutputs: TOutput[]\n candidateOutputs: TOutput[]\n /** Stronger judge — async to allow LLM calls. Return a 0..N scalar score. */\n judge: (output: TOutput) => Promise<number> | number\n alpha?: number\n iterations?: number\n /** RNG seed for reproducibility. */\n seed?: number\n /** Maximum concurrent judge calls. Default 4. */\n judgeConcurrency?: number\n}\n\n/**\n * Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.\n */\nexport async function judgeReplayGate<TOutput>(\n args: JudgeReplayGateArgs<TOutput>,\n): Promise<BootstrapResult & { baselineSamples: number; candidateSamples: number }> {\n const concurrency = args.judgeConcurrency ?? 4\n const baselineScores = await scoreAll(args.baselineOutputs, args.judge, concurrency)\n const candidateScores = await scoreAll(args.candidateOutputs, args.judge, concurrency)\n const ci = bootstrapCi(baselineScores, candidateScores, {\n ...(args.alpha !== undefined ? { alpha: args.alpha } : {}),\n ...(args.iterations !== undefined ? { iterations: args.iterations } : {}),\n ...(args.seed !== undefined ? { seed: args.seed } : {}),\n })\n return {\n ...ci,\n baselineSamples: baselineScores.length,\n candidateSamples: candidateScores.length,\n }\n}\n\nasync function scoreAll<TOutput>(\n outputs: TOutput[],\n judge: (output: TOutput) => Promise<number> | number,\n concurrency: number,\n): Promise<number[]> {\n const results: number[] = new Array(outputs.length)\n let next = 0\n async function worker(): Promise<void> {\n while (true) {\n const i = next++\n if (i >= outputs.length) return\n const v = await judge(outputs[i]!)\n results[i] = Number.isFinite(v) ? v : 0\n }\n }\n await Promise.all(Array.from({ length: Math.max(1, concurrency) }, () => worker()))\n return results\n}\n","/**\n * Release confidence gate.\n *\n * This is the production-facing composition layer over the lower-level\n * primitives:\n * - Dataset manifests prove corpus/version coverage.\n * - RunRecord rows prove reproducible search/holdout outcomes.\n * - Multi-shot trace evidence carries turn counts and ASI diagnostics.\n * - HeldOutGate decisions remain the paired promotion authority.\n *\n * The gate is intentionally pure and conservative. Missing declared evidence\n * fails closed instead of being treated as a neutral zero.\n */\n\nimport type { DatasetManifest, DatasetScenario, DatasetSplit } from './dataset'\nimport { ValidationError, VerificationError } from './errors'\nimport type { GateDecision } from './held-out-gate'\nimport { isRealnessGated, observedSplitScore } from './rollout/reward'\nimport { type RunRecord, type RunSplitTag, validateRunRecord } from './run-record'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Severity of an actionable finding attached to a run/trace. */\nexport type AsiSeverity = 'info' | 'warning' | 'error' | 'critical'\n\n/** Actionable side-info — a diagnosed finding the loop can act on. */\nexport interface ActionableSideInfo {\n /** Stable expectation/check id when available. */\n expectationId?: string\n /** Human-readable diagnosis of what happened. */\n message: string\n severity?: AsiSeverity\n /** Concrete trace excerpt, file path, tool call, screenshot id, etc. */\n evidence?: string\n /** Prompt/tool/context surface likely responsible. */\n responsibleSurface?: string\n /** Suggested fix in natural language. */\n suggestion?: string\n /** Whether this expectation was satisfied. Defaults to false for ASI rows. */\n matched?: boolean\n metadata?: Record<string, unknown>\n}\n\nexport type ReleaseConfidenceStatus = 'pass' | 'warn' | 'fail'\nexport type ReleaseConfidenceAxisName =\n | 'corpus'\n | 'quality'\n | 'reliability'\n | 'generalization'\n | 'diagnostics'\n | 'efficiency'\n\nexport interface ReleaseTraceEvidence {\n scenarioId: string\n candidateId?: string\n split?: RunSplitTag\n score?: number\n ok?: boolean\n turnCount?: number\n costUsd?: number\n durationMs?: number\n /** Canonical task-failure class. Free-form detail belongs in ASI. */\n failureClass?: FailureClass\n asi?: ActionableSideInfo[]\n metadata?: Record<string, unknown>\n}\n\nexport interface ReleaseConfidenceThresholds {\n /** Require a Dataset manifest or explicit scenarios. Default true. */\n requireCorpus?: boolean\n minScenarioCount?: number\n minSearchRuns?: number\n minHoldoutRuns?: number\n /** Require at least one holdout scenario/run. Default true. */\n requireHoldout?: boolean\n minPassRate?: number\n minMeanScore?: number\n /** Search mean may exceed holdout mean by at most this much. */\n maxOverfitGap?: number\n maxMeanCostUsd?: number\n maxP95WallMs?: number\n /** Low-score/failed rows must carry ASI. Default true. */\n requireAsiForFailures?: boolean\n /** Score below this is considered a failure for ASI coverage. Default 0.5. */\n failureScoreThreshold?: number\n}\n\nexport interface ReleaseConfidenceInput {\n target: string\n candidateId?: string\n baselineId?: string\n dataset?: DatasetManifest\n scenarios?: readonly DatasetScenario[]\n runs?: readonly RunRecord[]\n traces?: readonly ReleaseTraceEvidence[]\n gateDecision?: GateDecision | null\n thresholds?: ReleaseConfidenceThresholds\n}\n\nexport interface ReleaseConfidenceAxis {\n name: ReleaseConfidenceAxisName\n status: ReleaseConfidenceStatus\n score: number | null\n detail: string\n}\n\nexport interface ReleaseConfidenceIssue {\n axis: ReleaseConfidenceAxisName\n severity: 'critical' | 'warning'\n code: string\n detail: string\n}\n\nexport interface ReleaseConfidenceMetrics {\n scenarioCount: number\n /** Search rows with a finite search score. */\n searchRuns: number\n /** Holdout rows with a finite holdout score. */\n holdoutRuns: number\n /** Runs with neither a split-matched score nor an explicit task failure. */\n unscoredRuns: number\n /** Run rows, or trace rows when no runs exist, with no classified terminal result. */\n unclassifiedTerminalRuns: number\n /** Run rows, or trace rows when no runs exist, that ended unsuccessfully. */\n terminalFailureRuns: number\n /** Success fraction when every run or fallback trace row has a classified result. */\n reliabilityRate: number | null\n passRate: number | null\n meanScore: number | null\n searchMeanScore: number | null\n holdoutMeanScore: number | null\n overfitGap: number | null\n meanCostUsd: number | null\n p95WallMs: number | null\n failedRows: number\n failuresWithAsi: number\n singleShotTraces: number\n multiShotTraces: number\n splitCounts: Record<DatasetSplit, number>\n domainCounts: Record<string, number>\n failureClassCounts: Partial<Record<FailureClass, number>>\n responsibleSurfaceCounts: Record<string, number>\n /**\n * Runs excluded from `passRate` because the authenticity gate flagged them as\n * gamed. Surfaced, never silent: a release whose pass rate is computed over a\n * shrunken denominator has to say by how much, or the exclusion is just a\n * different way of hiding the same runs.\n */\n realnessGatedRuns: number\n}\n\nexport interface ReleaseConfidenceScorecard {\n target: string\n candidateId: string | null\n baselineId: string | null\n status: ReleaseConfidenceStatus\n promote: boolean\n axes: ReleaseConfidenceAxis[]\n issues: ReleaseConfidenceIssue[]\n metrics: ReleaseConfidenceMetrics\n dataset: DatasetManifest | null\n gateDecision: GateDecision | null\n summary: string\n}\n\nconst DEFAULT_THRESHOLDS: Required<ReleaseConfidenceThresholds> = {\n requireCorpus: true,\n minScenarioCount: 1,\n minSearchRuns: 1,\n minHoldoutRuns: 1,\n requireHoldout: true,\n minPassRate: 0.8,\n minMeanScore: 0.7,\n maxOverfitGap: 0.15,\n maxMeanCostUsd: Number.POSITIVE_INFINITY,\n maxP95WallMs: Number.POSITIVE_INFINITY,\n requireAsiForFailures: true,\n failureScoreThreshold: 0.5,\n}\n\nexport function evaluateReleaseConfidence(\n input: ReleaseConfidenceInput,\n): ReleaseConfidenceScorecard {\n const thresholds = { ...DEFAULT_THRESHOLDS, ...input.thresholds }\n const candidateId = input.candidateId ?? null\n const runs = filterCandidate(\n (input.runs ?? []).map(validateRunRecord),\n candidateId,\n input.baselineId,\n )\n const traces = filterTraceCandidate(\n (input.traces ?? []).map(validateReleaseTraceEvidence),\n candidateId,\n input.baselineId,\n )\n const scenarios = input.scenarios ?? []\n const scenarioCount = input.dataset?.scenarioCount ?? scenarios.length\n const splitCounts = input.dataset?.splitCounts ?? countScenarioSplits(scenarios)\n const searchScores = scoresFor(runs, 'search')\n const holdoutScores = scoresFor(runs, 'holdout')\n const runScores = runs.map(runSplitScore).filter(isFiniteNumber)\n const traceScores = traces.map((t) => t.score).filter(isFiniteNumber)\n const scoreUniverse = runs.length > 0 ? runScores : traceScores\n const qualityRuns = runs.filter((run) => runSplitScore(run) !== undefined)\n const unscoredRuns = runs.filter(\n (run) =>\n runSplitScore(run) === undefined &&\n !hasExplicitTaskFailure(run) &&\n !isFailedTerminalOutcome(run.terminalOutcome),\n ).length\n // Realness-gated runs are EXCLUDED from the pass rate — numerator AND\n // denominator — and the count of what was dropped ships beside the rate as\n // `metrics.realnessGatedRuns`. Counting a gamed run as a pass made faking a\n // success the cheapest way to improve a release scorecard; scoring it 0\n // instead would be the other error, silently deflating the rate with a run\n // the gate says carries no usable verdict at all.\n const honestRuns = runs.filter((run) => !isRealnessGated(run))\n const passOutcomes =\n runs.length > 0\n ? honestRuns.map((run) => runPassOutcome(run, thresholds.failureScoreThreshold))\n : traces.map((trace) => tracePassOutcome(trace, thresholds.failureScoreThreshold))\n const reliabilityRows =\n runs.length > 0\n ? runs.map((run) => terminalSuccess(run.terminalOutcome))\n : traces.map((trace) => trace.ok)\n const unclassifiedTerminalRuns = reliabilityRows.filter((outcome) => outcome === undefined).length\n const terminalFailureRuns = reliabilityRows.filter((outcome) => outcome === false).length\n const reliabilityRate =\n reliabilityRows.length === 0 || unclassifiedTerminalRuns > 0\n ? null\n : (reliabilityRows.length - terminalFailureRuns) / reliabilityRows.length\n const searchRuns = qualityRuns.filter((r) => r.splitTag === 'search').length\n const holdoutRuns = qualityRuns.filter((r) => r.splitTag === 'holdout').length\n const failed = failedRows(runs, traces, thresholds.failureScoreThreshold)\n const searchMeanScore = meanOrNull(searchScores)\n const holdoutMeanScore = meanOrNull(holdoutScores)\n const runCosts = runs.flatMap((run) =>\n run.costProvenance.kind === 'uncaptured' ? [] : [run.costProvenance.usd],\n )\n const traceCosts = traces.map((trace) => trace.costUsd).filter(isFiniteNumber)\n const meanCostUsd =\n runs.length > 0\n ? runCosts.length === runs.length\n ? meanOrNull(runCosts)\n : null\n : meanOrNull(traceCosts)\n const wallTimes =\n runs.length > 0\n ? runs.map((run) => run.wallMs)\n : traces.map((trace) => trace.durationMs).filter(isFiniteNumber)\n const metrics: ReleaseConfidenceMetrics = {\n scenarioCount,\n searchRuns,\n holdoutRuns,\n unscoredRuns,\n unclassifiedTerminalRuns,\n terminalFailureRuns,\n reliabilityRate,\n passRate: passOutcomeRate(passOutcomes),\n realnessGatedRuns: runs.length - honestRuns.length,\n meanScore: meanOrNull(scoreUniverse),\n searchMeanScore,\n holdoutMeanScore,\n overfitGap: diffOrNull(searchMeanScore, holdoutMeanScore),\n meanCostUsd,\n p95WallMs: percentileOrNull(wallTimes, 0.95),\n failedRows: failed.length,\n failuresWithAsi: failed.filter((row) => row.hasAsi).length,\n singleShotTraces: traces.filter((t) => t.turnCount === 1).length,\n multiShotTraces: traces.filter((t) => (t.turnCount ?? 0) > 1).length,\n splitCounts,\n domainCounts: countDomains(scenarios),\n failureClassCounts: countFailureClasses(runs, traces, thresholds.failureScoreThreshold),\n responsibleSurfaceCounts: countResponsibleSurfaces(traces),\n }\n\n const issues: ReleaseConfidenceIssue[] = []\n checkCorpus(input, thresholds, metrics, issues)\n checkQuality(thresholds, metrics, issues)\n checkReliability(metrics, issues)\n checkGeneralization(input.gateDecision ?? null, thresholds, metrics, issues)\n checkDiagnostics(thresholds, metrics, issues)\n checkEfficiency(thresholds, metrics, issues)\n\n const axes = buildAxes(metrics, thresholds, issues)\n const status = issues.some((i) => i.severity === 'critical')\n ? 'fail'\n : issues.length > 0\n ? 'warn'\n : 'pass'\n\n return {\n target: input.target,\n candidateId,\n baselineId: input.baselineId ?? null,\n status,\n promote: status === 'pass' && (input.gateDecision ? input.gateDecision.promote : true),\n axes,\n issues,\n metrics,\n dataset: input.dataset ?? null,\n gateDecision: input.gateDecision ?? null,\n summary: renderSummary(input.target, status, metrics, issues),\n }\n}\n\nexport function assertReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard {\n const scorecard = evaluateReleaseConfidence(input)\n if (scorecard.status === 'fail') {\n throw new VerificationError(scorecard.summary)\n }\n return scorecard\n}\n\nfunction filterCandidate(\n runs: readonly RunRecord[],\n candidateId: string | null,\n baselineId?: string,\n): RunRecord[] {\n if (candidateId) return runs.filter((r) => r.candidateId === candidateId)\n if (baselineId) return runs.filter((r) => r.candidateId !== baselineId)\n return [...runs]\n}\n\nfunction filterTraceCandidate(\n traces: readonly ReleaseTraceEvidence[],\n candidateId: string | null,\n baselineId?: string,\n): ReleaseTraceEvidence[] {\n if (candidateId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId === candidateId)\n if (baselineId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId !== baselineId)\n return [...traces]\n}\n\nfunction validateReleaseTraceEvidence(\n trace: ReleaseTraceEvidence,\n index: number,\n): ReleaseTraceEvidence {\n const value = trace as unknown as Record<string, unknown>\n if (Object.hasOwn(value, 'failureMode')) {\n throw new ValidationError(\n `traces[${index}].failureMode is not supported; use canonical failureClass`,\n )\n }\n if (trace.failureClass !== undefined && !FAILURE_CLASSES.includes(trace.failureClass)) {\n throw new ValidationError(\n `traces[${index}].failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n )\n }\n return trace\n}\n\nfunction checkCorpus(\n input: ReleaseConfidenceInput,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireCorpus && !input.dataset && (input.scenarios?.length ?? 0) === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_corpus',\n detail: 'No Dataset manifest or scenarios supplied.',\n })\n }\n if (metrics.scenarioCount < thresholds.minScenarioCount) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'few_scenarios',\n detail: `${metrics.scenarioCount} scenario(s) < min ${thresholds.minScenarioCount}.`,\n })\n }\n if (thresholds.requireHoldout && metrics.splitCounts.holdout === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_holdout_split',\n detail: 'Corpus has no holdout scenarios.',\n })\n }\n}\n\nfunction checkQuality(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.searchRuns < thresholds.minSearchRuns) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'few_search_runs',\n detail: `${metrics.searchRuns} search run(s) < min ${thresholds.minSearchRuns}.`,\n })\n }\n if (metrics.unscoredRuns > 0) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'unscored_runs',\n detail: `${metrics.unscoredRuns} supplied run(s) have no task result.`,\n })\n }\n if (metrics.passRate === null || metrics.meanScore === null) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'missing_quality_scores',\n detail: 'No task-quality scores are available for pass-rate and mean-score checks.',\n })\n }\n if (metrics.passRate !== null && metrics.passRate < thresholds.minPassRate) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_pass_rate',\n detail: `passRate ${fmt(metrics.passRate)} < ${fmt(thresholds.minPassRate)}.`,\n })\n }\n if (metrics.meanScore !== null && metrics.meanScore < thresholds.minMeanScore) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_mean_score',\n detail: `meanScore ${fmt(metrics.meanScore)} < ${fmt(thresholds.minMeanScore)}.`,\n })\n }\n}\n\nfunction checkReliability(\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.reliabilityRate === null) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'missing_reliability_evidence',\n detail:\n metrics.unclassifiedTerminalRuns > 0\n ? `${metrics.unclassifiedTerminalRuns} supplied run(s) have no classified terminal result.`\n : 'No classified terminal results are available.',\n })\n }\n if (metrics.terminalFailureRuns > 0) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'terminal_run_failures',\n detail: `${metrics.terminalFailureRuns} run(s) ended failed, cancelled, or incomplete.`,\n })\n }\n}\n\nfunction checkGeneralization(\n gateDecision: GateDecision | null,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireHoldout && metrics.holdoutRuns < thresholds.minHoldoutRuns) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'few_holdout_runs',\n detail: `${metrics.holdoutRuns} holdout run(s) < min ${thresholds.minHoldoutRuns}.`,\n })\n }\n if (metrics.overfitGap !== null && metrics.overfitGap > thresholds.maxOverfitGap) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'overfit_gap',\n detail: `search-holdout gap ${fmt(metrics.overfitGap)} > ${fmt(thresholds.maxOverfitGap)}.`,\n })\n }\n if (gateDecision && !gateDecision.promote) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: `gate_${gateDecision.rejectionCode ?? 'reject'}`,\n detail: gateDecision.reason,\n })\n }\n}\n\nfunction checkDiagnostics(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (!thresholds.requireAsiForFailures) return\n if (metrics.failedRows > metrics.failuresWithAsi) {\n issues.push({\n axis: 'diagnostics',\n severity: 'critical',\n code: 'missing_failure_asi',\n detail: `${metrics.failedRows - metrics.failuresWithAsi} failed row(s) have no actionable side information.`,\n })\n }\n}\n\nfunction checkEfficiency(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (Number.isFinite(thresholds.maxMeanCostUsd) && metrics.meanCostUsd === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_cost',\n detail: 'A finite cost limit was configured but no cost evidence is available.',\n })\n } else if (metrics.meanCostUsd !== null && metrics.meanCostUsd > thresholds.maxMeanCostUsd) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'cost_budget',\n detail: `meanCostUsd ${fmt(metrics.meanCostUsd)} > ${fmt(thresholds.maxMeanCostUsd)}.`,\n })\n }\n if (Number.isFinite(thresholds.maxP95WallMs) && metrics.p95WallMs === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_latency',\n detail: 'A finite latency limit was configured but no latency evidence is available.',\n })\n } else if (metrics.p95WallMs !== null && metrics.p95WallMs > thresholds.maxP95WallMs) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'latency_budget',\n detail: `p95WallMs ${fmt(metrics.p95WallMs)} > ${fmt(thresholds.maxP95WallMs)}.`,\n })\n }\n}\n\nfunction buildAxes(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n issues: ReleaseConfidenceIssue[],\n): ReleaseConfidenceAxis[] {\n return [\n axis(\n 'corpus',\n issues,\n bounded(metrics.scenarioCount / Math.max(1, thresholds.minScenarioCount)),\n `${metrics.scenarioCount} scenarios; holdout=${metrics.splitCounts.holdout}`,\n ),\n axis(\n 'quality',\n issues,\n metrics.passRate === null || metrics.meanScore === null\n ? null\n : Math.min(metrics.passRate, metrics.meanScore),\n `passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`,\n ),\n axis(\n 'reliability',\n issues,\n metrics.reliabilityRate,\n `successRate=${fmt(metrics.reliabilityRate)} terminalFailures=${metrics.terminalFailureRuns} unclassified=${metrics.unclassifiedTerminalRuns}`,\n ),\n axis(\n 'generalization',\n issues,\n gapScore(metrics.overfitGap, thresholds.maxOverfitGap),\n `holdoutRuns=${metrics.holdoutRuns} overfitGap=${fmt(metrics.overfitGap)}`,\n ),\n axis(\n 'diagnostics',\n issues,\n metrics.failedRows === 0 ? 1 : metrics.failuresWithAsi / metrics.failedRows,\n `failuresWithAsi=${metrics.failuresWithAsi}/${metrics.failedRows}`,\n ),\n axis(\n 'efficiency',\n issues,\n efficiencyScore(metrics, thresholds),\n `meanCostUsd=${fmt(metrics.meanCostUsd)} p95WallMs=${fmt(metrics.p95WallMs)}`,\n ),\n ]\n}\n\nfunction axis(\n name: ReleaseConfidenceAxisName,\n issues: ReleaseConfidenceIssue[],\n score: number | null,\n detail: string,\n): ReleaseConfidenceAxis {\n const own = issues.filter((i) => i.axis === name)\n const status = own.some((i) => i.severity === 'critical')\n ? 'fail'\n : own.length > 0\n ? 'warn'\n : 'pass'\n return { name, status, score: score === null ? null : bounded(score), detail }\n}\n\nfunction countScenarioSplits(scenarios: readonly DatasetScenario[]): Record<DatasetSplit, number> {\n const counts: Record<DatasetSplit, number> = { train: 0, dev: 0, test: 0, holdout: 0 }\n for (const scenario of scenarios) counts[scenario.split ?? 'train']++\n return counts\n}\n\nfunction countDomains(scenarios: readonly DatasetScenario[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const scenario of scenarios) {\n const domain = scenario.tags?.domain ?? scenario.tags?.category ?? 'uncategorized'\n out[domain] = (out[domain] ?? 0) + 1\n }\n return out\n}\n\nfunction countFailureClasses(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Partial<Record<FailureClass, number>> {\n const out: Partial<Record<FailureClass, number>> = {}\n for (const run of runs) {\n // Ungated: a failure-mode census counts what the runs REPORTED, which is\n // the only way a gamed run's inflated score is visible at all. It is not a\n // promotion number — `passRate` is, and that one excludes gated runs and\n // publishes the excluded count as `metrics.realnessGatedRuns`.\n if (runPassOutcome(run, threshold) === false) {\n const failureClass =\n run.failureClass !== undefined && run.failureClass !== 'success'\n ? run.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n const failureClass =\n trace.failureClass !== undefined && trace.failureClass !== 'success'\n ? trace.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction countResponsibleSurfaces(traces: readonly ReleaseTraceEvidence[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const trace of traces) {\n for (const asi of trace.asi ?? []) {\n const surface = asi.responsibleSurface ?? 'unknown'\n out[surface] = (out[surface] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction failedRows(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Array<{ hasAsi: boolean }> {\n const out: Array<{ hasAsi: boolean }> = []\n for (const run of runs) {\n if (runPassOutcome(run, threshold) === false) {\n const asiMetric = run.outcome.raw.asi\n out.push({ hasAsi: typeof asiMetric === 'number' && asiMetric > 0 })\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n out.push({ hasAsi: (trace.asi?.length ?? 0) > 0 })\n }\n }\n return out\n}\n\nfunction passOutcomeRate(outcomes: readonly (boolean | null)[]): number | null {\n const classified = outcomes.filter((outcome): outcome is boolean => outcome !== null)\n if (classified.length === 0) return null\n return classified.filter(Boolean).length / classified.length\n}\n\nfunction runPassOutcome(run: RunRecord, threshold: number): boolean | null {\n if (hasExplicitTaskFailure(run)) return false\n const score = runSplitScore(run)\n return score === undefined ? null : score >= threshold\n}\n\nfunction hasExplicitTaskFailure(run: RunRecord): boolean {\n return run.failureClass !== undefined && run.failureClass !== 'success'\n}\n\nfunction tracePassOutcome(trace: ReleaseTraceEvidence, threshold: number): boolean | null {\n if (trace.failureClass !== undefined && trace.failureClass !== 'success') return false\n if (trace.ok === false) return false\n if (isFiniteNumber(trace.score)) return trace.score >= threshold\n return trace.ok === true ? true : null\n}\n\nfunction isFailedTerminalOutcome(\n outcome: RunRecord['terminalOutcome'],\n): outcome is 'failed' | 'cancelled' | 'incomplete' {\n return outcome === 'failed' || outcome === 'cancelled' || outcome === 'incomplete'\n}\n\nfunction terminalSuccess(outcome: RunRecord['terminalOutcome']): boolean | undefined {\n if (outcome === 'succeeded') return true\n if (isFailedTerminalOutcome(outcome)) return false\n return undefined\n}\n\nfunction scoresFor(runs: readonly RunRecord[], split: RunSplitTag): number[] {\n return runs\n .filter((run) => run.splitTag === split)\n .map(runSplitScore)\n .filter(isFiniteNumber)\n}\n\n/**\n * RAW and split-exact (`observedSplitScore`): this feeds the per-split means,\n * the overfit gap, and the pass threshold — descriptions of what the runs\n * reported. A gamed run inflating them is visible next to `realnessGatedRuns`,\n * and the promotion number (`passRate`) excludes gated runs entirely.\n */\nfunction runSplitScore(run: RunRecord): number | undefined {\n const score = observedSplitScore(run, run.splitTag === 'holdout' ? 'holdout' : 'search')\n return isFiniteNumber(score) ? score : undefined\n}\n\nfunction meanOrNull(xs: readonly number[]): number | null {\n if (xs.length === 0) return null\n return xs.reduce((sum, x) => sum + x, 0) / xs.length\n}\n\nfunction percentileOrNull(xs: readonly number[], p: number): number | null {\n if (xs.length === 0) return null\n const sorted = [...xs].sort((a, b) => a - b)\n return sorted[Math.min(sorted.length - 1, Math.max(0, Math.ceil(p * sorted.length) - 1))]!\n}\n\nfunction isFiniteNumber(value: unknown): value is number {\n return typeof value === 'number' && Number.isFinite(value)\n}\n\nfunction diffOrNull(a: number | null, b: number | null): number | null {\n if (a === null || b === null) return null\n return a - b\n}\n\nfunction gapScore(gap: number | null, maxGap: number): number | null {\n if (gap === null) return null\n if (maxGap <= 0) return gap <= 0 ? 1 : 0\n return bounded(1 - Math.max(0, gap) / maxGap)\n}\n\nfunction efficiencyScore(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n): number | null {\n const cost = Number.isFinite(thresholds.maxMeanCostUsd)\n ? metrics.meanCostUsd === null\n ? null\n : bounded(thresholds.maxMeanCostUsd / Math.max(metrics.meanCostUsd, 1e-12))\n : 1\n const latency = Number.isFinite(thresholds.maxP95WallMs)\n ? metrics.p95WallMs === null\n ? null\n : bounded(thresholds.maxP95WallMs / Math.max(metrics.p95WallMs, 1e-12))\n : 1\n if (cost === null || latency === null) return null\n return Math.min(cost, latency)\n}\n\nfunction bounded(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n\nfunction renderSummary(\n target: string,\n status: ReleaseConfidenceStatus,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): string {\n const prefix = `release confidence ${status}: ${target}`\n const metricText = `scenarios=${metrics.scenarioCount} searchRuns=${metrics.searchRuns} holdoutRuns=${metrics.holdoutRuns} passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`\n if (issues.length === 0) return `${prefix}; ${metricText}`\n return `${prefix}; ${metricText}; issues=${issues.map((i) => i.code).join(',')}`\n}\n\nfunction fmt(x: number | null): string {\n if (x === null) return 'n/a'\n return x.toFixed(4)\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmEA,SAAgB,YACd,UACA,WACA,UAA4B,CAAC,GACZ;CACjB,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,aAAa,QAAQ,cAAc;CACzC,MAAM,WAAW,QAAQ,mBAAmB;CAC5C,MAAM,MAAM,WAAW,QAAQ,QAAQ,SAAS,UAAU,SAAS,CAAC;CAEpE,MAAM,eAAe,KAAK,QAAQ;CAClC,MAAM,gBAAgB,KAAK,SAAS;CACpC,MAAM,QAAQ,gBAAgB;CAE9B,IACE,SAAS,SAAS,UAAU,SAAS,YACrC,SAAS,WAAW,KACpB,UAAU,WAAW,GAErB,OAAO;EACL;EACA;EACA;EACA,SAAS;EACT,SAAS;EACT,YAAY;EACZ;EACA,SAAS;CACX;CAGF,MAAM,SAAmB,IAAI,MAAM,UAAU;CAC7C,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,YAAY,SAAS,UAAU,GAAG;EAExC,OAAO,KAAK,KADM,SAAS,WAAW,GACb,CAAC,IAAI,KAAK,SAAS;CAC9C;CACA,OAAO,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3B,MAAM,WAAW,KAAK,MAAO,QAAQ,IAAK,UAAU;CACpD,MAAM,WAAW,KAAK,OAAO,IAAI,QAAQ,KAAK,UAAU,IAAI;CAC5D,MAAM,UAAU,OAAO,KAAK,IAAI,GAAG,QAAQ;CAC3C,MAAM,UAAU,OAAO,KAAK,IAAI,aAAa,GAAG,QAAQ;CAExD,IAAI;CAMJ,IAAI,CAAC,OAAO,SAAS,OAAO,KAAK,CAAC,OAAO,SAAS,OAAO,KAAK,YAAY,SACxE,UAAU;MACL,IAAI,UAAU,GAAG,UAAU;MAC7B,IAAI,UAAU,GAAG,UAAU;MAC3B,IAAI,SAAS,GAAG,UAAU;MAC1B,UAAU;CAEf,OAAO;EACL;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,IAAI,KAAK;CACzB,OAAO,IAAI,GAAG;AAChB;AAEA,SAAS,SAAS,IAAc,KAA6B;CAC3D,MAAM,MAAM,IAAI,MAAM,GAAG,MAAM;CAC/B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,QAAQ,KAAK,IAAI,KAAK,GAAG,KAAK,MAAM,IAAI,IAAI,GAAG,MAAM;CAC5E,OAAO;AACT;;AAGA,SAAS,SAAS,GAAa,GAAqB;CAClD,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,CAAC,GAAG,GAAG,GAAG,CAAC,GAAG;EAC5B,MAAM,OAAO,IAAI,aAAa,CAAC,CAAC,CAAC;EACjC,MAAM,QAAQ,IAAI,WAAW,KAAK,MAAM;EACxC,KAAK,MAAM,QAAQ,OAAO;GACxB,KAAK;GACL,IAAI,KAAK,KAAK,GAAG,QAAQ;EAC3B;CACF;CACA,OAAO,MAAM;AACf;;;;AAiCA,eAAsB,gBACpB,MACkF;CAClF,MAAM,cAAc,KAAK,oBAAoB;CAC7C,MAAM,iBAAiB,MAAM,SAAS,KAAK,iBAAiB,KAAK,OAAO,WAAW;CACnF,MAAM,kBAAkB,MAAM,SAAS,KAAK,kBAAkB,KAAK,OAAO,WAAW;CAMrF,OAAO;EACL,GANS,YAAY,gBAAgB,iBAAiB;GACtD,GAAI,KAAK,UAAU,KAAA,IAAY,EAAE,OAAO,KAAK,MAAM,IAAI,CAAC;GACxD,GAAI,KAAK,eAAe,KAAA,IAAY,EAAE,YAAY,KAAK,WAAW,IAAI,CAAC;GACvE,GAAI,KAAK,SAAS,KAAA,IAAY,EAAE,MAAM,KAAK,KAAK,IAAI,CAAC;EACvD,CAEM;EACJ,iBAAiB,eAAe;EAChC,kBAAkB,gBAAgB;CACpC;AACF;AAEA,eAAe,SACb,SACA,OACA,aACmB;CACnB,MAAM,UAAoB,IAAI,MAAM,QAAQ,MAAM;CAClD,IAAI,OAAO;CACX,eAAe,SAAwB;EACrC,OAAO,MAAM;GACX,MAAM,IAAI;GACV,IAAI,KAAK,QAAQ,QAAQ;GACzB,MAAM,IAAI,MAAM,MAAM,QAAQ,EAAG;GACjC,QAAQ,KAAK,OAAO,SAAS,CAAC,IAAI,IAAI;EACxC;CACF;CACA,MAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,GAAG,WAAW,EAAE,SAAS,OAAO,CAAC,CAAC;CAClF,OAAO;AACT;;;AChEA,MAAM,qBAA4D;CAChE,eAAe;CACf,kBAAkB;CAClB,eAAe;CACf,gBAAgB;CAChB,gBAAgB;CAChB,aAAa;CACb,cAAc;CACd,eAAe;CACf,gBAAgB,OAAO;CACvB,cAAc,OAAO;CACrB,uBAAuB;CACvB,uBAAuB;AACzB;AAEA,SAAgB,0BACd,OAC4B;CAC5B,MAAM,aAAa;EAAE,GAAG;EAAoB,GAAG,MAAM;CAAW;CAChE,MAAM,cAAc,MAAM,eAAe;CACzC,MAAM,OAAO,iBACV,MAAM,QAAQ,CAAC,EAAA,CAAG,IAAI,iBAAiB,GACxC,aACA,MAAM,UACR;CACA,MAAM,SAAS,sBACZ,MAAM,UAAU,CAAC,EAAA,CAAG,IAAI,4BAA4B,GACrD,aACA,MAAM,UACR;CACA,MAAM,YAAY,MAAM,aAAa,CAAC;CACtC,MAAM,gBAAgB,MAAM,SAAS,iBAAiB,UAAU;CAChE,MAAM,cAAc,MAAM,SAAS,eAAe,oBAAoB,SAAS;CAC/E,MAAM,eAAe,UAAU,MAAM,QAAQ;CAC7C,MAAM,gBAAgB,UAAU,MAAM,SAAS;CAC/C,MAAM,YAAY,KAAK,IAAI,aAAa,CAAC,CAAC,OAAO,cAAc;CAC/D,MAAM,cAAc,OAAO,KAAK,MAAM,EAAE,KAAK,CAAC,CAAC,OAAO,cAAc;CACpE,MAAM,gBAAgB,KAAK,SAAS,IAAI,YAAY;CACpD,MAAM,cAAc,KAAK,QAAQ,QAAQ,cAAc,GAAG,MAAM,KAAA,CAAS;CACzE,MAAM,eAAe,KAAK,QACvB,QACC,cAAc,GAAG,MAAM,KAAA,KACvB,CAAC,uBAAuB,GAAG,KAC3B,CAAC,wBAAwB,IAAI,eAAe,CAChD,CAAC,CAAC;CAOF,MAAM,aAAa,KAAK,QAAQ,QAAQ,CAAC,gBAAgB,GAAG,CAAC;CAC7D,MAAM,eACJ,KAAK,SAAS,IACV,WAAW,KAAK,QAAQ,eAAe,KAAK,WAAW,qBAAqB,CAAC,IAC7E,OAAO,KAAK,UAAU,iBAAiB,OAAO,WAAW,qBAAqB,CAAC;CACrF,MAAM,kBACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,gBAAgB,IAAI,eAAe,CAAC,IACtD,OAAO,KAAK,UAAU,MAAM,EAAE;CACpC,MAAM,2BAA2B,gBAAgB,QAAQ,YAAY,YAAY,KAAA,CAAS,CAAC,CAAC;CAC5F,MAAM,sBAAsB,gBAAgB,QAAQ,YAAY,YAAY,KAAK,CAAC,CAAC;CACnF,MAAM,kBACJ,gBAAgB,WAAW,KAAK,2BAA2B,IACvD,QACC,gBAAgB,SAAS,uBAAuB,gBAAgB;CACvE,MAAM,aAAa,YAAY,QAAQ,MAAM,EAAE,aAAa,QAAQ,CAAC,CAAC;CACtE,MAAM,cAAc,YAAY,QAAQ,MAAM,EAAE,aAAa,SAAS,CAAC,CAAC;CACxE,MAAM,SAAS,WAAW,MAAM,QAAQ,WAAW,qBAAqB;CACxE,MAAM,kBAAkB,WAAW,YAAY;CAC/C,MAAM,mBAAmB,WAAW,aAAa;CACjD,MAAM,WAAW,KAAK,SAAS,QAC7B,IAAI,eAAe,SAAS,eAAe,CAAC,IAAI,CAAC,IAAI,eAAe,GAAG,CACzE;CACA,MAAM,aAAa,OAAO,KAAK,UAAU,MAAM,OAAO,CAAC,CAAC,OAAO,cAAc;CAC7E,MAAM,cACJ,KAAK,SAAS,IACV,SAAS,WAAW,KAAK,SACvB,WAAW,QAAQ,IACnB,OACF,WAAW,UAAU;CAC3B,MAAM,YACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,IAAI,MAAM,IAC5B,OAAO,KAAK,UAAU,MAAM,UAAU,CAAC,CAAC,OAAO,cAAc;CACnE,MAAM,UAAoC;EACxC;EACA;EACA;EACA;EACA;EACA;EACA;EACA,UAAU,gBAAgB,YAAY;EACtC,mBAAmB,KAAK,SAAS,WAAW;EAC5C,WAAW,WAAW,aAAa;EACnC;EACA;EACA,YAAY,WAAW,iBAAiB,gBAAgB;EACxD;EACA,WAAW,iBAAiB,WAAW,GAAI;EAC3C,YAAY,OAAO;EACnB,iBAAiB,OAAO,QAAQ,QAAQ,IAAI,MAAM,CAAC,CAAC;EACpD,kBAAkB,OAAO,QAAQ,MAAM,EAAE,cAAc,CAAC,CAAC,CAAC;EAC1D,iBAAiB,OAAO,QAAQ,OAAO,EAAE,aAAa,KAAK,CAAC,CAAC,CAAC;EAC9D;EACA,cAAc,aAAa,SAAS;EACpC,oBAAoB,oBAAoB,MAAM,QAAQ,WAAW,qBAAqB;EACtF,0BAA0B,yBAAyB,MAAM;CAC3D;CAEA,MAAM,SAAmC,CAAC;CAC1C,YAAY,OAAO,YAAY,SAAS,MAAM;CAC9C,aAAa,YAAY,SAAS,MAAM;CACxC,iBAAiB,SAAS,MAAM;CAChC,oBAAoB,MAAM,gBAAgB,MAAM,YAAY,SAAS,MAAM;CAC3E,iBAAiB,YAAY,SAAS,MAAM;CAC5C,gBAAgB,YAAY,SAAS,MAAM;CAE3C,MAAM,OAAO,UAAU,SAAS,YAAY,MAAM;CAClD,MAAM,SAAS,OAAO,MAAM,MAAM,EAAE,aAAa,UAAU,IACvD,SACA,OAAO,SAAS,IACd,SACA;CAEN,OAAO;EACL,QAAQ,MAAM;EACd;EACA,YAAY,MAAM,cAAc;EAChC;EACA,SAAS,WAAW,WAAW,MAAM,eAAe,MAAM,aAAa,UAAU;EACjF;EACA;EACA;EACA,SAAS,MAAM,WAAW;EAC1B,cAAc,MAAM,gBAAgB;EACpC,SAAS,cAAc,MAAM,QAAQ,QAAQ,SAAS,MAAM;CAC9D;AACF;AAEA,SAAgB,wBAAwB,OAA2D;CACjG,MAAM,YAAY,0BAA0B,KAAK;CACjD,IAAI,UAAU,WAAW,QACvB,MAAM,IAAI,kBAAkB,UAAU,OAAO;CAE/C,OAAO;AACT;AAEA,SAAS,gBACP,MACA,aACA,YACa;CACb,IAAI,aAAa,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,WAAW;CACxE,IAAI,YAAY,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,UAAU;CACtE,OAAO,CAAC,GAAG,IAAI;AACjB;AAEA,SAAS,qBACP,QACA,aACA,YACwB;CACxB,IAAI,aACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,WAAW;CAC1F,IAAI,YACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,UAAU;CACzF,OAAO,CAAC,GAAG,MAAM;AACnB;AAEA,SAAS,6BACP,OACA,OACsB;CACtB,MAAM,QAAQ;CACd,IAAI,OAAO,OAAO,OAAO,aAAa,GACpC,MAAM,IAAI,gBACR,UAAU,MAAM,2DAClB;CAEF,IAAI,MAAM,iBAAiB,KAAA,KAAa,CAAC,gBAAgB,SAAS,MAAM,YAAY,GAClF,MAAM,IAAI,gBACR,UAAU,MAAM,gCAAgC,gBAAgB,KAAK,IAAI,GAC3E;CAEF,OAAO;AACT;AAEA,SAAS,YACP,OACA,YACA,SACA,QACM;CACN,IAAI,WAAW,iBAAiB,CAAC,MAAM,YAAY,MAAM,WAAW,UAAU,OAAO,GACnF,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,gBAAgB,WAAW,kBACrC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,cAAc,qBAAqB,WAAW,iBAAiB;CACpF,CAAC;CAEH,IAAI,WAAW,kBAAkB,QAAQ,YAAY,YAAY,GAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;AAEL;AAEA,SAAS,aACP,YACA,SACA,QACM;CACN,IAAI,QAAQ,aAAa,WAAW,eAClC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,WAAW,uBAAuB,WAAW,cAAc;CAChF,CAAC;CAEH,IAAI,QAAQ,eAAe,GACzB,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa;CAClC,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,cAAc,MACrD,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,WAAW,WAAW,aAC7D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,YAAY,IAAI,QAAQ,QAAQ,EAAE,KAAK,IAAI,WAAW,WAAW,EAAE;CAC7E,CAAC;CAEH,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,iBACP,SACA,QACM;CACN,IAAI,QAAQ,oBAAoB,MAC9B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QACE,QAAQ,2BAA2B,IAC/B,GAAG,QAAQ,yBAAyB,wDACpC;CACR,CAAC;CAEH,IAAI,QAAQ,sBAAsB,GAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,oBAAoB;CACzC,CAAC;AAEL;AAEA,SAAS,oBACP,cACA,YACA,SACA,QACM;CACN,IAAI,WAAW,kBAAkB,QAAQ,cAAc,WAAW,gBAChE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,YAAY,wBAAwB,WAAW,eAAe;CACnF,CAAC;CAEH,IAAI,QAAQ,eAAe,QAAQ,QAAQ,aAAa,WAAW,eACjE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,sBAAsB,IAAI,QAAQ,UAAU,EAAE,KAAK,IAAI,WAAW,aAAa,EAAE;CAC3F,CAAC;CAEH,IAAI,gBAAgB,CAAC,aAAa,SAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM,QAAQ,aAAa,iBAAiB;EAC5C,QAAQ,aAAa;CACvB,CAAC;AAEL;AAEA,SAAS,iBACP,YACA,SACA,QACM;CACN,IAAI,CAAC,WAAW,uBAAuB;CACvC,IAAI,QAAQ,aAAa,QAAQ,iBAC/B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa,QAAQ,gBAAgB;CAC1D,CAAC;AAEL;AAEA,SAAS,gBACP,YACA,SACA,QACM;CACN,IAAI,OAAO,SAAS,WAAW,cAAc,KAAK,QAAQ,gBAAgB,MACxE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,gBAAgB,QAAQ,QAAQ,cAAc,WAAW,gBAC1E,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,eAAe,IAAI,QAAQ,WAAW,EAAE,KAAK,IAAI,WAAW,cAAc,EAAE;CACtF,CAAC;CAEH,IAAI,OAAO,SAAS,WAAW,YAAY,KAAK,QAAQ,cAAc,MACpE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cACtE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,UACP,SACA,YACA,QACyB;CACzB,OAAO;EACL,KACE,UACA,QACA,QAAQ,QAAQ,gBAAgB,KAAK,IAAI,GAAG,WAAW,gBAAgB,CAAC,GACxE,GAAG,QAAQ,cAAc,sBAAsB,QAAQ,YAAY,SACrE;EACA,KACE,WACA,QACA,QAAQ,aAAa,QAAQ,QAAQ,cAAc,OAC/C,OACA,KAAK,IAAI,QAAQ,UAAU,QAAQ,SAAS,GAChD,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS,GACtE;EACA,KACE,eACA,QACA,QAAQ,iBACR,eAAe,IAAI,QAAQ,eAAe,EAAE,oBAAoB,QAAQ,oBAAoB,gBAAgB,QAAQ,0BACtH;EACA,KACE,kBACA,QACA,SAAS,QAAQ,YAAY,WAAW,aAAa,GACrD,eAAe,QAAQ,YAAY,cAAc,IAAI,QAAQ,UAAU,GACzE;EACA,KACE,eACA,QACA,QAAQ,eAAe,IAAI,IAAI,QAAQ,kBAAkB,QAAQ,YACjE,mBAAmB,QAAQ,gBAAgB,GAAG,QAAQ,YACxD;EACA,KACE,cACA,QACA,gBAAgB,SAAS,UAAU,GACnC,eAAe,IAAI,QAAQ,WAAW,EAAE,aAAa,IAAI,QAAQ,SAAS,GAC5E;CACF;AACF;AAEA,SAAS,KACP,MACA,QACA,OACA,QACuB;CACvB,MAAM,MAAM,OAAO,QAAQ,MAAM,EAAE,SAAS,IAAI;CAMhD,OAAO;EAAE;EAAM,QALA,IAAI,MAAM,MAAM,EAAE,aAAa,UAAU,IACpD,SACA,IAAI,SAAS,IACX,SACA;EACiB,OAAO,UAAU,OAAO,OAAO,QAAQ,KAAK;EAAG;CAAO;AAC/E;AAEA,SAAS,oBAAoB,WAAqE;CAChG,MAAM,SAAuC;EAAE,OAAO;EAAG,KAAK;EAAG,MAAM;EAAG,SAAS;CAAE;CACrF,KAAK,MAAM,YAAY,WAAW,OAAO,SAAS,SAAS,QAAQ;CACnE,OAAO;AACT;AAEA,SAAS,aAAa,WAA+D;CACnF,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,YAAY,WAAW;EAChC,MAAM,SAAS,SAAS,MAAM,UAAU,SAAS,MAAM,YAAY;EACnE,IAAI,WAAW,IAAI,WAAW,KAAK;CACrC;CACA,OAAO;AACT;AAEA,SAAS,oBACP,MACA,QACA,WACuC;CACvC,MAAM,MAA6C,CAAC;CACpD,KAAK,MAAM,OAAO,MAKhB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,eACJ,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,YACnD,IAAI,eACJ;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OAAO;EAChD,MAAM,eACJ,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,YACvD,MAAM,eACN;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,QAAiE;CACjG,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,SAAS,QAClB,KAAK,MAAM,OAAO,MAAM,OAAO,CAAC,GAAG;EACjC,MAAM,UAAU,IAAI,sBAAsB;EAC1C,IAAI,YAAY,IAAI,YAAY,KAAK;CACvC;CAEF,OAAO;AACT;AAEA,SAAS,WACP,MACA,QACA,WAC4B;CAC5B,MAAM,MAAkC,CAAC;CACzC,KAAK,MAAM,OAAO,MAChB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,YAAY,IAAI,QAAQ,IAAI;EAClC,IAAI,KAAK,EAAE,QAAQ,OAAO,cAAc,YAAY,YAAY,EAAE,CAAC;CACrE;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OACzC,IAAI,KAAK,EAAE,SAAS,MAAM,KAAK,UAAU,KAAK,EAAE,CAAC;CAGrD,OAAO;AACT;AAEA,SAAS,gBAAgB,UAAsD;CAC7E,MAAM,aAAa,SAAS,QAAQ,YAAgC,YAAY,IAAI;CACpF,IAAI,WAAW,WAAW,GAAG,OAAO;CACpC,OAAO,WAAW,OAAO,OAAO,CAAC,CAAC,SAAS,WAAW;AACxD;AAEA,SAAS,eAAe,KAAgB,WAAmC;CACzE,IAAI,uBAAuB,GAAG,GAAG,OAAO;CACxC,MAAM,QAAQ,cAAc,GAAG;CAC/B,OAAO,UAAU,KAAA,IAAY,OAAO,SAAS;AAC/C;AAEA,SAAS,uBAAuB,KAAyB;CACvD,OAAO,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB;AAChE;AAEA,SAAS,iBAAiB,OAA6B,WAAmC;CACxF,IAAI,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,WAAW,OAAO;CACjF,IAAI,MAAM,OAAO,OAAO,OAAO;CAC/B,IAAI,eAAe,MAAM,KAAK,GAAG,OAAO,MAAM,SAAS;CACvD,OAAO,MAAM,OAAO,OAAO,OAAO;AACpC;AAEA,SAAS,wBACP,SACkD;CAClD,OAAO,YAAY,YAAY,YAAY,eAAe,YAAY;AACxE;AAEA,SAAS,gBAAgB,SAA4D;CACnF,IAAI,YAAY,aAAa,OAAO;CACpC,IAAI,wBAAwB,OAAO,GAAG,OAAO;AAE/C;AAEA,SAAS,UAAU,MAA4B,OAA8B;CAC3E,OAAO,KACJ,QAAQ,QAAQ,IAAI,aAAa,KAAK,CAAC,CACvC,IAAI,aAAa,CAAC,CAClB,OAAO,cAAc;AAC1B;;;;;;;AAQA,SAAS,cAAc,KAAoC;CACzD,MAAM,QAAQ,mBAAmB,KAAK,IAAI,aAAa,YAAY,YAAY,QAAQ;CACvF,OAAO,eAAe,KAAK,IAAI,QAAQ,KAAA;AACzC;AAEA,SAAS,WAAW,IAAsC;CACxD,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,OAAO,GAAG,QAAQ,KAAK,MAAM,MAAM,GAAG,CAAC,IAAI,GAAG;AAChD;AAEA,SAAS,iBAAiB,IAAuB,GAA0B;CACzE,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,MAAM,SAAS,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,OAAO,OAAO,KAAK,IAAI,OAAO,SAAS,GAAG,KAAK,IAAI,GAAG,KAAK,KAAK,IAAI,OAAO,MAAM,IAAI,CAAC,CAAC;AACzF;AAEA,SAAS,eAAe,OAAiC;CACvD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK;AAC3D;AAEA,SAAS,WAAW,GAAkB,GAAiC;CACrE,IAAI,MAAM,QAAQ,MAAM,MAAM,OAAO;CACrC,OAAO,IAAI;AACb;AAEA,SAAS,SAAS,KAAoB,QAA+B;CACnE,IAAI,QAAQ,MAAM,OAAO;CACzB,IAAI,UAAU,GAAG,OAAO,OAAO,IAAI,IAAI;CACvC,OAAO,QAAQ,IAAI,KAAK,IAAI,GAAG,GAAG,IAAI,MAAM;AAC9C;AAEA,SAAS,gBACP,SACA,YACe;CACf,MAAM,OAAO,OAAO,SAAS,WAAW,cAAc,IAClD,QAAQ,gBAAgB,OACtB,OACA,QAAQ,WAAW,iBAAiB,KAAK,IAAI,QAAQ,aAAa,KAAK,CAAC,IAC1E;CACJ,MAAM,UAAU,OAAO,SAAS,WAAW,YAAY,IACnD,QAAQ,cAAc,OACpB,OACA,QAAQ,WAAW,eAAe,KAAK,IAAI,QAAQ,WAAW,KAAK,CAAC,IACtE;CACJ,IAAI,SAAS,QAAQ,YAAY,MAAM,OAAO;CAC9C,OAAO,KAAK,IAAI,MAAM,OAAO;AAC/B;AAEA,SAAS,QAAQ,GAAmB;CAClC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;AAEA,SAAS,cACP,QACA,QACA,SACA,QACQ;CACR,MAAM,SAAS,sBAAsB,OAAO,IAAI;CAChD,MAAM,aAAa,aAAa,QAAQ,cAAc,cAAc,QAAQ,WAAW,eAAe,QAAQ,YAAY,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS;CAC9L,IAAI,OAAO,WAAW,GAAG,OAAO,GAAG,OAAO,IAAI;CAC9C,OAAO,GAAG,OAAO,IAAI,WAAW,WAAW,OAAO,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,GAAG;AAC/E;AAEA,SAAS,IAAI,GAA0B;CACrC,IAAI,MAAM,MAAM,OAAO;CACvB,OAAO,EAAE,QAAQ,CAAC;AACpB"}
|
package/dist/reporting.js
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
import { o as benjaminiHochberg } from "./power-and-mde-
|
|
2
|
-
import { l as wilcoxonSignedRank } from "./paired-arms-
|
|
3
|
-
import { r as pairedBootstrap } from "./paired-tests-
|
|
4
|
-
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-
|
|
1
|
+
import { o as benjaminiHochberg } from "./power-and-mde-B8F2RdcD.js";
|
|
2
|
+
import { l as wilcoxonSignedRank } from "./paired-arms-D4aeIHUy.js";
|
|
3
|
+
import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
|
|
4
|
+
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-BXeQ5Ues.js";
|
|
5
5
|
import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-CzK5DarL.js";
|
|
6
|
-
import { i as judgeReplayGate, n as evaluateReleaseConfidence, r as bootstrapCi, t as assertReleaseConfidence } from "./release-confidence-
|
|
7
|
-
import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-
|
|
6
|
+
import { i as judgeReplayGate, n as evaluateReleaseConfidence, r as bootstrapCi, t as assertReleaseConfidence } from "./release-confidence-nGDJiiwc.js";
|
|
7
|
+
import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-2D5Gw9z9.js";
|
|
8
8
|
export { RESEARCH_REPORT_HARD_PAIR_FLOOR, assertReleaseConfidence, benjaminiHochberg, bootstrapCi, evaluateInterimReleaseConfidence, evaluateReleaseConfidence, gainHistogram, judgeReplayGate, pairedBootstrap, pairedEvalueSequence, paretoChart, researchReport, rubricPredictiveValidity, summaryTable, wilcoxonSignedRank };
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { r as pearsonR } from "./descriptive-
|
|
1
|
+
import { r as pearsonR } from "./descriptive-1V17A-qa.js";
|
|
2
2
|
import { n as observedScore, s as trainingScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
|
|
3
3
|
import { c as validateRunRecord } from "./run-record-BC0ebuRP.js";
|
|
4
4
|
//#region src/campaign/run-record.ts
|
|
@@ -641,4 +641,4 @@ function defaultSecondary(verifiableOpts) {
|
|
|
641
641
|
//#endregion
|
|
642
642
|
export { campaignCellCostProvenance as a, campaignCellTaskScore as c, filterDeterministicallyRewarded as i, campaignCellToRunRecord as l, extractVerifiableReward as n, campaignCellExecutionEvidence as o, extractVerifiableRewardsFromRecords as r, campaignCellJudgeDimensions as s, detectRewardHacking as t, projectCampaignCellQuality as u };
|
|
643
643
|
|
|
644
|
-
//# sourceMappingURL=reward-hacking-
|
|
644
|
+
//# sourceMappingURL=reward-hacking-O5zKDANP.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"reward-hacking-t4lB1yt8.js","names":["mean","clamp01"],"sources":["../src/campaign/run-record.ts","../src/rl/verifiable-reward.ts","../src/rl/reward-hacking.ts"],"sourcesContent":["import type { AgentProfileCell } from '../agent-profile-cell'\nimport type { CostProvenance } from '../cost-ledger'\nimport type {\n JudgeScoresRecord,\n RunOutcome,\n RunRecord,\n RunSplitTag,\n RunTerminalOutcome,\n} from '../run-record'\nimport { validateRunRecord } from '../run-record'\nimport type { CampaignCellResult, JudgeScore } from './types'\n\nexport interface CampaignCellRunRecordOptions {\n runId: string\n experimentId: string\n candidateId: string\n model: string\n promptHash: string\n configHash: string\n commitSha: string\n splitTag: RunSplitTag\n seed?: number\n scenarioId?: string\n defaultCostUsd?: number\n agentProfile?: AgentProfileCell\n raw?: Record<string, number>\n}\n\nexport interface CampaignCellQualityProjection {\n score?: number\n judgeScores?: JudgeScoresRecord\n successfulJudgeScores: Record<string, JudgeScore>\n failedJudges: string[]\n raw: Record<string, number>\n}\n\n/**\n * A campaign cell carried a judge score without a `dimensions` record.\n *\n * Two exported types share the name `JudgeScore`: the campaign verdict\n * (`{ dimensions, composite, notes }` from `@tangle-network/agent-eval/campaign`)\n * and the root export's flat per-dimension row (`{ judgeName, dimension, score }`\n * from `@tangle-network/agent-eval`). Campaign aggregation accepts only the\n * campaign shape; the flat shape previously crashed here with an opaque\n * TypeError deep inside aggregation.\n */\nexport class CampaignJudgeScoreShapeError extends TypeError {\n constructor(judgeName: string) {\n super(\n `campaign cell judge '${judgeName}' carries a score without a 'dimensions' record. ` +\n `Use the campaign JudgeScore ({ dimensions, composite, notes }) from ` +\n `'@tangle-network/agent-eval/campaign'; the root export's JudgeScore ` +\n `({ judgeName, dimension, score }) is a different type with the same name.`,\n )\n this.name = 'CampaignJudgeScoreShapeError'\n }\n}\n\nexport interface CampaignCellExecutionEvidence {\n terminalOutcome: RunTerminalOutcome\n executionErrorCount?: number\n judgeErrorCount?: number\n unclassifiedErrorCount?: number\n terminalFailureReason?: string\n}\n\n/**\n * Project one campaign cell into the canonical run format.\n *\n * A dispatch error establishes terminal execution failure. A judge error only\n * establishes that quality measurement failed after dispatch completed.\n * Failures without a stage remain unknown. No failure becomes a zero-quality\n * label.\n */\nexport function campaignCellToRunRecord<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n options: CampaignCellRunRecordOptions,\n): RunRecord {\n const quality = projectCampaignCellQuality(cell)\n const execution = campaignCellExecutionEvidence(cell)\n const judgeErrorCount = Math.max(\n quality.raw.judge_error_count ?? 0,\n execution.judgeErrorCount ?? 0,\n )\n const cellCostProvenance = campaignCellCostProvenance(cell)\n const costProvenance: CostProvenance =\n cellCostProvenance.kind === 'uncaptured' && options.defaultCostUsd !== undefined\n ? { kind: 'estimated', usd: options.defaultCostUsd }\n : cellCostProvenance\n const costUsd = costProvenance.kind === 'uncaptured' ? null : costProvenance.usd\n const raw: Record<string, number> = {\n ...finiteMetrics(options.raw),\n ...quality.raw,\n rep: cell.rep,\n duration_ms: cell.durationMs,\n ...(costUsd === null ? {} : { cost_usd: costUsd }),\n // Retain the observed subtotal even when the caller supplies an estimated total.\n ...(cellCostProvenance.kind === 'uncaptured' ? { cost_known_subtotal_usd: cell.costUsd } : {}),\n cost_observed: costProvenance.kind === 'observed' ? 1 : 0,\n cost_estimated: costProvenance.kind === 'estimated' ? 1 : 0,\n cost_uncaptured: costProvenance.kind === 'uncaptured' ? 1 : 0,\n tokens_input: cell.tokenUsage.input,\n tokens_output: cell.tokenUsage.output,\n tokens_known: cell.tokenUsage.tokensKnown === false ? 0 : 1,\n latency_ms: cell.durationMs,\n ...(execution.executionErrorCount === undefined\n ? {}\n : { execution_error_count: execution.executionErrorCount }),\n ...(judgeErrorCount > 0 ? { judge_error_count: judgeErrorCount } : {}),\n ...(execution.unclassifiedErrorCount === undefined\n ? {}\n : { unclassified_error_count: execution.unclassifiedErrorCount }),\n }\n if (typeof cell.generation === 'number') raw.generation = cell.generation\n if (cell.tokenUsage.reasoning !== undefined) {\n raw.tokens_reasoning = cell.tokenUsage.reasoning\n }\n if (cell.tokenUsage.cached !== undefined) raw.tokens_cached = cell.tokenUsage.cached\n if (cell.tokenUsage.cacheWrite !== undefined) {\n raw.tokens_cache_write = cell.tokenUsage.cacheWrite\n }\n if (cell.tokenUsage.tokensKnown !== false && costUsd !== null && costUsd > 0) {\n raw.tokens_per_dollar = (cell.tokenUsage.input + cell.tokenUsage.output) / costUsd\n }\n if (costUsd !== null && quality.score !== undefined && quality.score > 0.01) {\n raw.cost_per_quality = costUsd / quality.score\n }\n\n const outcome: RunOutcome = {\n raw,\n ...(quality.judgeScores ? { judgeScores: quality.judgeScores } : {}),\n }\n if (quality.score !== undefined) {\n if (options.splitTag === 'holdout') outcome.holdoutScore = quality.score\n else outcome.searchScore = quality.score\n }\n\n return validateRunRecord({\n runId: options.runId,\n experimentId: options.experimentId,\n candidateId: options.candidateId,\n seed: options.seed ?? cell.seed,\n model: options.model,\n promptHash: options.promptHash,\n configHash: options.configHash,\n commitSha: options.commitSha,\n wallMs: cell.durationMs,\n costUsd,\n costProvenance,\n tokenUsage: { ...cell.tokenUsage },\n terminalOutcome: execution.terminalOutcome,\n ...(execution.terminalFailureReason\n ? { terminalFailureReason: execution.terminalFailureReason }\n : {}),\n outcome,\n splitTag: options.splitTag,\n scenarioId: options.scenarioId ?? cell.scenarioId,\n ...(options.agentProfile ? { agentProfile: options.agentProfile } : {}),\n })\n}\n\n/**\n * Validate the cost fields that cross campaign cache and RunRecord boundaries.\n * `costUsd` is a known subtotal for uncaptured cells, but it must equal the\n * authoritative total whenever that total is observed or estimated.\n */\nexport function campaignCellCostProvenance<TArtifact>(\n cell: Pick<CampaignCellResult<TArtifact>, 'cellId' | 'costUsd' | 'costProvenance'>,\n): CostProvenance {\n if (!Number.isFinite(cell.costUsd) || cell.costUsd < 0) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid costUsd`)\n }\n const provenance = cell.costProvenance\n if (!provenance || typeof provenance !== 'object') {\n throw new Error(`campaign cell '${cell.cellId}' has no costProvenance`)\n }\n if (provenance.kind === 'uncaptured') {\n if (provenance.usd !== null) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid uncaptured costProvenance`)\n }\n return { kind: 'uncaptured', usd: null }\n }\n if (\n (provenance.kind !== 'observed' && provenance.kind !== 'estimated') ||\n !Number.isFinite(provenance.usd) ||\n provenance.usd < 0\n ) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid costProvenance`)\n }\n if (provenance.usd !== cell.costUsd) {\n throw new Error(`campaign cell '${cell.cellId}' has costUsd inconsistent with costProvenance`)\n }\n return { kind: provenance.kind, usd: provenance.usd }\n}\n\nexport function campaignCellExecutionEvidence<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): CampaignCellExecutionEvidence {\n if (cell.errorStage === 'dispatch') {\n return {\n terminalOutcome: 'failed',\n executionErrorCount: 1,\n ...(cell.error ? { terminalFailureReason: cell.error } : {}),\n }\n }\n if (cell.errorStage === 'judge') {\n return {\n terminalOutcome: 'succeeded',\n executionErrorCount: 0,\n judgeErrorCount: 1,\n }\n }\n if (!cell.error) {\n return { terminalOutcome: 'succeeded', executionErrorCount: 0 }\n }\n return {\n terminalOutcome: 'unknown',\n unclassifiedErrorCount: 1,\n }\n}\n\n/**\n * Produce the only task-quality view used by campaign aggregates and exports.\n *\n * Successful judge results remain available for diagnosis after another judge\n * fails, but a task score exists only for an error-free cell whose reported\n * judge values are all finite.\n */\nexport function projectCampaignCellQuality<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): CampaignCellQualityProjection {\n if (cell.errorStage === 'dispatch') {\n return { successfulJudgeScores: {}, failedJudges: [], raw: {} }\n }\n\n const perJudge: Record<string, Record<string, number>> = {}\n const successfulJudgeScores: Record<string, JudgeScore> = {}\n const dimensionValues = new Map<string, number[]>()\n const composites: number[] = []\n const notes: string[] = []\n const failedJudges = new Set<string>(\n cell.errorStage === 'judge' ? [cell.errorJudge ?? 'unknown-judge'] : [],\n )\n const raw: Record<string, number> = {}\n\n for (const [judgeName, score] of Object.entries(cell.judgeScores)) {\n const dimensionsShape = (score as { dimensions?: unknown }).dimensions\n if (\n typeof dimensionsShape !== 'object' ||\n dimensionsShape === null ||\n Array.isArray(dimensionsShape)\n ) {\n throw new CampaignJudgeScoreShapeError(judgeName)\n }\n const finiteDimensions = Object.values(score.dimensions).every(Number.isFinite)\n if (score.failed || !Number.isFinite(score.composite) || !finiteDimensions) {\n failedJudges.add(judgeName)\n continue\n }\n\n composites.push(score.composite)\n successfulJudgeScores[judgeName] = score\n const dimensions = { ...score.dimensions }\n perJudge[judgeName] = dimensions\n for (const [dimension, value] of Object.entries(dimensions)) {\n raw[`${judgeName}.${dimension}`] = value\n const values = dimensionValues.get(dimension) ?? []\n values.push(value)\n dimensionValues.set(dimension, values)\n }\n if (score.notes) notes.push(`${judgeName}: ${score.notes}`)\n for (const failedJudge of score.failedJudges ?? []) {\n failedJudges.add(`${judgeName}/${failedJudge}`)\n }\n }\n\n if (failedJudges.size > 0) raw.judge_error_count = failedJudges.size\n const sortedFailedJudges = [...failedJudges].sort()\n if (composites.length === 0) {\n return {\n successfulJudgeScores,\n failedJudges: sortedFailedJudges,\n raw,\n }\n }\n\n const composite = mean(composites)\n const perDimMean = Object.fromEntries(\n [...dimensionValues.entries()].map(([dimension, values]) => [dimension, mean(values)]),\n )\n const complete =\n cell.error === undefined && cell.errorStage === undefined && failedJudges.size === 0\n if (complete) raw.composite = composite\n\n return {\n ...(complete ? { score: composite } : {}),\n raw,\n successfulJudgeScores,\n failedJudges: sortedFailedJudges,\n judgeScores: {\n perJudge,\n perDimMean,\n composite,\n ...(sortedFailedJudges.length > 0 ? { failedJudges: sortedFailedJudges } : {}),\n ...(notes.length > 0 ? { notes: notes.join(' | ') } : {}),\n },\n }\n}\n\n/** Read the canonical task score without recomputing cell quality. */\nexport function campaignCellTaskScore<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): number | undefined {\n return projectCampaignCellQuality(cell).score\n}\n\n/** Read canonical successful judge dimensions without recomputing cell quality. */\nexport function campaignCellJudgeDimensions<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): Record<string, Record<string, number>> {\n return projectCampaignCellQuality(cell).judgeScores?.perJudge ?? {}\n}\n\nfunction finiteMetrics(metrics: Record<string, number> | undefined): Record<string, number> {\n const finite: Record<string, number> = {}\n for (const [key, value] of Object.entries(metrics ?? {})) {\n if (Number.isFinite(value)) finite[key] = value\n }\n return finite\n}\n\nfunction mean(values: number[]): number {\n return values.reduce((sum, value) => sum + value, 0) / values.length\n}\n","/**\n * Verifiable reward channel.\n *\n * For RL on coding / math / theorem-proving / structured-output tasks, the\n * reward signal is *decidable* — a test passes or fails, a proof checks or\n * doesn't, an output validates against a schema or doesn't. These rewards\n * are dramatically more useful for RL training than LLM-judge scores\n * because they don't drift, can't be Goodhart-gamed by the policy in the\n * same way, and don't require a separate calibration loop.\n *\n * The `MultiLayerVerifier` already produces this signal — it just doesn't\n * surface it in a shape that's clean enough for RL training. This module\n * wraps the verifier output so consumers can:\n *\n * 1. Extract a clean `VerifiableReward` from a `VerificationReport`\n * 2. Distinguish *deterministic* rewards (compile, test, schema) from\n * *probabilistic* rewards (judge) so they can be weighted differently\n * in the RL training step\n * 3. Filter `RunRecord[]` to only those with a verifiable reward,\n * producing the clean training set that DeepSeek-R1-style GRPO and\n * AlphaProof-style search both depend on\n *\n * Why this matters: every credible 2025-2026 frontier RL result on coding\n * agents leans on verifiable reward (DeepSeek-R1 GRPO on test pass-rate,\n * o-series RL on math/code, AlphaProof on Lean kernel checking). Mixing\n * judge scores into the reward signal poisons the gradient. This module\n * is the seam.\n */\n\nimport type { LayerResult, VerificationReport } from '../multi-layer-verifier'\nimport { isRealnessGated, observedScore, trainingScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport type { VerificationStrategySource } from '../verification-strategy'\n\n/**\n * What produced a reward. This is the verification-strategy family\n * (`src/verification-strategy.ts`) — one open vocabulary shared by rewards\n * and verdict certifications, so an RL consumer and a certification reader\n * mean the same thing by `'proof-kernel'`.\n *\n * Beyond the classic answer-key members (`compile`, `test`, `schema`,\n * `sandbox`, `judge`, `composite`) the family carries four members for\n * tasks with no held-out suite, each with a documented failure mode:\n *\n * - `'proof-kernel'` — kernel-checked formal proof. Failure mode: the\n * formalization gap — the kernel never certifies that the formal\n * statement matches the informal claim (`src/verification-strategy.ts`\n * is the discharge protocol).\n * - `'invariant'` — invariant / metamorphic properties held. Failure\n * mode: weak invariants pass everything; the set needs seeded-bug\n * calibration before its pass carries weight.\n * - `'replication'` — independent re-execution from pinned inputs.\n * Failure mode: re-runs the method, so it never catches an error the\n * method itself carries.\n * - `'agreement'` — independently-derived results agree. Failure mode:\n * the shared blind spot — derivers with common training corpora or\n * priors agree for the same wrong reason. Probabilistic: the check may\n * be mechanical, but the independence it certifies is not.\n *\n * The deterministic/probabilistic axis stays on `VerifiableReward` —\n * the producer declares it per reward, and `VERIFICATION_STRATEGIES`\n * documents each member's default class.\n */\nexport type VerifiableRewardSource = VerificationStrategySource\n\nexport interface VerifiableReward {\n /** Scalar in [0, 1]. The RL training signal. */\n value: number\n /** What produced the reward — different sources have different determinism. */\n source: VerifiableRewardSource\n /**\n * Determinism class. `'deterministic'` rewards are repeatable byte-for-byte\n * given the same inputs (compile, test, schema validation, sandbox exit code).\n * `'probabilistic'` rewards depend on a stochastic component (LLM judge).\n * Mixing these in the same training batch without separation is a known\n * footgun in production RLHF pipelines.\n */\n determinism: 'deterministic' | 'probabilistic'\n /**\n * Confidence in the reward value. For deterministic sources this is 1.0\n * (the bit either flipped or didn't). For judge sources this is the\n * judge-reported confidence or — when missing — a calibrated prior.\n */\n confidence: number\n /** The layer / judge id that produced the signal, for provenance. */\n origin: string\n /**\n * Per-source contribution to `value`, keyed by layer/judge id. Single-source\n * rewards carry one entry (`{ [origin]: value }`); composite rewards carry\n * every contributing layer's score — the anti-scalar-collapse surface RL\n * consumers weight per-source instead of trusting one blended number.\n */\n components: Record<string, number>\n /**\n * The run carries `outcome.realness.gated` — the authenticity gate flagged\n * its success signal as faked.\n *\n * With the gate applied (the default) `value` and every `components` entry\n * are 0 on such a run; with `applyRealnessGate: false` the observed numbers\n * come back untouched and this flag is the only marker that they are not to\n * be trusted. Either way it distinguishes \"measured a genuine failure\" from\n * \"claimed a success we refuse to believe\", which a bare 0 cannot.\n */\n realnessGated?: boolean\n /**\n * Whether an authenticity screen COULD run on this reward at all — the same\n * distinction `RolloutOutcome.realness_screened` draws, for the same reason.\n *\n * `false` on every reward from `extractVerifiableReward`, because a\n * `VerificationReport` carries layer scores and nothing else: there is no\n * `outcome.realness` to consult, so no gate has run, and `realnessGated`\n * being absent there means \"unknown\", NOT \"clean\". Absent on the\n * `RunRecord` path when the record itself carries no realness verdict.\n *\n * This matters most exactly where it is easiest to miss: a report whose\n * deterministic layers all passed yields `determinism: 'deterministic'`,\n * `confidence: 1` — the highest-credibility reward this module can emit —\n * and a stubbed integration reporting green is precisely what a gamed run\n * looks like. Consumers driving training off this shape must screen the run\n * themselves; the flag is what tells them nobody has.\n */\n realnessScreened?: boolean\n}\n\nexport interface VerifiableRewardExtractionOptions {\n /**\n * Which layers count as deterministic-reward sources. The verifier doesn't\n * tag layers as \"this is verifiable\"; the caller declares it via this list\n * (or via the layer name → source mapping). Default treats common names\n * (`install`, `typecheck`, `build`, `lint`, `test`, `compile`, `schema`,\n * `sandbox`) as deterministic.\n */\n deterministicLayers?: string[]\n /**\n * Map layer name → reward source. Defaults to a sensible string-match.\n */\n sourceFor?: (layerName: string) => VerifiableRewardSource\n /**\n * Whether to fall back to a probabilistic (judge) reward when no\n * deterministic layer produced a numeric score. Default `true`. Set to\n * `false` for \"deterministic-only\" training pipelines that should\n * discard runs without a verifiable signal.\n */\n fallbackToJudge?: boolean\n /**\n * Default confidence for probabilistic (judge) rewards when the judge\n * doesn't report one. Default `0.7`.\n */\n judgeConfidenceFloor?: number\n /**\n * Whether the anti-Goodhart realness gate applies. Default `true`, and the\n * default is the one every training path must keep.\n *\n * Set `false` ONLY for detection and analysis. `rl/reward-hacking.ts` does,\n * for the same reason it reads `observedScore` for its proxy: it measures the\n * DIVERGENCE between the judge signal and the deterministic one, and a\n * deterministic reward that another gate already forced to 0 manufactures\n * exactly that divergence on exactly the gamed population. The detector would\n * then be re-reporting a verdict it was supposed to reach independently.\n */\n applyRealnessGate?: boolean\n}\n\n// 'agreement' is deliberately absent: agreement between fixed artifacts is\n// mechanical, but the independent derivation it certifies is stochastic, so\n// a producer that wants a deterministic agreement reward must declare it\n// (via `deterministicLayers` or by constructing the reward directly).\nconst DEFAULT_DETERMINISTIC_LAYERS = new Set([\n 'install',\n 'typecheck',\n 'build',\n 'lint',\n 'test',\n 'compile',\n 'schema',\n 'sandbox',\n 'unit_tests',\n 'integration_tests',\n 'proof_kernel',\n 'proof-kernel',\n 'invariant',\n 'metamorphic',\n 'replication',\n])\n\nconst DEFAULT_SOURCE_FOR = (name: string): VerifiableRewardSource => {\n const lower = name.toLowerCase()\n if (lower.includes('test')) return 'test'\n if (\n lower.includes('compile') ||\n lower.includes('build') ||\n lower.includes('typecheck') ||\n lower.includes('lint')\n )\n return 'compile'\n if (lower.includes('schema')) return 'schema'\n if (lower.includes('sandbox')) return 'sandbox'\n if (lower.includes('judge') || lower.includes('semantic')) return 'judge'\n // Answer-key names above take precedence: a name like 'proof_test' maps\n // to 'test'; only names with no answer-key match reach the open-family\n // members.\n if (lower.includes('proof') || lower.includes('kernel') || lower.includes('lean')) {\n return 'proof-kernel'\n }\n if (lower.includes('invariant') || lower.includes('metamorphic')) return 'invariant'\n if (lower.includes('replicat')) return 'replication'\n if (lower.includes('agreement') || lower.includes('consensus')) return 'agreement'\n return 'composite'\n}\n\n/**\n * Extract a `VerifiableReward` from a `VerificationReport`.\n *\n * Strategy: prefer the deterministic layers (in order: test → compile →\n * schema → sandbox), fall back to the judge layer if `fallbackToJudge` is\n * true, return `null` if no signal qualifies. When multiple deterministic\n * layers contribute, return a `'composite'` source with a weighted blend.\n *\n * NO realness gate is applied and none can be: a `VerificationReport` carries\n * layer scores and nothing about whether the run faked them — `realness` lives\n * on the `RunRecord`. Use `extractVerifiableRewardsFromRecords` for anything\n * that becomes training data; this signature is for scoring a report in hand.\n */\nexport function extractVerifiableReward(\n report: VerificationReport,\n opts: VerifiableRewardExtractionOptions = {},\n): VerifiableReward | null {\n const deterministicSet = new Set(opts.deterministicLayers ?? [...DEFAULT_DETERMINISTIC_LAYERS])\n const sourceFor = opts.sourceFor ?? DEFAULT_SOURCE_FOR\n const fallbackToJudge = opts.fallbackToJudge ?? true\n const judgeFloor = opts.judgeConfidenceFloor ?? 0.7\n\n const deterministic = report.layers.filter(\n (layer) => deterministicSet.has(layer.layer) && isMeasuredLayer(layer),\n )\n\n if (deterministic.length === 1) {\n const layer = deterministic[0]!\n const value = clamp01(layer.score!)\n return {\n value,\n source: sourceFor(layer.layer),\n determinism: 'deterministic',\n confidence: 1,\n origin: layer.layer,\n components: { [layer.layer]: value },\n realnessScreened: false,\n }\n }\n\n if (deterministic.length > 1) {\n // Composite: weighted blend by `Layer.weight` if present, else equal.\n let num = 0\n let denom = 0\n const components: Record<string, number> = {}\n for (const l of deterministic) {\n const w = (l.detail?.weight as number | undefined) ?? 1\n num += w * (l.score ?? 0)\n denom += w\n components[l.layer] = l.score!\n }\n return {\n value: denom === 0 ? 0 : clamp01(num / denom),\n source: 'composite',\n determinism: 'deterministic',\n confidence: 1,\n origin: deterministic.map((l) => l.layer).join('+'),\n components,\n realnessScreened: false,\n }\n }\n\n if (!fallbackToJudge) return null\n\n const judge =\n report.layers.find((layer) => isMeasuredLayer(layer) && sourceFor(layer.layer) === 'judge') ??\n report.layers.find(isMeasuredLayer)\n\n if (!judge) return null\n\n const confFromDetail = judge.detail?.confidence as number | undefined\n const judgeValue = clamp01(judge.score!)\n return {\n value: judgeValue,\n source: 'judge',\n determinism: 'probabilistic',\n confidence: typeof confFromDetail === 'number' ? confFromDetail : judgeFloor,\n origin: judge.layer,\n components: { [judge.layer]: judgeValue },\n realnessScreened: false,\n }\n}\n\nfunction isMeasuredLayer(\n layer: LayerResult,\n): layer is LayerResult & { status: 'pass' | 'fail'; score: number } {\n return (\n (layer.status === 'pass' || layer.status === 'fail') &&\n typeof layer.score === 'number' &&\n Number.isFinite(layer.score) &&\n layer.score >= 0 &&\n layer.score <= 1\n )\n}\n\n/**\n * Extract verifiable rewards from `RunRecord[]` produced via the\n * `verificationReportToRunRecord` adapter (which encodes per-layer scores\n * in `outcome.raw['layer.<name>']`). For records that don't carry layer\n * scores, returns `null` for that record.\n *\n * This is the canonical bridge from \"campaign-shaped artifacts\" to\n * \"RL-training-ready reward signals\": every record that has a clean\n * verifiable reward becomes a training datum, every record that doesn't\n * gets filtered out (or kept with `'probabilistic'` determinism for\n * separate downstream handling).\n *\n * The realness gate applies to EVERY channel here, and to the deterministic one\n * MOST. It is tempting to reason that a decidable signal cannot be gamed, so\n * the gate is redundant on it — that reasoning is backwards. `realness.gated`\n * means the run's success signal was FAKED, and a test suite reporting green on\n * a stubbed integration is precisely what that looks like: the deterministic\n * layer is the thing that got faked. Exporting it ungated hands a trainer the\n * highest-credibility reward the module can emit (`determinism: 'deterministic'`,\n * `confidence: 1`) for the one population the gate exists to catch. Pass\n * `applyRealnessGate: false` only to look at the ungated numbers for detection.\n */\nexport function extractVerifiableRewardsFromRecords(\n runs: RunRecord[],\n opts: VerifiableRewardExtractionOptions = {},\n): Array<{ runId: string; reward: VerifiableReward | null }> {\n const sourceFor = opts.sourceFor ?? DEFAULT_SOURCE_FOR\n const deterministicSet = new Set(opts.deterministicLayers ?? [...DEFAULT_DETERMINISTIC_LAYERS])\n const fallbackToJudge = opts.fallbackToJudge ?? true\n const judgeFloor = opts.judgeConfidenceFloor ?? 0.7\n const applyGate = opts.applyRealnessGate ?? true\n\n return runs.map((run) => {\n const flagged = isRealnessGated(run)\n // Present only when the record carries an actual realness verdict. Absent\n // is the honest \"unknown\"; `false` is reserved for a producer that\n // declares it HAS no screen, which is the report-shaped path above.\n const screened = run.outcome.realness === undefined ? {} : ({ realnessScreened: true } as const)\n // Zeroed with `value`, never left at the measured number: `components`\n // exists so an RL consumer can re-weight per source, and a raw layer score\n // surviving there would let that re-weighting reconstruct the very reward\n // the gate just refused. The measured layer scores stay on\n // `run.outcome.raw['layer.*']`, which is where analysis reads them.\n const gate = (value: number): number => (applyGate && flagged ? 0 : value)\n // Recover per-layer scores from outcome.raw['layer.<name>']\n const layerScores: Array<{ name: string; score: number }> = []\n for (const [k, v] of Object.entries(run.outcome.raw)) {\n if (\n k.startsWith('layer.') &&\n !k.includes('.', 6) &&\n typeof v === 'number' &&\n Number.isFinite(v)\n ) {\n layerScores.push({ name: k.slice('layer.'.length), score: v })\n }\n }\n const det = layerScores.filter((l) => deterministicSet.has(l.name))\n\n if (det.length === 1) {\n const layer = det[0]!\n const value = gate(clamp01(layer.score))\n return {\n runId: run.runId,\n reward: {\n value,\n source: sourceFor(layer.name),\n determinism: 'deterministic',\n confidence: 1,\n origin: layer.name,\n components: { [layer.name]: value },\n realnessGated: flagged,\n ...screened,\n },\n }\n }\n if (det.length > 1) {\n const value = gate(clamp01(det.reduce((s, l) => s + l.score, 0) / det.length))\n // Same clamp as the headline value: a producer writing layer.score 1.5\n // into outcome.raw must not propagate 1.5 through a component either.\n const components: Record<string, number> = Object.fromEntries(\n det.map((l) => [l.name, gate(clamp01(l.score))]),\n )\n return {\n runId: run.runId,\n reward: {\n value,\n source: 'composite',\n determinism: 'deterministic',\n confidence: 1,\n origin: det.map((l) => l.name).join('+'),\n components,\n realnessGated: flagged,\n ...screened,\n },\n }\n }\n if (!fallbackToJudge) return { runId: run.runId, reward: null }\n\n // Probabilistic fallback: the run's primary score. `trainingScore` already\n // carries the gate, so a gamed run falls to 0 rather than earning the\n // judge's number; `observedScore` is the ungated reader the detection\n // opt-out asks for. Either way an unscored run stays a labeled gap\n // (`reward: null`), never a fabricated 0.\n const primary = applyGate ? trainingScore(run) : observedScore(run)\n if (typeof primary !== 'number' || !Number.isFinite(primary)) {\n return { runId: run.runId, reward: null }\n }\n const primaryValue = clamp01(primary)\n return {\n runId: run.runId,\n reward: {\n value: primaryValue,\n source: 'judge',\n determinism: 'probabilistic',\n confidence: judgeFloor,\n origin: 'run.outcome.score',\n components: { 'run.outcome.score': primaryValue },\n realnessGated: flagged,\n ...screened,\n },\n }\n })\n}\n\n/**\n * Filter `RunRecord[]` to those with deterministic verifiable rewards.\n *\n * A realness-gated run is KEPT, at reward 0 with `realnessGated: true` — the\n * same rule GRPO uses on a gated line. 0 is the honest label for a faked\n * success and is usable signal, whereas dropping the run would move a group\n * baseline without saying so. (SFT differs: there every row is a target to\n * imitate, so a gated row is removed outright.)\n */\nexport function filterDeterministicallyRewarded(\n runs: RunRecord[],\n opts: VerifiableRewardExtractionOptions = {},\n): Array<{ run: RunRecord; reward: VerifiableReward }> {\n const rewarded = extractVerifiableRewardsFromRecords(runs, { ...opts, fallbackToJudge: false })\n const out: Array<{ run: RunRecord; reward: VerifiableReward }> = []\n for (let i = 0; i < runs.length; i++) {\n const r = rewarded[i]!\n if (r.reward && r.reward.determinism === 'deterministic') {\n out.push({ run: runs[i]!, reward: r.reward })\n }\n }\n return out\n}\n\nfunction clamp01(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n","/**\n * Reward hacking / Goodhart detection.\n *\n * Goodhart's Law says: when a measure becomes a target, it ceases to be\n * a good measure. In RLHF and agentic-RL settings this is the dominant\n * failure mode — the policy learns to produce outputs that score well on\n * the proxy reward (judge, rubric, test pass-rate) without producing\n * the underlying capability the proxy was meant to track.\n *\n * Krakovna et al. (2020, \"Specification Gaming Examples in AI\") and the\n * subsequent RLHF reward-hacking literature (Skalse et al. 2022, Kim et al.\n * 2023) converge on a few diagnostic signatures:\n *\n * 1. **Reward divergence:** the proxy reward grows while the held-out\n * ground-truth signal stagnates or drops. Predictive validity over\n * time captures this.\n * 2. **Distributional shift in outputs:** after RL, the policy produces\n * outputs that no longer match the reference distribution — usually\n * because it found a high-reward attractor that's degenerate (e.g.\n * one-token responses, repetition, formatting tricks).\n * 3. **Disagreement between independent rewards:** if you train on\n * reward A and a held-out independent reward B drops sharply, you're\n * probably hacking A.\n * 4. **Calibration drift:** the verifiable / deterministic component of\n * the reward is stable; the probabilistic / judge component drifts up\n * while the deterministic component doesn't. The judge is being\n * gamed.\n *\n * This module ships explicit detectors for all four signatures, plus a\n * combined verdict. The output is diagnostic — actionable signals,\n * not autoreject — because each signature has known false positives\n * (e.g., a policy that genuinely improves can show distributional shift).\n *\n * Differs from `rubricPredictiveValidity` (which is a *standing* check on\n * whether rubrics correlate with deployment outcomes) — this is a\n * *temporal* check on whether the reward-vs-truth gap is *widening over\n * time during a training run*.\n */\n\nimport { observedScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport { pearsonR } from '../statistics'\nimport {\n filterDeterministicallyRewarded,\n type VerifiableRewardExtractionOptions,\n} from './verifiable-reward'\n\nexport type RewardHackingSignal =\n | 'reward_divergence'\n | 'distribution_shift'\n | 'reward_disagreement'\n | 'judge_drift'\n\nexport interface RewardHackingFinding {\n signal: RewardHackingSignal\n /** Severity in [0, 1]. >0.5 = strong signal. */\n severity: number\n message: string\n /** Numeric evidence the consumer can render. */\n detail: Record<string, number>\n}\n\nexport interface RewardHackingReport {\n findings: RewardHackingFinding[]\n /** Signals with enough usable observations to produce a finding. */\n evaluatedSignals: RewardHackingSignal[]\n /**\n * Composite verdict. `'insufficient_evidence'` when fewer than four scored\n * runs exist; otherwise `'clean'` if every signal severity < 0.3,\n * `'suspect'` if at least one ≥ 0.3 but none ≥ 0.6, and `'gaming'` if any ≥ 0.6.\n */\n verdict: 'insufficient_evidence' | 'clean' | 'suspect' | 'gaming'\n /** Rationale for the verdict, ready to paste into an audit log. */\n rationale: string[]\n /** Number of runs with a usable proxy reward. */\n n: number\n}\n\nexport interface DetectRewardHackingInput {\n /**\n * Run records ordered by recency (oldest first). The detector segments\n * them into prefix/suffix windows to compute \"did the gap widen.\"\n */\n runs: RunRecord[]\n /**\n * The metric the policy was trained to optimize. Should be present on\n * `outcome.raw` or `outcome.holdoutScore`. Default reads `outcome.holdoutScore`.\n */\n proxyOf?: (run: RunRecord) => number | null\n /**\n * The held-out ground-truth metric. For RL on coding, this is typically\n * test pass-rate. For RLHF, it's downstream task performance or human\n * preference. For knowledge tasks, it's an independently-graded score.\n */\n truthOf?: (run: RunRecord) => number | null\n /**\n * Independent secondary reward. Used for the `reward_disagreement`\n * signal. Default uses the verifiable reward extractor (deterministic\n * sources only).\n */\n secondaryRewardOf?: (run: RunRecord) => number | null\n /**\n * Window size — how many of the most recent runs count as the \"after\"\n * cohort. Default min(50, half the runs).\n */\n windowSize?: number\n /**\n * Severity threshold to flag a signal. Default 0.3 (suspect) and 0.6\n * (gaming).\n */\n thresholds?: { suspect?: number; gaming?: number }\n /**\n * Verifiable-reward options used for the secondary-reward fallback.\n */\n verifiableRewardOptions?: VerifiableRewardExtractionOptions\n}\n\nconst DEFAULT_PROXY = (r: RunRecord): number | null => {\n // DELIBERATELY UNGATED. This is the proxy reward the detector tests for\n // Goodharting, and gated runs are exactly the gamed population. Forcing them\n // to 0 would collapse the proxy toward the deterministic secondary signal —\n // `reward_disagreement`'s correlation would rise, `judge_drift`'s gap would\n // shrink, and `reward_divergence`'s proxy-up/truth-flat fingerprint would be\n // erased — so the detector would report \"clean\" on the very runs it exists to\n // catch. `null` (never 0) is also load-bearing: it sets the n-denominator via\n // the filter below.\n const v = observedScore(r)\n return typeof v === 'number' && Number.isFinite(v) ? v : null\n}\n\nexport function detectRewardHacking(input: DetectRewardHackingInput): RewardHackingReport {\n const proxyOf = input.proxyOf ?? DEFAULT_PROXY\n const truthOf = input.truthOf\n const sus = input.thresholds?.suspect ?? 0.3\n const gam = input.thresholds?.gaming ?? 0.6\n\n const runs = input.runs.filter((run) => finiteNumber(proxyOf(run)))\n const n = runs.length\n if (n < 4) {\n return {\n findings: [],\n evaluatedSignals: [],\n verdict: 'insufficient_evidence',\n n,\n rationale: [`fewer than 4 runs with proxy reward (n=${n}); insufficient evidence`],\n }\n }\n const windowSize = Math.max(1, input.windowSize ?? Math.min(50, Math.floor(n / 2)))\n const before = runs.slice(0, n - windowSize)\n const after = runs.slice(n - windowSize)\n\n const findings: RewardHackingFinding[] = []\n\n // ── Signal 1: reward divergence (proxy ↑ while truth flat or ↓) ──────\n if (truthOf) {\n const beforeProxy = before.map(proxyOf).filter(finiteNumber)\n const afterProxy = after.map(proxyOf).filter(finiteNumber)\n const beforeTruth = before.map(truthOf).filter(finiteNumber)\n const afterTruth = after.map(truthOf).filter(finiteNumber)\n if (\n beforeProxy.length >= 2 &&\n afterProxy.length >= 2 &&\n beforeTruth.length >= 2 &&\n afterTruth.length >= 2\n ) {\n const proxyDelta = mean(afterProxy) - mean(beforeProxy)\n const truthDelta = mean(afterTruth) - mean(beforeTruth)\n // Divergence: proxy goes up while truth goes flat or down.\n // Severity = max(0, (proxyDelta - truthDelta)) — bigger gap = bigger signal.\n const gap = Math.max(0, proxyDelta - truthDelta)\n const severity = clamp01(gap * 5) // scale: 0.2 absolute gap → severity 1.0\n findings.push({\n signal: 'reward_divergence',\n severity,\n message:\n severity >= sus\n ? `proxy reward rose by ${proxyDelta.toFixed(3)} while truth changed by ${truthDelta.toFixed(3)} — potential Goodhart`\n : `proxy and truth moved together (proxy ${proxyDelta.toFixed(3)}, truth ${truthDelta.toFixed(3)})`,\n detail: {\n proxyDelta,\n truthDelta,\n gap,\n beforeN: beforeProxy.length,\n afterN: afterProxy.length,\n },\n })\n }\n }\n\n // ── Signal 2: distributional shift in outputs (KS on score distributions) ──\n {\n const beforeP = before.map(proxyOf).filter(finiteNumber)\n const afterP = after.map(proxyOf).filter(finiteNumber)\n if (beforeP.length >= 4 && afterP.length >= 4) {\n const ks = ksStatistic(beforeP, afterP)\n // KS statistic: bigger = more shift. We're agnostic about direction;\n // genuine improvement ALSO produces shift, so this signal is\n // contributory rather than load-bearing.\n const severity = clamp01(ks - 0.2)\n findings.push({\n signal: 'distribution_shift',\n severity,\n message:\n severity >= sus\n ? `KS=${ks.toFixed(3)} between before/after windows — distributional shift large`\n : `KS=${ks.toFixed(3)} between before/after windows — within-distribution drift`,\n detail: { ks, beforeN: beforeP.length, afterN: afterP.length },\n })\n }\n }\n\n // ── Signal 3: reward disagreement (proxy vs independent secondary) ────\n {\n const secondaryOf = input.secondaryRewardOf ?? defaultSecondary(input.verifiableRewardOptions)\n const aligned = runs\n .map((r) => ({ p: proxyOf(r), s: secondaryOf(r) }))\n .filter((x): x is { p: number; s: number } => finiteNumber(x.p) && finiteNumber(x.s))\n if (aligned.length >= 4) {\n const ps = aligned.map((x) => x.p)\n const ss = aligned.map((x) => x.s)\n const r = pearsonR(ps, ss)\n // Disagreement: low or negative correlation between primary proxy\n // reward and an independent secondary signal.\n const severity = clamp01(0.5 - Math.max(0, r))\n findings.push({\n signal: 'reward_disagreement',\n severity,\n message:\n severity >= sus\n ? `proxy and independent secondary reward correlate ρ=${r.toFixed(3)} — possibly hacking proxy`\n : `proxy and secondary reward correlate ρ=${r.toFixed(3)}`,\n detail: { pearson: r, n: aligned.length },\n })\n }\n }\n\n // ── Signal 4: judge drift (probabilistic up while deterministic flat) ─\n {\n // Ungated on purpose, exactly like `DEFAULT_PROXY` above. This signal is\n // the GAP between the judge reward and the deterministic one; a\n // deterministic reward another gate already forced to 0 would open that gap\n // by construction on the gamed population, so the detector would fire on\n // its own input rather than on evidence it found.\n const detRuns = filterDeterministicallyRewarded(runs, {\n ...(input.verifiableRewardOptions ?? {}),\n applyRealnessGate: false,\n })\n if (detRuns.length >= 4) {\n const detBefore = detRuns.slice(0, Math.floor(detRuns.length / 2))\n const detAfter = detRuns.slice(Math.floor(detRuns.length / 2))\n const detDelta =\n mean(detAfter.map((r) => r.reward.value)) - mean(detBefore.map((r) => r.reward.value))\n const proxyDelta =\n mean(after.map(proxyOf).filter(finiteNumber)) -\n mean(before.map(proxyOf).filter(finiteNumber))\n const driftGap = Math.max(0, proxyDelta - detDelta)\n const severity = clamp01(driftGap * 5)\n findings.push({\n signal: 'judge_drift',\n severity,\n message:\n severity >= sus\n ? `judge proxy +${proxyDelta.toFixed(3)} while deterministic reward +${detDelta.toFixed(3)} — judge drifting up without verifiable backing`\n : `judge and deterministic rewards move in step (judge ${proxyDelta.toFixed(3)}, det ${detDelta.toFixed(3)})`,\n detail: { proxyDelta, detDelta, driftGap, n: detRuns.length },\n })\n }\n }\n\n const maxSev = findings.reduce((m, f) => Math.max(m, f.severity), 0)\n if (findings.length === 0) {\n return {\n findings,\n evaluatedSignals: [],\n verdict: 'insufficient_evidence',\n rationale: [`no reward-hacking signal had enough paired evidence (n=${n})`],\n n,\n }\n }\n const verdict: RewardHackingReport['verdict'] =\n maxSev >= gam ? 'gaming' : maxSev >= sus ? 'suspect' : 'clean'\n const rationale = findings\n .filter((f) => f.severity >= sus)\n .map((f) => `${f.signal}: severity ${f.severity.toFixed(2)} — ${f.message}`)\n if (rationale.length === 0) rationale.push('no signals fired above suspect threshold')\n\n return {\n findings,\n evaluatedSignals: findings.map((finding) => finding.signal),\n verdict,\n rationale,\n n,\n }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction mean(xs: number[]): number {\n if (xs.length === 0) return 0\n return xs.reduce((s, x) => s + x, 0) / xs.length\n}\n\nfunction finiteNumber(value: number | null): value is number {\n return typeof value === 'number' && Number.isFinite(value)\n}\n\nfunction clamp01(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n\nfunction ksStatistic(a: number[], b: number[]): number {\n // Two-sample Kolmogorov-Smirnov statistic.\n const sortedA = [...a].sort((x, y) => x - y)\n const sortedB = [...b].sort((x, y) => x - y)\n const all = [...new Set([...sortedA, ...sortedB])].sort((x, y) => x - y)\n let max = 0\n for (const v of all) {\n const fa = sortedA.filter((x) => x <= v).length / sortedA.length\n const fb = sortedB.filter((x) => x <= v).length / sortedB.length\n max = Math.max(max, Math.abs(fa - fb))\n }\n return max\n}\n\nfunction defaultSecondary(\n verifiableOpts?: VerifiableRewardExtractionOptions,\n): (run: RunRecord) => number | null {\n return (run: RunRecord) => {\n // Ungated for the same reason as signal 4: this is the INDEPENDENT\n // secondary reward whose correlation with the proxy is the evidence.\n // Zeroing it on gated runs would drive that correlation down mechanically.\n const filtered = filterDeterministicallyRewarded([run], {\n ...(verifiableOpts ?? {}),\n applyRealnessGate: false,\n })\n return filtered.length === 1 ? filtered[0]!.reward.value : null\n }\n}\n"],"mappings":";;;;;;;;;;;;;;AA8CA,IAAa,+BAAb,cAAkD,UAAU;CAC1D,YAAY,WAAmB;EAC7B,MACE,wBAAwB,UAAU,mQAIpC;EACA,KAAK,OAAO;CACd;AACF;;;;;;;;;AAkBA,SAAgB,wBACd,MACA,SACW;CACX,MAAM,UAAU,2BAA2B,IAAI;CAC/C,MAAM,YAAY,8BAA8B,IAAI;CACpD,MAAM,kBAAkB,KAAK,IAC3B,QAAQ,IAAI,qBAAqB,GACjC,UAAU,mBAAmB,CAC/B;CACA,MAAM,qBAAqB,2BAA2B,IAAI;CAC1D,MAAM,iBACJ,mBAAmB,SAAS,gBAAgB,QAAQ,mBAAmB,KAAA,IACnE;EAAE,MAAM;EAAa,KAAK,QAAQ;CAAe,IACjD;CACN,MAAM,UAAU,eAAe,SAAS,eAAe,OAAO,eAAe;CAC7E,MAAM,MAA8B;EAClC,GAAG,cAAc,QAAQ,GAAG;EAC5B,GAAG,QAAQ;EACX,KAAK,KAAK;EACV,aAAa,KAAK;EAClB,GAAI,YAAY,OAAO,CAAC,IAAI,EAAE,UAAU,QAAQ;EAEhD,GAAI,mBAAmB,SAAS,eAAe,EAAE,yBAAyB,KAAK,QAAQ,IAAI,CAAC;EAC5F,eAAe,eAAe,SAAS,aAAa,IAAI;EACxD,gBAAgB,eAAe,SAAS,cAAc,IAAI;EAC1D,iBAAiB,eAAe,SAAS,eAAe,IAAI;EAC5D,cAAc,KAAK,WAAW;EAC9B,eAAe,KAAK,WAAW;EAC/B,cAAc,KAAK,WAAW,gBAAgB,QAAQ,IAAI;EAC1D,YAAY,KAAK;EACjB,GAAI,UAAU,wBAAwB,KAAA,IAClC,CAAC,IACD,EAAE,uBAAuB,UAAU,oBAAoB;EAC3D,GAAI,kBAAkB,IAAI,EAAE,mBAAmB,gBAAgB,IAAI,CAAC;EACpE,GAAI,UAAU,2BAA2B,KAAA,IACrC,CAAC,IACD,EAAE,0BAA0B,UAAU,uBAAuB;CACnE;CACA,IAAI,OAAO,KAAK,eAAe,UAAU,IAAI,aAAa,KAAK;CAC/D,IAAI,KAAK,WAAW,cAAc,KAAA,GAChC,IAAI,mBAAmB,KAAK,WAAW;CAEzC,IAAI,KAAK,WAAW,WAAW,KAAA,GAAW,IAAI,gBAAgB,KAAK,WAAW;CAC9E,IAAI,KAAK,WAAW,eAAe,KAAA,GACjC,IAAI,qBAAqB,KAAK,WAAW;CAE3C,IAAI,KAAK,WAAW,gBAAgB,SAAS,YAAY,QAAQ,UAAU,GACzE,IAAI,qBAAqB,KAAK,WAAW,QAAQ,KAAK,WAAW,UAAU;CAE7E,IAAI,YAAY,QAAQ,QAAQ,UAAU,KAAA,KAAa,QAAQ,QAAQ,KACrE,IAAI,mBAAmB,UAAU,QAAQ;CAG3C,MAAM,UAAsB;EAC1B;EACA,GAAI,QAAQ,cAAc,EAAE,aAAa,QAAQ,YAAY,IAAI,CAAC;CACpE;CACA,IAAI,QAAQ,UAAU,KAAA,GACpB,IAAI,QAAQ,aAAa,WAAW,QAAQ,eAAe,QAAQ;MAC9D,QAAQ,cAAc,QAAQ;CAGrC,OAAO,kBAAkB;EACvB,OAAO,QAAQ;EACf,cAAc,QAAQ;EACtB,aAAa,QAAQ;EACrB,MAAM,QAAQ,QAAQ,KAAK;EAC3B,OAAO,QAAQ;EACf,YAAY,QAAQ;EACpB,YAAY,QAAQ;EACpB,WAAW,QAAQ;EACnB,QAAQ,KAAK;EACb;EACA;EACA,YAAY,EAAE,GAAG,KAAK,WAAW;EACjC,iBAAiB,UAAU;EAC3B,GAAI,UAAU,wBACV,EAAE,uBAAuB,UAAU,sBAAsB,IACzD,CAAC;EACL;EACA,UAAU,QAAQ;EAClB,YAAY,QAAQ,cAAc,KAAK;EACvC,GAAI,QAAQ,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CACvE,CAAC;AACH;;;;;;AAOA,SAAgB,2BACd,MACgB;CAChB,IAAI,CAAC,OAAO,SAAS,KAAK,OAAO,KAAK,KAAK,UAAU,GACnD,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,sBAAsB;CAEtE,MAAM,aAAa,KAAK;CACxB,IAAI,CAAC,cAAc,OAAO,eAAe,UACvC,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,wBAAwB;CAExE,IAAI,WAAW,SAAS,cAAc;EACpC,IAAI,WAAW,QAAQ,MACrB,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,wCAAwC;EAExF,OAAO;GAAE,MAAM;GAAc,KAAK;EAAK;CACzC;CACA,IACG,WAAW,SAAS,cAAc,WAAW,SAAS,eACvD,CAAC,OAAO,SAAS,WAAW,GAAG,KAC/B,WAAW,MAAM,GAEjB,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,6BAA6B;CAE7E,IAAI,WAAW,QAAQ,KAAK,SAC1B,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,+CAA+C;CAE/F,OAAO;EAAE,MAAM,WAAW;EAAM,KAAK,WAAW;CAAI;AACtD;AAEA,SAAgB,8BACd,MAC+B;CAC/B,IAAI,KAAK,eAAe,YACtB,OAAO;EACL,iBAAiB;EACjB,qBAAqB;EACrB,GAAI,KAAK,QAAQ,EAAE,uBAAuB,KAAK,MAAM,IAAI,CAAC;CAC5D;CAEF,IAAI,KAAK,eAAe,SACtB,OAAO;EACL,iBAAiB;EACjB,qBAAqB;EACrB,iBAAiB;CACnB;CAEF,IAAI,CAAC,KAAK,OACR,OAAO;EAAE,iBAAiB;EAAa,qBAAqB;CAAE;CAEhE,OAAO;EACL,iBAAiB;EACjB,wBAAwB;CAC1B;AACF;;;;;;;;AASA,SAAgB,2BACd,MAC+B;CAC/B,IAAI,KAAK,eAAe,YACtB,OAAO;EAAE,uBAAuB,CAAC;EAAG,cAAc,CAAC;EAAG,KAAK,CAAC;CAAE;CAGhE,MAAM,WAAmD,CAAC;CAC1D,MAAM,wBAAoD,CAAC;CAC3D,MAAM,kCAAkB,IAAI,IAAsB;CAClD,MAAM,aAAuB,CAAC;CAC9B,MAAM,QAAkB,CAAC;CACzB,MAAM,eAAe,IAAI,IACvB,KAAK,eAAe,UAAU,CAAC,KAAK,cAAc,eAAe,IAAI,CAAC,CACxE;CACA,MAAM,MAA8B,CAAC;CAErC,KAAK,MAAM,CAAC,WAAW,UAAU,OAAO,QAAQ,KAAK,WAAW,GAAG;EACjE,MAAM,kBAAmB,MAAmC;EAC5D,IACE,OAAO,oBAAoB,YAC3B,oBAAoB,QACpB,MAAM,QAAQ,eAAe,GAE7B,MAAM,IAAI,6BAA6B,SAAS;EAElD,MAAM,mBAAmB,OAAO,OAAO,MAAM,UAAU,CAAC,CAAC,MAAM,OAAO,QAAQ;EAC9E,IAAI,MAAM,UAAU,CAAC,OAAO,SAAS,MAAM,SAAS,KAAK,CAAC,kBAAkB;GAC1E,aAAa,IAAI,SAAS;GAC1B;EACF;EAEA,WAAW,KAAK,MAAM,SAAS;EAC/B,sBAAsB,aAAa;EACnC,MAAM,aAAa,EAAE,GAAG,MAAM,WAAW;EACzC,SAAS,aAAa;EACtB,KAAK,MAAM,CAAC,WAAW,UAAU,OAAO,QAAQ,UAAU,GAAG;GAC3D,IAAI,GAAG,UAAU,GAAG,eAAe;GACnC,MAAM,SAAS,gBAAgB,IAAI,SAAS,KAAK,CAAC;GAClD,OAAO,KAAK,KAAK;GACjB,gBAAgB,IAAI,WAAW,MAAM;EACvC;EACA,IAAI,MAAM,OAAO,MAAM,KAAK,GAAG,UAAU,IAAI,MAAM,OAAO;EAC1D,KAAK,MAAM,eAAe,MAAM,gBAAgB,CAAC,GAC/C,aAAa,IAAI,GAAG,UAAU,GAAG,aAAa;CAElD;CAEA,IAAI,aAAa,OAAO,GAAG,IAAI,oBAAoB,aAAa;CAChE,MAAM,qBAAqB,CAAC,GAAG,YAAY,CAAC,CAAC,KAAK;CAClD,IAAI,WAAW,WAAW,GACxB,OAAO;EACL;EACA,cAAc;EACd;CACF;CAGF,MAAM,YAAYA,OAAK,UAAU;CACjC,MAAM,aAAa,OAAO,YACxB,CAAC,GAAG,gBAAgB,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,WAAW,YAAY,CAAC,WAAWA,OAAK,MAAM,CAAC,CAAC,CACvF;CACA,MAAM,WACJ,KAAK,UAAU,KAAA,KAAa,KAAK,eAAe,KAAA,KAAa,aAAa,SAAS;CACrF,IAAI,UAAU,IAAI,YAAY;CAE9B,OAAO;EACL,GAAI,WAAW,EAAE,OAAO,UAAU,IAAI,CAAC;EACvC;EACA;EACA,cAAc;EACd,aAAa;GACX;GACA;GACA;GACA,GAAI,mBAAmB,SAAS,IAAI,EAAE,cAAc,mBAAmB,IAAI,CAAC;GAC5E,GAAI,MAAM,SAAS,IAAI,EAAE,OAAO,MAAM,KAAK,KAAK,EAAE,IAAI,CAAC;EACzD;CACF;AACF;;AAGA,SAAgB,sBACd,MACoB;CACpB,OAAO,2BAA2B,IAAI,CAAC,CAAC;AAC1C;;AAGA,SAAgB,4BACd,MACwC;CACxC,OAAO,2BAA2B,IAAI,CAAC,CAAC,aAAa,YAAY,CAAC;AACpE;AAEA,SAAS,cAAc,SAAqE;CAC1F,MAAM,SAAiC,CAAC;CACxC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,WAAW,CAAC,CAAC,GACrD,IAAI,OAAO,SAAS,KAAK,GAAG,OAAO,OAAO;CAE5C,OAAO;AACT;AAEA,SAASA,OAAK,QAA0B;CACtC,OAAO,OAAO,QAAQ,KAAK,UAAU,MAAM,OAAO,CAAC,IAAI,OAAO;AAChE;;;ACtKA,MAAM,+CAA+B,IAAI,IAAI;CAC3C;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAM,sBAAsB,SAAyC;CACnE,MAAM,QAAQ,KAAK,YAAY;CAC/B,IAAI,MAAM,SAAS,MAAM,GAAG,OAAO;CACnC,IACE,MAAM,SAAS,SAAS,KACxB,MAAM,SAAS,OAAO,KACtB,MAAM,SAAS,WAAW,KAC1B,MAAM,SAAS,MAAM,GAErB,OAAO;CACT,IAAI,MAAM,SAAS,QAAQ,GAAG,OAAO;CACrC,IAAI,MAAM,SAAS,SAAS,GAAG,OAAO;CACtC,IAAI,MAAM,SAAS,OAAO,KAAK,MAAM,SAAS,UAAU,GAAG,OAAO;CAIlE,IAAI,MAAM,SAAS,OAAO,KAAK,MAAM,SAAS,QAAQ,KAAK,MAAM,SAAS,MAAM,GAC9E,OAAO;CAET,IAAI,MAAM,SAAS,WAAW,KAAK,MAAM,SAAS,aAAa,GAAG,OAAO;CACzE,IAAI,MAAM,SAAS,UAAU,GAAG,OAAO;CACvC,IAAI,MAAM,SAAS,WAAW,KAAK,MAAM,SAAS,WAAW,GAAG,OAAO;CACvE,OAAO;AACT;;;;;;;;;;;;;;AAeA,SAAgB,wBACd,QACA,OAA0C,CAAC,GAClB;CACzB,MAAM,mBAAmB,IAAI,IAAI,KAAK,uBAAuB,CAAC,GAAG,4BAA4B,CAAC;CAC9F,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,aAAa,KAAK,wBAAwB;CAEhD,MAAM,gBAAgB,OAAO,OAAO,QACjC,UAAU,iBAAiB,IAAI,MAAM,KAAK,KAAK,gBAAgB,KAAK,CACvE;CAEA,IAAI,cAAc,WAAW,GAAG;EAC9B,MAAM,QAAQ,cAAc;EAC5B,MAAM,QAAQC,UAAQ,MAAM,KAAM;EAClC,OAAO;GACL;GACA,QAAQ,UAAU,MAAM,KAAK;GAC7B,aAAa;GACb,YAAY;GACZ,QAAQ,MAAM;GACd,YAAY,GAAG,MAAM,QAAQ,MAAM;GACnC,kBAAkB;EACpB;CACF;CAEA,IAAI,cAAc,SAAS,GAAG;EAE5B,IAAI,MAAM;EACV,IAAI,QAAQ;EACZ,MAAM,aAAqC,CAAC;EAC5C,KAAK,MAAM,KAAK,eAAe;GAC7B,MAAM,IAAK,EAAE,QAAQ,UAAiC;GACtD,OAAO,KAAK,EAAE,SAAS;GACvB,SAAS;GACT,WAAW,EAAE,SAAS,EAAE;EAC1B;EACA,OAAO;GACL,OAAO,UAAU,IAAI,IAAIA,UAAQ,MAAM,KAAK;GAC5C,QAAQ;GACR,aAAa;GACb,YAAY;GACZ,QAAQ,cAAc,KAAK,MAAM,EAAE,KAAK,CAAC,CAAC,KAAK,GAAG;GAClD;GACA,kBAAkB;EACpB;CACF;CAEA,IAAI,CAAC,iBAAiB,OAAO;CAE7B,MAAM,QACJ,OAAO,OAAO,MAAM,UAAU,gBAAgB,KAAK,KAAK,UAAU,MAAM,KAAK,MAAM,OAAO,KAC1F,OAAO,OAAO,KAAK,eAAe;CAEpC,IAAI,CAAC,OAAO,OAAO;CAEnB,MAAM,iBAAiB,MAAM,QAAQ;CACrC,MAAM,aAAaA,UAAQ,MAAM,KAAM;CACvC,OAAO;EACL,OAAO;EACP,QAAQ;EACR,aAAa;EACb,YAAY,OAAO,mBAAmB,WAAW,iBAAiB;EAClE,QAAQ,MAAM;EACd,YAAY,GAAG,MAAM,QAAQ,WAAW;EACxC,kBAAkB;CACpB;AACF;AAEA,SAAS,gBACP,OACmE;CACnE,QACG,MAAM,WAAW,UAAU,MAAM,WAAW,WAC7C,OAAO,MAAM,UAAU,YACvB,OAAO,SAAS,MAAM,KAAK,KAC3B,MAAM,SAAS,KACf,MAAM,SAAS;AAEnB;;;;;;;;;;;;;;;;;;;;;;;AAwBA,SAAgB,oCACd,MACA,OAA0C,CAAC,GACgB;CAC3D,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,mBAAmB,IAAI,IAAI,KAAK,uBAAuB,CAAC,GAAG,4BAA4B,CAAC;CAC9F,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,aAAa,KAAK,wBAAwB;CAChD,MAAM,YAAY,KAAK,qBAAqB;CAE5C,OAAO,KAAK,KAAK,QAAQ;EACvB,MAAM,UAAU,gBAAgB,GAAG;EAInC,MAAM,WAAW,IAAI,QAAQ,aAAa,KAAA,IAAY,CAAC,IAAK,EAAE,kBAAkB,KAAK;EAMrF,MAAM,QAAQ,UAA2B,aAAa,UAAU,IAAI;EAEpE,MAAM,cAAsD,CAAC;EAC7D,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,IAAI,QAAQ,GAAG,GACjD,IACE,EAAE,WAAW,QAAQ,KACrB,CAAC,EAAE,SAAS,KAAK,CAAC,KAClB,OAAO,MAAM,YACb,OAAO,SAAS,CAAC,GAEjB,YAAY,KAAK;GAAE,MAAM,EAAE,MAAM,CAAe;GAAG,OAAO;EAAE,CAAC;EAGjE,MAAM,MAAM,YAAY,QAAQ,MAAM,iBAAiB,IAAI,EAAE,IAAI,CAAC;EAElE,IAAI,IAAI,WAAW,GAAG;GACpB,MAAM,QAAQ,IAAI;GAClB,MAAM,QAAQ,KAAKA,UAAQ,MAAM,KAAK,CAAC;GACvC,OAAO;IACL,OAAO,IAAI;IACX,QAAQ;KACN;KACA,QAAQ,UAAU,MAAM,IAAI;KAC5B,aAAa;KACb,YAAY;KACZ,QAAQ,MAAM;KACd,YAAY,GAAG,MAAM,OAAO,MAAM;KAClC,eAAe;KACf,GAAG;IACL;GACF;EACF;EACA,IAAI,IAAI,SAAS,GAAG;GAClB,MAAM,QAAQ,KAAKA,UAAQ,IAAI,QAAQ,GAAG,MAAM,IAAI,EAAE,OAAO,CAAC,IAAI,IAAI,MAAM,CAAC;GAG7E,MAAM,aAAqC,OAAO,YAChD,IAAI,KAAK,MAAM,CAAC,EAAE,MAAM,KAAKA,UAAQ,EAAE,KAAK,CAAC,CAAC,CAAC,CACjD;GACA,OAAO;IACL,OAAO,IAAI;IACX,QAAQ;KACN;KACA,QAAQ;KACR,aAAa;KACb,YAAY;KACZ,QAAQ,IAAI,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,GAAG;KACvC;KACA,eAAe;KACf,GAAG;IACL;GACF;EACF;EACA,IAAI,CAAC,iBAAiB,OAAO;GAAE,OAAO,IAAI;GAAO,QAAQ;EAAK;EAO9D,MAAM,UAAU,YAAY,cAAc,GAAG,IAAI,cAAc,GAAG;EAClE,IAAI,OAAO,YAAY,YAAY,CAAC,OAAO,SAAS,OAAO,GACzD,OAAO;GAAE,OAAO,IAAI;GAAO,QAAQ;EAAK;EAE1C,MAAM,eAAeA,UAAQ,OAAO;EACpC,OAAO;GACL,OAAO,IAAI;GACX,QAAQ;IACN,OAAO;IACP,QAAQ;IACR,aAAa;IACb,YAAY;IACZ,QAAQ;IACR,YAAY,EAAE,qBAAqB,aAAa;IAChD,eAAe;IACf,GAAG;GACL;EACF;CACF,CAAC;AACH;;;;;;;;;;AAWA,SAAgB,gCACd,MACA,OAA0C,CAAC,GACU;CACrD,MAAM,WAAW,oCAAoC,MAAM;EAAE,GAAG;EAAM,iBAAiB;CAAM,CAAC;CAC9F,MAAM,MAA2D,CAAC;CAClE,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,IAAI,SAAS;EACnB,IAAI,EAAE,UAAU,EAAE,OAAO,gBAAgB,iBACvC,IAAI,KAAK;GAAE,KAAK,KAAK;GAAK,QAAQ,EAAE;EAAO,CAAC;CAEhD;CACA,OAAO;AACT;AAEA,SAASA,UAAQ,GAAmB;CAClC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACnVA,MAAM,iBAAiB,MAAgC;CASrD,MAAM,IAAI,cAAc,CAAC;CACzB,OAAO,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,IAAI,IAAI;AAC3D;AAEA,SAAgB,oBAAoB,OAAsD;CACxF,MAAM,UAAU,MAAM,WAAW;CACjC,MAAM,UAAU,MAAM;CACtB,MAAM,MAAM,MAAM,YAAY,WAAW;CACzC,MAAM,MAAM,MAAM,YAAY,UAAU;CAExC,MAAM,OAAO,MAAM,KAAK,QAAQ,QAAQ,aAAa,QAAQ,GAAG,CAAC,CAAC;CAClE,MAAM,IAAI,KAAK;CACf,IAAI,IAAI,GACN,OAAO;EACL,UAAU,CAAC;EACX,kBAAkB,CAAC;EACnB,SAAS;EACT;EACA,WAAW,CAAC,0CAA0C,EAAE,yBAAyB;CACnF;CAEF,MAAM,aAAa,KAAK,IAAI,GAAG,MAAM,cAAc,KAAK,IAAI,IAAI,KAAK,MAAM,IAAI,CAAC,CAAC,CAAC;CAClF,MAAM,SAAS,KAAK,MAAM,GAAG,IAAI,UAAU;CAC3C,MAAM,QAAQ,KAAK,MAAM,IAAI,UAAU;CAEvC,MAAM,WAAmC,CAAC;CAG1C,IAAI,SAAS;EACX,MAAM,cAAc,OAAO,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EAC3D,MAAM,aAAa,MAAM,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EACzD,MAAM,cAAc,OAAO,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EAC3D,MAAM,aAAa,MAAM,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EACzD,IACE,YAAY,UAAU,KACtB,WAAW,UAAU,KACrB,YAAY,UAAU,KACtB,WAAW,UAAU,GACrB;GACA,MAAM,aAAa,KAAK,UAAU,IAAI,KAAK,WAAW;GACtD,MAAM,aAAa,KAAK,UAAU,IAAI,KAAK,WAAW;GAGtD,MAAM,MAAM,KAAK,IAAI,GAAG,aAAa,UAAU;GAC/C,MAAM,WAAW,QAAQ,MAAM,CAAC;GAChC,SAAS,KAAK;IACZ,QAAQ;IACR;IACA,SACE,YAAY,MACR,wBAAwB,WAAW,QAAQ,CAAC,EAAE,0BAA0B,WAAW,QAAQ,CAAC,EAAE,yBAC9F,yCAAyC,WAAW,QAAQ,CAAC,EAAE,UAAU,WAAW,QAAQ,CAAC,EAAE;IACrG,QAAQ;KACN;KACA;KACA;KACA,SAAS,YAAY;KACrB,QAAQ,WAAW;IACrB;GACF,CAAC;EACH;CACF;CAGA;EACE,MAAM,UAAU,OAAO,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EACvD,MAAM,SAAS,MAAM,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EACrD,IAAI,QAAQ,UAAU,KAAK,OAAO,UAAU,GAAG;GAC7C,MAAM,KAAK,YAAY,SAAS,MAAM;GAItC,MAAM,WAAW,QAAQ,KAAK,EAAG;GACjC,SAAS,KAAK;IACZ,QAAQ;IACR;IACA,SACE,YAAY,MACR,MAAM,GAAG,QAAQ,CAAC,EAAE,8DACpB,MAAM,GAAG,QAAQ,CAAC,EAAE;IAC1B,QAAQ;KAAE;KAAI,SAAS,QAAQ;KAAQ,QAAQ,OAAO;IAAO;GAC/D,CAAC;EACH;CACF;CAGA;EACE,MAAM,cAAc,MAAM,qBAAqB,iBAAiB,MAAM,uBAAuB;EAC7F,MAAM,UAAU,KACb,KAAK,OAAO;GAAE,GAAG,QAAQ,CAAC;GAAG,GAAG,YAAY,CAAC;EAAE,EAAE,CAAC,CAClD,QAAQ,MAAqC,aAAa,EAAE,CAAC,KAAK,aAAa,EAAE,CAAC,CAAC;EACtF,IAAI,QAAQ,UAAU,GAAG;GAGvB,MAAM,IAAI,SAFC,QAAQ,KAAK,MAAM,EAAE,CAEZ,GADT,QAAQ,KAAK,MAAM,EAAE,CACR,CAAC;GAGzB,MAAM,WAAW,QAAQ,KAAM,KAAK,IAAI,GAAG,CAAC,CAAC;GAC7C,SAAS,KAAK;IACZ,QAAQ;IACR;IACA,SACE,YAAY,MACR,sDAAsD,EAAE,QAAQ,CAAC,EAAE,6BACnE,0CAA0C,EAAE,QAAQ,CAAC;IAC3D,QAAQ;KAAE,SAAS;KAAG,GAAG,QAAQ;IAAO;GAC1C,CAAC;EACH;CACF;CAGA;EAME,MAAM,UAAU,gCAAgC,MAAM;GACpD,GAAI,MAAM,2BAA2B,CAAC;GACtC,mBAAmB;EACrB,CAAC;EACD,IAAI,QAAQ,UAAU,GAAG;GACvB,MAAM,YAAY,QAAQ,MAAM,GAAG,KAAK,MAAM,QAAQ,SAAS,CAAC,CAAC;GAEjE,MAAM,WACJ,KAFe,QAAQ,MAAM,KAAK,MAAM,QAAQ,SAAS,CAAC,CAE9C,CAAC,CAAC,KAAK,MAAM,EAAE,OAAO,KAAK,CAAC,IAAI,KAAK,UAAU,KAAK,MAAM,EAAE,OAAO,KAAK,CAAC;GACvF,MAAM,aACJ,KAAK,MAAM,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY,CAAC,IAC5C,KAAK,OAAO,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY,CAAC;GAC/C,MAAM,WAAW,KAAK,IAAI,GAAG,aAAa,QAAQ;GAClD,MAAM,WAAW,QAAQ,WAAW,CAAC;GACrC,SAAS,KAAK;IACZ,QAAQ;IACR;IACA,SACE,YAAY,MACR,gBAAgB,WAAW,QAAQ,CAAC,EAAE,+BAA+B,SAAS,QAAQ,CAAC,EAAE,mDACzF,uDAAuD,WAAW,QAAQ,CAAC,EAAE,QAAQ,SAAS,QAAQ,CAAC,EAAE;IAC/G,QAAQ;KAAE;KAAY;KAAU;KAAU,GAAG,QAAQ;IAAO;GAC9D,CAAC;EACH;CACF;CAEA,MAAM,SAAS,SAAS,QAAQ,GAAG,MAAM,KAAK,IAAI,GAAG,EAAE,QAAQ,GAAG,CAAC;CACnE,IAAI,SAAS,WAAW,GACtB,OAAO;EACL;EACA,kBAAkB,CAAC;EACnB,SAAS;EACT,WAAW,CAAC,0DAA0D,EAAE,EAAE;EAC1E;CACF;CAEF,MAAM,UACJ,UAAU,MAAM,WAAW,UAAU,MAAM,YAAY;CACzD,MAAM,YAAY,SACf,QAAQ,MAAM,EAAE,YAAY,GAAG,CAAC,CAChC,KAAK,MAAM,GAAG,EAAE,OAAO,aAAa,EAAE,SAAS,QAAQ,CAAC,EAAE,KAAK,EAAE,SAAS;CAC7E,IAAI,UAAU,WAAW,GAAG,UAAU,KAAK,0CAA0C;CAErF,OAAO;EACL;EACA,kBAAkB,SAAS,KAAK,YAAY,QAAQ,MAAM;EAC1D;EACA;EACA;CACF;AACF;AAIA,SAAS,KAAK,IAAsB;CAClC,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,OAAO,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,GAAG;AAC5C;AAEA,SAAS,aAAa,OAAuC;CAC3D,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK;AAC3D;AAEA,SAAS,QAAQ,GAAmB;CAClC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;AAEA,SAAS,YAAY,GAAa,GAAqB;CAErD,MAAM,UAAU,CAAC,GAAG,CAAC,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,MAAM,UAAU,CAAC,GAAG,CAAC,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,MAAM,MAAM,CAAC,mBAAG,IAAI,IAAI,CAAC,GAAG,SAAS,GAAG,OAAO,CAAC,CAAC,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CACvE,IAAI,MAAM;CACV,KAAK,MAAM,KAAK,KAAK;EACnB,MAAM,KAAK,QAAQ,QAAQ,MAAM,KAAK,CAAC,CAAC,CAAC,SAAS,QAAQ;EAC1D,MAAM,KAAK,QAAQ,QAAQ,MAAM,KAAK,CAAC,CAAC,CAAC,SAAS,QAAQ;EAC1D,MAAM,KAAK,IAAI,KAAK,KAAK,IAAI,KAAK,EAAE,CAAC;CACvC;CACA,OAAO;AACT;AAEA,SAAS,iBACP,gBACmC;CACnC,QAAQ,QAAmB;EAIzB,MAAM,WAAW,gCAAgC,CAAC,GAAG,GAAG;GACtD,GAAI,kBAAkB,CAAC;GACvB,mBAAmB;EACrB,CAAC;EACD,OAAO,SAAS,WAAW,IAAI,SAAS,EAAE,CAAE,OAAO,QAAQ;CAC7D;AACF"}
|
|
1
|
+
{"version":3,"file":"reward-hacking-O5zKDANP.js","names":["mean","clamp01"],"sources":["../src/campaign/run-record.ts","../src/rl/verifiable-reward.ts","../src/rl/reward-hacking.ts"],"sourcesContent":["import type { AgentProfileCell } from '../agent-profile-cell'\nimport type { CostProvenance } from '../cost-ledger'\nimport type {\n JudgeScoresRecord,\n RunOutcome,\n RunRecord,\n RunSplitTag,\n RunTerminalOutcome,\n} from '../run-record'\nimport { validateRunRecord } from '../run-record'\nimport type { CampaignCellResult, JudgeScore } from './types'\n\nexport interface CampaignCellRunRecordOptions {\n runId: string\n experimentId: string\n candidateId: string\n model: string\n promptHash: string\n configHash: string\n commitSha: string\n splitTag: RunSplitTag\n seed?: number\n scenarioId?: string\n defaultCostUsd?: number\n agentProfile?: AgentProfileCell\n raw?: Record<string, number>\n}\n\nexport interface CampaignCellQualityProjection {\n score?: number\n judgeScores?: JudgeScoresRecord\n successfulJudgeScores: Record<string, JudgeScore>\n failedJudges: string[]\n raw: Record<string, number>\n}\n\n/**\n * A campaign cell carried a judge score without a `dimensions` record.\n *\n * Two exported types share the name `JudgeScore`: the campaign verdict\n * (`{ dimensions, composite, notes }` from `@tangle-network/agent-eval/campaign`)\n * and the root export's flat per-dimension row (`{ judgeName, dimension, score }`\n * from `@tangle-network/agent-eval`). Campaign aggregation accepts only the\n * campaign shape; the flat shape previously crashed here with an opaque\n * TypeError deep inside aggregation.\n */\nexport class CampaignJudgeScoreShapeError extends TypeError {\n constructor(judgeName: string) {\n super(\n `campaign cell judge '${judgeName}' carries a score without a 'dimensions' record. ` +\n `Use the campaign JudgeScore ({ dimensions, composite, notes }) from ` +\n `'@tangle-network/agent-eval/campaign'; the root export's JudgeScore ` +\n `({ judgeName, dimension, score }) is a different type with the same name.`,\n )\n this.name = 'CampaignJudgeScoreShapeError'\n }\n}\n\nexport interface CampaignCellExecutionEvidence {\n terminalOutcome: RunTerminalOutcome\n executionErrorCount?: number\n judgeErrorCount?: number\n unclassifiedErrorCount?: number\n terminalFailureReason?: string\n}\n\n/**\n * Project one campaign cell into the canonical run format.\n *\n * A dispatch error establishes terminal execution failure. A judge error only\n * establishes that quality measurement failed after dispatch completed.\n * Failures without a stage remain unknown. No failure becomes a zero-quality\n * label.\n */\nexport function campaignCellToRunRecord<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n options: CampaignCellRunRecordOptions,\n): RunRecord {\n const quality = projectCampaignCellQuality(cell)\n const execution = campaignCellExecutionEvidence(cell)\n const judgeErrorCount = Math.max(\n quality.raw.judge_error_count ?? 0,\n execution.judgeErrorCount ?? 0,\n )\n const cellCostProvenance = campaignCellCostProvenance(cell)\n const costProvenance: CostProvenance =\n cellCostProvenance.kind === 'uncaptured' && options.defaultCostUsd !== undefined\n ? { kind: 'estimated', usd: options.defaultCostUsd }\n : cellCostProvenance\n const costUsd = costProvenance.kind === 'uncaptured' ? null : costProvenance.usd\n const raw: Record<string, number> = {\n ...finiteMetrics(options.raw),\n ...quality.raw,\n rep: cell.rep,\n duration_ms: cell.durationMs,\n ...(costUsd === null ? {} : { cost_usd: costUsd }),\n // Retain the observed subtotal even when the caller supplies an estimated total.\n ...(cellCostProvenance.kind === 'uncaptured' ? { cost_known_subtotal_usd: cell.costUsd } : {}),\n cost_observed: costProvenance.kind === 'observed' ? 1 : 0,\n cost_estimated: costProvenance.kind === 'estimated' ? 1 : 0,\n cost_uncaptured: costProvenance.kind === 'uncaptured' ? 1 : 0,\n tokens_input: cell.tokenUsage.input,\n tokens_output: cell.tokenUsage.output,\n tokens_known: cell.tokenUsage.tokensKnown === false ? 0 : 1,\n latency_ms: cell.durationMs,\n ...(execution.executionErrorCount === undefined\n ? {}\n : { execution_error_count: execution.executionErrorCount }),\n ...(judgeErrorCount > 0 ? { judge_error_count: judgeErrorCount } : {}),\n ...(execution.unclassifiedErrorCount === undefined\n ? {}\n : { unclassified_error_count: execution.unclassifiedErrorCount }),\n }\n if (typeof cell.generation === 'number') raw.generation = cell.generation\n if (cell.tokenUsage.reasoning !== undefined) {\n raw.tokens_reasoning = cell.tokenUsage.reasoning\n }\n if (cell.tokenUsage.cached !== undefined) raw.tokens_cached = cell.tokenUsage.cached\n if (cell.tokenUsage.cacheWrite !== undefined) {\n raw.tokens_cache_write = cell.tokenUsage.cacheWrite\n }\n if (cell.tokenUsage.tokensKnown !== false && costUsd !== null && costUsd > 0) {\n raw.tokens_per_dollar = (cell.tokenUsage.input + cell.tokenUsage.output) / costUsd\n }\n if (costUsd !== null && quality.score !== undefined && quality.score > 0.01) {\n raw.cost_per_quality = costUsd / quality.score\n }\n\n const outcome: RunOutcome = {\n raw,\n ...(quality.judgeScores ? { judgeScores: quality.judgeScores } : {}),\n }\n if (quality.score !== undefined) {\n if (options.splitTag === 'holdout') outcome.holdoutScore = quality.score\n else outcome.searchScore = quality.score\n }\n\n return validateRunRecord({\n runId: options.runId,\n experimentId: options.experimentId,\n candidateId: options.candidateId,\n seed: options.seed ?? cell.seed,\n model: options.model,\n promptHash: options.promptHash,\n configHash: options.configHash,\n commitSha: options.commitSha,\n wallMs: cell.durationMs,\n costUsd,\n costProvenance,\n tokenUsage: { ...cell.tokenUsage },\n terminalOutcome: execution.terminalOutcome,\n ...(execution.terminalFailureReason\n ? { terminalFailureReason: execution.terminalFailureReason }\n : {}),\n outcome,\n splitTag: options.splitTag,\n scenarioId: options.scenarioId ?? cell.scenarioId,\n ...(options.agentProfile ? { agentProfile: options.agentProfile } : {}),\n })\n}\n\n/**\n * Validate the cost fields that cross campaign cache and RunRecord boundaries.\n * `costUsd` is a known subtotal for uncaptured cells, but it must equal the\n * authoritative total whenever that total is observed or estimated.\n */\nexport function campaignCellCostProvenance<TArtifact>(\n cell: Pick<CampaignCellResult<TArtifact>, 'cellId' | 'costUsd' | 'costProvenance'>,\n): CostProvenance {\n if (!Number.isFinite(cell.costUsd) || cell.costUsd < 0) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid costUsd`)\n }\n const provenance = cell.costProvenance\n if (!provenance || typeof provenance !== 'object') {\n throw new Error(`campaign cell '${cell.cellId}' has no costProvenance`)\n }\n if (provenance.kind === 'uncaptured') {\n if (provenance.usd !== null) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid uncaptured costProvenance`)\n }\n return { kind: 'uncaptured', usd: null }\n }\n if (\n (provenance.kind !== 'observed' && provenance.kind !== 'estimated') ||\n !Number.isFinite(provenance.usd) ||\n provenance.usd < 0\n ) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid costProvenance`)\n }\n if (provenance.usd !== cell.costUsd) {\n throw new Error(`campaign cell '${cell.cellId}' has costUsd inconsistent with costProvenance`)\n }\n return { kind: provenance.kind, usd: provenance.usd }\n}\n\nexport function campaignCellExecutionEvidence<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): CampaignCellExecutionEvidence {\n if (cell.errorStage === 'dispatch') {\n return {\n terminalOutcome: 'failed',\n executionErrorCount: 1,\n ...(cell.error ? { terminalFailureReason: cell.error } : {}),\n }\n }\n if (cell.errorStage === 'judge') {\n return {\n terminalOutcome: 'succeeded',\n executionErrorCount: 0,\n judgeErrorCount: 1,\n }\n }\n if (!cell.error) {\n return { terminalOutcome: 'succeeded', executionErrorCount: 0 }\n }\n return {\n terminalOutcome: 'unknown',\n unclassifiedErrorCount: 1,\n }\n}\n\n/**\n * Produce the only task-quality view used by campaign aggregates and exports.\n *\n * Successful judge results remain available for diagnosis after another judge\n * fails, but a task score exists only for an error-free cell whose reported\n * judge values are all finite.\n */\nexport function projectCampaignCellQuality<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): CampaignCellQualityProjection {\n if (cell.errorStage === 'dispatch') {\n return { successfulJudgeScores: {}, failedJudges: [], raw: {} }\n }\n\n const perJudge: Record<string, Record<string, number>> = {}\n const successfulJudgeScores: Record<string, JudgeScore> = {}\n const dimensionValues = new Map<string, number[]>()\n const composites: number[] = []\n const notes: string[] = []\n const failedJudges = new Set<string>(\n cell.errorStage === 'judge' ? [cell.errorJudge ?? 'unknown-judge'] : [],\n )\n const raw: Record<string, number> = {}\n\n for (const [judgeName, score] of Object.entries(cell.judgeScores)) {\n const dimensionsShape = (score as { dimensions?: unknown }).dimensions\n if (\n typeof dimensionsShape !== 'object' ||\n dimensionsShape === null ||\n Array.isArray(dimensionsShape)\n ) {\n throw new CampaignJudgeScoreShapeError(judgeName)\n }\n const finiteDimensions = Object.values(score.dimensions).every(Number.isFinite)\n if (score.failed || !Number.isFinite(score.composite) || !finiteDimensions) {\n failedJudges.add(judgeName)\n continue\n }\n\n composites.push(score.composite)\n successfulJudgeScores[judgeName] = score\n const dimensions = { ...score.dimensions }\n perJudge[judgeName] = dimensions\n for (const [dimension, value] of Object.entries(dimensions)) {\n raw[`${judgeName}.${dimension}`] = value\n const values = dimensionValues.get(dimension) ?? []\n values.push(value)\n dimensionValues.set(dimension, values)\n }\n if (score.notes) notes.push(`${judgeName}: ${score.notes}`)\n for (const failedJudge of score.failedJudges ?? []) {\n failedJudges.add(`${judgeName}/${failedJudge}`)\n }\n }\n\n if (failedJudges.size > 0) raw.judge_error_count = failedJudges.size\n const sortedFailedJudges = [...failedJudges].sort()\n if (composites.length === 0) {\n return {\n successfulJudgeScores,\n failedJudges: sortedFailedJudges,\n raw,\n }\n }\n\n const composite = mean(composites)\n const perDimMean = Object.fromEntries(\n [...dimensionValues.entries()].map(([dimension, values]) => [dimension, mean(values)]),\n )\n const complete =\n cell.error === undefined && cell.errorStage === undefined && failedJudges.size === 0\n if (complete) raw.composite = composite\n\n return {\n ...(complete ? { score: composite } : {}),\n raw,\n successfulJudgeScores,\n failedJudges: sortedFailedJudges,\n judgeScores: {\n perJudge,\n perDimMean,\n composite,\n ...(sortedFailedJudges.length > 0 ? { failedJudges: sortedFailedJudges } : {}),\n ...(notes.length > 0 ? { notes: notes.join(' | ') } : {}),\n },\n }\n}\n\n/** Read the canonical task score without recomputing cell quality. */\nexport function campaignCellTaskScore<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): number | undefined {\n return projectCampaignCellQuality(cell).score\n}\n\n/** Read canonical successful judge dimensions without recomputing cell quality. */\nexport function campaignCellJudgeDimensions<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): Record<string, Record<string, number>> {\n return projectCampaignCellQuality(cell).judgeScores?.perJudge ?? {}\n}\n\nfunction finiteMetrics(metrics: Record<string, number> | undefined): Record<string, number> {\n const finite: Record<string, number> = {}\n for (const [key, value] of Object.entries(metrics ?? {})) {\n if (Number.isFinite(value)) finite[key] = value\n }\n return finite\n}\n\nfunction mean(values: number[]): number {\n return values.reduce((sum, value) => sum + value, 0) / values.length\n}\n","/**\n * Verifiable reward channel.\n *\n * For RL on coding / math / theorem-proving / structured-output tasks, the\n * reward signal is *decidable* — a test passes or fails, a proof checks or\n * doesn't, an output validates against a schema or doesn't. These rewards\n * are dramatically more useful for RL training than LLM-judge scores\n * because they don't drift, can't be Goodhart-gamed by the policy in the\n * same way, and don't require a separate calibration loop.\n *\n * The `MultiLayerVerifier` already produces this signal — it just doesn't\n * surface it in a shape that's clean enough for RL training. This module\n * wraps the verifier output so consumers can:\n *\n * 1. Extract a clean `VerifiableReward` from a `VerificationReport`\n * 2. Distinguish *deterministic* rewards (compile, test, schema) from\n * *probabilistic* rewards (judge) so they can be weighted differently\n * in the RL training step\n * 3. Filter `RunRecord[]` to only those with a verifiable reward,\n * producing the clean training set that DeepSeek-R1-style GRPO and\n * AlphaProof-style search both depend on\n *\n * Why this matters: every credible 2025-2026 frontier RL result on coding\n * agents leans on verifiable reward (DeepSeek-R1 GRPO on test pass-rate,\n * o-series RL on math/code, AlphaProof on Lean kernel checking). Mixing\n * judge scores into the reward signal poisons the gradient. This module\n * is the seam.\n */\n\nimport type { LayerResult, VerificationReport } from '../multi-layer-verifier'\nimport { isRealnessGated, observedScore, trainingScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport type { VerificationStrategySource } from '../verification-strategy'\n\n/**\n * What produced a reward. This is the verification-strategy family\n * (`src/verification-strategy.ts`) — one open vocabulary shared by rewards\n * and verdict certifications, so an RL consumer and a certification reader\n * mean the same thing by `'proof-kernel'`.\n *\n * Beyond the classic answer-key members (`compile`, `test`, `schema`,\n * `sandbox`, `judge`, `composite`) the family carries four members for\n * tasks with no held-out suite, each with a documented failure mode:\n *\n * - `'proof-kernel'` — kernel-checked formal proof. Failure mode: the\n * formalization gap — the kernel never certifies that the formal\n * statement matches the informal claim (`src/verification-strategy.ts`\n * is the discharge protocol).\n * - `'invariant'` — invariant / metamorphic properties held. Failure\n * mode: weak invariants pass everything; the set needs seeded-bug\n * calibration before its pass carries weight.\n * - `'replication'` — independent re-execution from pinned inputs.\n * Failure mode: re-runs the method, so it never catches an error the\n * method itself carries.\n * - `'agreement'` — independently-derived results agree. Failure mode:\n * the shared blind spot — derivers with common training corpora or\n * priors agree for the same wrong reason. Probabilistic: the check may\n * be mechanical, but the independence it certifies is not.\n *\n * The deterministic/probabilistic axis stays on `VerifiableReward` —\n * the producer declares it per reward, and `VERIFICATION_STRATEGIES`\n * documents each member's default class.\n */\nexport type VerifiableRewardSource = VerificationStrategySource\n\nexport interface VerifiableReward {\n /** Scalar in [0, 1]. The RL training signal. */\n value: number\n /** What produced the reward — different sources have different determinism. */\n source: VerifiableRewardSource\n /**\n * Determinism class. `'deterministic'` rewards are repeatable byte-for-byte\n * given the same inputs (compile, test, schema validation, sandbox exit code).\n * `'probabilistic'` rewards depend on a stochastic component (LLM judge).\n * Mixing these in the same training batch without separation is a known\n * footgun in production RLHF pipelines.\n */\n determinism: 'deterministic' | 'probabilistic'\n /**\n * Confidence in the reward value. For deterministic sources this is 1.0\n * (the bit either flipped or didn't). For judge sources this is the\n * judge-reported confidence or — when missing — a calibrated prior.\n */\n confidence: number\n /** The layer / judge id that produced the signal, for provenance. */\n origin: string\n /**\n * Per-source contribution to `value`, keyed by layer/judge id. Single-source\n * rewards carry one entry (`{ [origin]: value }`); composite rewards carry\n * every contributing layer's score — the anti-scalar-collapse surface RL\n * consumers weight per-source instead of trusting one blended number.\n */\n components: Record<string, number>\n /**\n * The run carries `outcome.realness.gated` — the authenticity gate flagged\n * its success signal as faked.\n *\n * With the gate applied (the default) `value` and every `components` entry\n * are 0 on such a run; with `applyRealnessGate: false` the observed numbers\n * come back untouched and this flag is the only marker that they are not to\n * be trusted. Either way it distinguishes \"measured a genuine failure\" from\n * \"claimed a success we refuse to believe\", which a bare 0 cannot.\n */\n realnessGated?: boolean\n /**\n * Whether an authenticity screen COULD run on this reward at all — the same\n * distinction `RolloutOutcome.realness_screened` draws, for the same reason.\n *\n * `false` on every reward from `extractVerifiableReward`, because a\n * `VerificationReport` carries layer scores and nothing else: there is no\n * `outcome.realness` to consult, so no gate has run, and `realnessGated`\n * being absent there means \"unknown\", NOT \"clean\". Absent on the\n * `RunRecord` path when the record itself carries no realness verdict.\n *\n * This matters most exactly where it is easiest to miss: a report whose\n * deterministic layers all passed yields `determinism: 'deterministic'`,\n * `confidence: 1` — the highest-credibility reward this module can emit —\n * and a stubbed integration reporting green is precisely what a gamed run\n * looks like. Consumers driving training off this shape must screen the run\n * themselves; the flag is what tells them nobody has.\n */\n realnessScreened?: boolean\n}\n\nexport interface VerifiableRewardExtractionOptions {\n /**\n * Which layers count as deterministic-reward sources. The verifier doesn't\n * tag layers as \"this is verifiable\"; the caller declares it via this list\n * (or via the layer name → source mapping). Default treats common names\n * (`install`, `typecheck`, `build`, `lint`, `test`, `compile`, `schema`,\n * `sandbox`) as deterministic.\n */\n deterministicLayers?: string[]\n /**\n * Map layer name → reward source. Defaults to a sensible string-match.\n */\n sourceFor?: (layerName: string) => VerifiableRewardSource\n /**\n * Whether to fall back to a probabilistic (judge) reward when no\n * deterministic layer produced a numeric score. Default `true`. Set to\n * `false` for \"deterministic-only\" training pipelines that should\n * discard runs without a verifiable signal.\n */\n fallbackToJudge?: boolean\n /**\n * Default confidence for probabilistic (judge) rewards when the judge\n * doesn't report one. Default `0.7`.\n */\n judgeConfidenceFloor?: number\n /**\n * Whether the anti-Goodhart realness gate applies. Default `true`, and the\n * default is the one every training path must keep.\n *\n * Set `false` ONLY for detection and analysis. `rl/reward-hacking.ts` does,\n * for the same reason it reads `observedScore` for its proxy: it measures the\n * DIVERGENCE between the judge signal and the deterministic one, and a\n * deterministic reward that another gate already forced to 0 manufactures\n * exactly that divergence on exactly the gamed population. The detector would\n * then be re-reporting a verdict it was supposed to reach independently.\n */\n applyRealnessGate?: boolean\n}\n\n// 'agreement' is deliberately absent: agreement between fixed artifacts is\n// mechanical, but the independent derivation it certifies is stochastic, so\n// a producer that wants a deterministic agreement reward must declare it\n// (via `deterministicLayers` or by constructing the reward directly).\nconst DEFAULT_DETERMINISTIC_LAYERS = new Set([\n 'install',\n 'typecheck',\n 'build',\n 'lint',\n 'test',\n 'compile',\n 'schema',\n 'sandbox',\n 'unit_tests',\n 'integration_tests',\n 'proof_kernel',\n 'proof-kernel',\n 'invariant',\n 'metamorphic',\n 'replication',\n])\n\nconst DEFAULT_SOURCE_FOR = (name: string): VerifiableRewardSource => {\n const lower = name.toLowerCase()\n if (lower.includes('test')) return 'test'\n if (\n lower.includes('compile') ||\n lower.includes('build') ||\n lower.includes('typecheck') ||\n lower.includes('lint')\n )\n return 'compile'\n if (lower.includes('schema')) return 'schema'\n if (lower.includes('sandbox')) return 'sandbox'\n if (lower.includes('judge') || lower.includes('semantic')) return 'judge'\n // Answer-key names above take precedence: a name like 'proof_test' maps\n // to 'test'; only names with no answer-key match reach the open-family\n // members.\n if (lower.includes('proof') || lower.includes('kernel') || lower.includes('lean')) {\n return 'proof-kernel'\n }\n if (lower.includes('invariant') || lower.includes('metamorphic')) return 'invariant'\n if (lower.includes('replicat')) return 'replication'\n if (lower.includes('agreement') || lower.includes('consensus')) return 'agreement'\n return 'composite'\n}\n\n/**\n * Extract a `VerifiableReward` from a `VerificationReport`.\n *\n * Strategy: prefer the deterministic layers (in order: test → compile →\n * schema → sandbox), fall back to the judge layer if `fallbackToJudge` is\n * true, return `null` if no signal qualifies. When multiple deterministic\n * layers contribute, return a `'composite'` source with a weighted blend.\n *\n * NO realness gate is applied and none can be: a `VerificationReport` carries\n * layer scores and nothing about whether the run faked them — `realness` lives\n * on the `RunRecord`. Use `extractVerifiableRewardsFromRecords` for anything\n * that becomes training data; this signature is for scoring a report in hand.\n */\nexport function extractVerifiableReward(\n report: VerificationReport,\n opts: VerifiableRewardExtractionOptions = {},\n): VerifiableReward | null {\n const deterministicSet = new Set(opts.deterministicLayers ?? [...DEFAULT_DETERMINISTIC_LAYERS])\n const sourceFor = opts.sourceFor ?? DEFAULT_SOURCE_FOR\n const fallbackToJudge = opts.fallbackToJudge ?? true\n const judgeFloor = opts.judgeConfidenceFloor ?? 0.7\n\n const deterministic = report.layers.filter(\n (layer) => deterministicSet.has(layer.layer) && isMeasuredLayer(layer),\n )\n\n if (deterministic.length === 1) {\n const layer = deterministic[0]!\n const value = clamp01(layer.score!)\n return {\n value,\n source: sourceFor(layer.layer),\n determinism: 'deterministic',\n confidence: 1,\n origin: layer.layer,\n components: { [layer.layer]: value },\n realnessScreened: false,\n }\n }\n\n if (deterministic.length > 1) {\n // Composite: weighted blend by `Layer.weight` if present, else equal.\n let num = 0\n let denom = 0\n const components: Record<string, number> = {}\n for (const l of deterministic) {\n const w = (l.detail?.weight as number | undefined) ?? 1\n num += w * (l.score ?? 0)\n denom += w\n components[l.layer] = l.score!\n }\n return {\n value: denom === 0 ? 0 : clamp01(num / denom),\n source: 'composite',\n determinism: 'deterministic',\n confidence: 1,\n origin: deterministic.map((l) => l.layer).join('+'),\n components,\n realnessScreened: false,\n }\n }\n\n if (!fallbackToJudge) return null\n\n const judge =\n report.layers.find((layer) => isMeasuredLayer(layer) && sourceFor(layer.layer) === 'judge') ??\n report.layers.find(isMeasuredLayer)\n\n if (!judge) return null\n\n const confFromDetail = judge.detail?.confidence as number | undefined\n const judgeValue = clamp01(judge.score!)\n return {\n value: judgeValue,\n source: 'judge',\n determinism: 'probabilistic',\n confidence: typeof confFromDetail === 'number' ? confFromDetail : judgeFloor,\n origin: judge.layer,\n components: { [judge.layer]: judgeValue },\n realnessScreened: false,\n }\n}\n\nfunction isMeasuredLayer(\n layer: LayerResult,\n): layer is LayerResult & { status: 'pass' | 'fail'; score: number } {\n return (\n (layer.status === 'pass' || layer.status === 'fail') &&\n typeof layer.score === 'number' &&\n Number.isFinite(layer.score) &&\n layer.score >= 0 &&\n layer.score <= 1\n )\n}\n\n/**\n * Extract verifiable rewards from `RunRecord[]` produced via the\n * `verificationReportToRunRecord` adapter (which encodes per-layer scores\n * in `outcome.raw['layer.<name>']`). For records that don't carry layer\n * scores, returns `null` for that record.\n *\n * This is the canonical bridge from \"campaign-shaped artifacts\" to\n * \"RL-training-ready reward signals\": every record that has a clean\n * verifiable reward becomes a training datum, every record that doesn't\n * gets filtered out (or kept with `'probabilistic'` determinism for\n * separate downstream handling).\n *\n * The realness gate applies to EVERY channel here, and to the deterministic one\n * MOST. It is tempting to reason that a decidable signal cannot be gamed, so\n * the gate is redundant on it — that reasoning is backwards. `realness.gated`\n * means the run's success signal was FAKED, and a test suite reporting green on\n * a stubbed integration is precisely what that looks like: the deterministic\n * layer is the thing that got faked. Exporting it ungated hands a trainer the\n * highest-credibility reward the module can emit (`determinism: 'deterministic'`,\n * `confidence: 1`) for the one population the gate exists to catch. Pass\n * `applyRealnessGate: false` only to look at the ungated numbers for detection.\n */\nexport function extractVerifiableRewardsFromRecords(\n runs: RunRecord[],\n opts: VerifiableRewardExtractionOptions = {},\n): Array<{ runId: string; reward: VerifiableReward | null }> {\n const sourceFor = opts.sourceFor ?? DEFAULT_SOURCE_FOR\n const deterministicSet = new Set(opts.deterministicLayers ?? [...DEFAULT_DETERMINISTIC_LAYERS])\n const fallbackToJudge = opts.fallbackToJudge ?? true\n const judgeFloor = opts.judgeConfidenceFloor ?? 0.7\n const applyGate = opts.applyRealnessGate ?? true\n\n return runs.map((run) => {\n const flagged = isRealnessGated(run)\n // Present only when the record carries an actual realness verdict. Absent\n // is the honest \"unknown\"; `false` is reserved for a producer that\n // declares it HAS no screen, which is the report-shaped path above.\n const screened = run.outcome.realness === undefined ? {} : ({ realnessScreened: true } as const)\n // Zeroed with `value`, never left at the measured number: `components`\n // exists so an RL consumer can re-weight per source, and a raw layer score\n // surviving there would let that re-weighting reconstruct the very reward\n // the gate just refused. The measured layer scores stay on\n // `run.outcome.raw['layer.*']`, which is where analysis reads them.\n const gate = (value: number): number => (applyGate && flagged ? 0 : value)\n // Recover per-layer scores from outcome.raw['layer.<name>']\n const layerScores: Array<{ name: string; score: number }> = []\n for (const [k, v] of Object.entries(run.outcome.raw)) {\n if (\n k.startsWith('layer.') &&\n !k.includes('.', 6) &&\n typeof v === 'number' &&\n Number.isFinite(v)\n ) {\n layerScores.push({ name: k.slice('layer.'.length), score: v })\n }\n }\n const det = layerScores.filter((l) => deterministicSet.has(l.name))\n\n if (det.length === 1) {\n const layer = det[0]!\n const value = gate(clamp01(layer.score))\n return {\n runId: run.runId,\n reward: {\n value,\n source: sourceFor(layer.name),\n determinism: 'deterministic',\n confidence: 1,\n origin: layer.name,\n components: { [layer.name]: value },\n realnessGated: flagged,\n ...screened,\n },\n }\n }\n if (det.length > 1) {\n const value = gate(clamp01(det.reduce((s, l) => s + l.score, 0) / det.length))\n // Same clamp as the headline value: a producer writing layer.score 1.5\n // into outcome.raw must not propagate 1.5 through a component either.\n const components: Record<string, number> = Object.fromEntries(\n det.map((l) => [l.name, gate(clamp01(l.score))]),\n )\n return {\n runId: run.runId,\n reward: {\n value,\n source: 'composite',\n determinism: 'deterministic',\n confidence: 1,\n origin: det.map((l) => l.name).join('+'),\n components,\n realnessGated: flagged,\n ...screened,\n },\n }\n }\n if (!fallbackToJudge) return { runId: run.runId, reward: null }\n\n // Probabilistic fallback: the run's primary score. `trainingScore` already\n // carries the gate, so a gamed run falls to 0 rather than earning the\n // judge's number; `observedScore` is the ungated reader the detection\n // opt-out asks for. Either way an unscored run stays a labeled gap\n // (`reward: null`), never a fabricated 0.\n const primary = applyGate ? trainingScore(run) : observedScore(run)\n if (typeof primary !== 'number' || !Number.isFinite(primary)) {\n return { runId: run.runId, reward: null }\n }\n const primaryValue = clamp01(primary)\n return {\n runId: run.runId,\n reward: {\n value: primaryValue,\n source: 'judge',\n determinism: 'probabilistic',\n confidence: judgeFloor,\n origin: 'run.outcome.score',\n components: { 'run.outcome.score': primaryValue },\n realnessGated: flagged,\n ...screened,\n },\n }\n })\n}\n\n/**\n * Filter `RunRecord[]` to those with deterministic verifiable rewards.\n *\n * A realness-gated run is KEPT, at reward 0 with `realnessGated: true` — the\n * same rule GRPO uses on a gated line. 0 is the honest label for a faked\n * success and is usable signal, whereas dropping the run would move a group\n * baseline without saying so. (SFT differs: there every row is a target to\n * imitate, so a gated row is removed outright.)\n */\nexport function filterDeterministicallyRewarded(\n runs: RunRecord[],\n opts: VerifiableRewardExtractionOptions = {},\n): Array<{ run: RunRecord; reward: VerifiableReward }> {\n const rewarded = extractVerifiableRewardsFromRecords(runs, { ...opts, fallbackToJudge: false })\n const out: Array<{ run: RunRecord; reward: VerifiableReward }> = []\n for (let i = 0; i < runs.length; i++) {\n const r = rewarded[i]!\n if (r.reward && r.reward.determinism === 'deterministic') {\n out.push({ run: runs[i]!, reward: r.reward })\n }\n }\n return out\n}\n\nfunction clamp01(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n","/**\n * Reward hacking / Goodhart detection.\n *\n * Goodhart's Law says: when a measure becomes a target, it ceases to be\n * a good measure. In RLHF and agentic-RL settings this is the dominant\n * failure mode — the policy learns to produce outputs that score well on\n * the proxy reward (judge, rubric, test pass-rate) without producing\n * the underlying capability the proxy was meant to track.\n *\n * Krakovna et al. (2020, \"Specification Gaming Examples in AI\") and the\n * subsequent RLHF reward-hacking literature (Skalse et al. 2022, Kim et al.\n * 2023) converge on a few diagnostic signatures:\n *\n * 1. **Reward divergence:** the proxy reward grows while the held-out\n * ground-truth signal stagnates or drops. Predictive validity over\n * time captures this.\n * 2. **Distributional shift in outputs:** after RL, the policy produces\n * outputs that no longer match the reference distribution — usually\n * because it found a high-reward attractor that's degenerate (e.g.\n * one-token responses, repetition, formatting tricks).\n * 3. **Disagreement between independent rewards:** if you train on\n * reward A and a held-out independent reward B drops sharply, you're\n * probably hacking A.\n * 4. **Calibration drift:** the verifiable / deterministic component of\n * the reward is stable; the probabilistic / judge component drifts up\n * while the deterministic component doesn't. The judge is being\n * gamed.\n *\n * This module ships explicit detectors for all four signatures, plus a\n * combined verdict. The output is diagnostic — actionable signals,\n * not autoreject — because each signature has known false positives\n * (e.g., a policy that genuinely improves can show distributional shift).\n *\n * Differs from `rubricPredictiveValidity` (which is a *standing* check on\n * whether rubrics correlate with deployment outcomes) — this is a\n * *temporal* check on whether the reward-vs-truth gap is *widening over\n * time during a training run*.\n */\n\nimport { observedScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport { pearsonR } from '../statistics'\nimport {\n filterDeterministicallyRewarded,\n type VerifiableRewardExtractionOptions,\n} from './verifiable-reward'\n\nexport type RewardHackingSignal =\n | 'reward_divergence'\n | 'distribution_shift'\n | 'reward_disagreement'\n | 'judge_drift'\n\nexport interface RewardHackingFinding {\n signal: RewardHackingSignal\n /** Severity in [0, 1]. >0.5 = strong signal. */\n severity: number\n message: string\n /** Numeric evidence the consumer can render. */\n detail: Record<string, number>\n}\n\nexport interface RewardHackingReport {\n findings: RewardHackingFinding[]\n /** Signals with enough usable observations to produce a finding. */\n evaluatedSignals: RewardHackingSignal[]\n /**\n * Composite verdict. `'insufficient_evidence'` when fewer than four scored\n * runs exist; otherwise `'clean'` if every signal severity < 0.3,\n * `'suspect'` if at least one ≥ 0.3 but none ≥ 0.6, and `'gaming'` if any ≥ 0.6.\n */\n verdict: 'insufficient_evidence' | 'clean' | 'suspect' | 'gaming'\n /** Rationale for the verdict, ready to paste into an audit log. */\n rationale: string[]\n /** Number of runs with a usable proxy reward. */\n n: number\n}\n\nexport interface DetectRewardHackingInput {\n /**\n * Run records ordered by recency (oldest first). The detector segments\n * them into prefix/suffix windows to compute \"did the gap widen.\"\n */\n runs: RunRecord[]\n /**\n * The metric the policy was trained to optimize. Should be present on\n * `outcome.raw` or `outcome.holdoutScore`. Default reads `outcome.holdoutScore`.\n */\n proxyOf?: (run: RunRecord) => number | null\n /**\n * The held-out ground-truth metric. For RL on coding, this is typically\n * test pass-rate. For RLHF, it's downstream task performance or human\n * preference. For knowledge tasks, it's an independently-graded score.\n */\n truthOf?: (run: RunRecord) => number | null\n /**\n * Independent secondary reward. Used for the `reward_disagreement`\n * signal. Default uses the verifiable reward extractor (deterministic\n * sources only).\n */\n secondaryRewardOf?: (run: RunRecord) => number | null\n /**\n * Window size — how many of the most recent runs count as the \"after\"\n * cohort. Default min(50, half the runs).\n */\n windowSize?: number\n /**\n * Severity threshold to flag a signal. Default 0.3 (suspect) and 0.6\n * (gaming).\n */\n thresholds?: { suspect?: number; gaming?: number }\n /**\n * Verifiable-reward options used for the secondary-reward fallback.\n */\n verifiableRewardOptions?: VerifiableRewardExtractionOptions\n}\n\nconst DEFAULT_PROXY = (r: RunRecord): number | null => {\n // DELIBERATELY UNGATED. This is the proxy reward the detector tests for\n // Goodharting, and gated runs are exactly the gamed population. Forcing them\n // to 0 would collapse the proxy toward the deterministic secondary signal —\n // `reward_disagreement`'s correlation would rise, `judge_drift`'s gap would\n // shrink, and `reward_divergence`'s proxy-up/truth-flat fingerprint would be\n // erased — so the detector would report \"clean\" on the very runs it exists to\n // catch. `null` (never 0) is also load-bearing: it sets the n-denominator via\n // the filter below.\n const v = observedScore(r)\n return typeof v === 'number' && Number.isFinite(v) ? v : null\n}\n\nexport function detectRewardHacking(input: DetectRewardHackingInput): RewardHackingReport {\n const proxyOf = input.proxyOf ?? DEFAULT_PROXY\n const truthOf = input.truthOf\n const sus = input.thresholds?.suspect ?? 0.3\n const gam = input.thresholds?.gaming ?? 0.6\n\n const runs = input.runs.filter((run) => finiteNumber(proxyOf(run)))\n const n = runs.length\n if (n < 4) {\n return {\n findings: [],\n evaluatedSignals: [],\n verdict: 'insufficient_evidence',\n n,\n rationale: [`fewer than 4 runs with proxy reward (n=${n}); insufficient evidence`],\n }\n }\n const windowSize = Math.max(1, input.windowSize ?? Math.min(50, Math.floor(n / 2)))\n const before = runs.slice(0, n - windowSize)\n const after = runs.slice(n - windowSize)\n\n const findings: RewardHackingFinding[] = []\n\n // ── Signal 1: reward divergence (proxy ↑ while truth flat or ↓) ──────\n if (truthOf) {\n const beforeProxy = before.map(proxyOf).filter(finiteNumber)\n const afterProxy = after.map(proxyOf).filter(finiteNumber)\n const beforeTruth = before.map(truthOf).filter(finiteNumber)\n const afterTruth = after.map(truthOf).filter(finiteNumber)\n if (\n beforeProxy.length >= 2 &&\n afterProxy.length >= 2 &&\n beforeTruth.length >= 2 &&\n afterTruth.length >= 2\n ) {\n const proxyDelta = mean(afterProxy) - mean(beforeProxy)\n const truthDelta = mean(afterTruth) - mean(beforeTruth)\n // Divergence: proxy goes up while truth goes flat or down.\n // Severity = max(0, (proxyDelta - truthDelta)) — bigger gap = bigger signal.\n const gap = Math.max(0, proxyDelta - truthDelta)\n const severity = clamp01(gap * 5) // scale: 0.2 absolute gap → severity 1.0\n findings.push({\n signal: 'reward_divergence',\n severity,\n message:\n severity >= sus\n ? `proxy reward rose by ${proxyDelta.toFixed(3)} while truth changed by ${truthDelta.toFixed(3)} — potential Goodhart`\n : `proxy and truth moved together (proxy ${proxyDelta.toFixed(3)}, truth ${truthDelta.toFixed(3)})`,\n detail: {\n proxyDelta,\n truthDelta,\n gap,\n beforeN: beforeProxy.length,\n afterN: afterProxy.length,\n },\n })\n }\n }\n\n // ── Signal 2: distributional shift in outputs (KS on score distributions) ──\n {\n const beforeP = before.map(proxyOf).filter(finiteNumber)\n const afterP = after.map(proxyOf).filter(finiteNumber)\n if (beforeP.length >= 4 && afterP.length >= 4) {\n const ks = ksStatistic(beforeP, afterP)\n // KS statistic: bigger = more shift. We're agnostic about direction;\n // genuine improvement ALSO produces shift, so this signal is\n // contributory rather than load-bearing.\n const severity = clamp01(ks - 0.2)\n findings.push({\n signal: 'distribution_shift',\n severity,\n message:\n severity >= sus\n ? `KS=${ks.toFixed(3)} between before/after windows — distributional shift large`\n : `KS=${ks.toFixed(3)} between before/after windows — within-distribution drift`,\n detail: { ks, beforeN: beforeP.length, afterN: afterP.length },\n })\n }\n }\n\n // ── Signal 3: reward disagreement (proxy vs independent secondary) ────\n {\n const secondaryOf = input.secondaryRewardOf ?? defaultSecondary(input.verifiableRewardOptions)\n const aligned = runs\n .map((r) => ({ p: proxyOf(r), s: secondaryOf(r) }))\n .filter((x): x is { p: number; s: number } => finiteNumber(x.p) && finiteNumber(x.s))\n if (aligned.length >= 4) {\n const ps = aligned.map((x) => x.p)\n const ss = aligned.map((x) => x.s)\n const r = pearsonR(ps, ss)\n // Disagreement: low or negative correlation between primary proxy\n // reward and an independent secondary signal.\n const severity = clamp01(0.5 - Math.max(0, r))\n findings.push({\n signal: 'reward_disagreement',\n severity,\n message:\n severity >= sus\n ? `proxy and independent secondary reward correlate ρ=${r.toFixed(3)} — possibly hacking proxy`\n : `proxy and secondary reward correlate ρ=${r.toFixed(3)}`,\n detail: { pearson: r, n: aligned.length },\n })\n }\n }\n\n // ── Signal 4: judge drift (probabilistic up while deterministic flat) ─\n {\n // Ungated on purpose, exactly like `DEFAULT_PROXY` above. This signal is\n // the GAP between the judge reward and the deterministic one; a\n // deterministic reward another gate already forced to 0 would open that gap\n // by construction on the gamed population, so the detector would fire on\n // its own input rather than on evidence it found.\n const detRuns = filterDeterministicallyRewarded(runs, {\n ...(input.verifiableRewardOptions ?? {}),\n applyRealnessGate: false,\n })\n if (detRuns.length >= 4) {\n const detBefore = detRuns.slice(0, Math.floor(detRuns.length / 2))\n const detAfter = detRuns.slice(Math.floor(detRuns.length / 2))\n const detDelta =\n mean(detAfter.map((r) => r.reward.value)) - mean(detBefore.map((r) => r.reward.value))\n const proxyDelta =\n mean(after.map(proxyOf).filter(finiteNumber)) -\n mean(before.map(proxyOf).filter(finiteNumber))\n const driftGap = Math.max(0, proxyDelta - detDelta)\n const severity = clamp01(driftGap * 5)\n findings.push({\n signal: 'judge_drift',\n severity,\n message:\n severity >= sus\n ? `judge proxy +${proxyDelta.toFixed(3)} while deterministic reward +${detDelta.toFixed(3)} — judge drifting up without verifiable backing`\n : `judge and deterministic rewards move in step (judge ${proxyDelta.toFixed(3)}, det ${detDelta.toFixed(3)})`,\n detail: { proxyDelta, detDelta, driftGap, n: detRuns.length },\n })\n }\n }\n\n const maxSev = findings.reduce((m, f) => Math.max(m, f.severity), 0)\n if (findings.length === 0) {\n return {\n findings,\n evaluatedSignals: [],\n verdict: 'insufficient_evidence',\n rationale: [`no reward-hacking signal had enough paired evidence (n=${n})`],\n n,\n }\n }\n const verdict: RewardHackingReport['verdict'] =\n maxSev >= gam ? 'gaming' : maxSev >= sus ? 'suspect' : 'clean'\n const rationale = findings\n .filter((f) => f.severity >= sus)\n .map((f) => `${f.signal}: severity ${f.severity.toFixed(2)} — ${f.message}`)\n if (rationale.length === 0) rationale.push('no signals fired above suspect threshold')\n\n return {\n findings,\n evaluatedSignals: findings.map((finding) => finding.signal),\n verdict,\n rationale,\n n,\n }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction mean(xs: number[]): number {\n if (xs.length === 0) return 0\n return xs.reduce((s, x) => s + x, 0) / xs.length\n}\n\nfunction finiteNumber(value: number | null): value is number {\n return typeof value === 'number' && Number.isFinite(value)\n}\n\nfunction clamp01(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n\nfunction ksStatistic(a: number[], b: number[]): number {\n // Two-sample Kolmogorov-Smirnov statistic.\n const sortedA = [...a].sort((x, y) => x - y)\n const sortedB = [...b].sort((x, y) => x - y)\n const all = [...new Set([...sortedA, ...sortedB])].sort((x, y) => x - y)\n let max = 0\n for (const v of all) {\n const fa = sortedA.filter((x) => x <= v).length / sortedA.length\n const fb = sortedB.filter((x) => x <= v).length / sortedB.length\n max = Math.max(max, Math.abs(fa - fb))\n }\n return max\n}\n\nfunction defaultSecondary(\n verifiableOpts?: VerifiableRewardExtractionOptions,\n): (run: RunRecord) => number | null {\n return (run: RunRecord) => {\n // Ungated for the same reason as signal 4: this is the INDEPENDENT\n // secondary reward whose correlation with the proxy is the evidence.\n // Zeroing it on gated runs would drive that correlation down mechanically.\n const filtered = filterDeterministicallyRewarded([run], {\n ...(verifiableOpts ?? {}),\n applyRealnessGate: false,\n })\n return filtered.length === 1 ? filtered[0]!.reward.value : null\n }\n}\n"],"mappings":";;;;;;;;;;;;;;AA8CA,IAAa,+BAAb,cAAkD,UAAU;CAC1D,YAAY,WAAmB;EAC7B,MACE,wBAAwB,UAAU,mQAIpC;EACA,KAAK,OAAO;CACd;AACF;;;;;;;;;AAkBA,SAAgB,wBACd,MACA,SACW;CACX,MAAM,UAAU,2BAA2B,IAAI;CAC/C,MAAM,YAAY,8BAA8B,IAAI;CACpD,MAAM,kBAAkB,KAAK,IAC3B,QAAQ,IAAI,qBAAqB,GACjC,UAAU,mBAAmB,CAC/B;CACA,MAAM,qBAAqB,2BAA2B,IAAI;CAC1D,MAAM,iBACJ,mBAAmB,SAAS,gBAAgB,QAAQ,mBAAmB,KAAA,IACnE;EAAE,MAAM;EAAa,KAAK,QAAQ;CAAe,IACjD;CACN,MAAM,UAAU,eAAe,SAAS,eAAe,OAAO,eAAe;CAC7E,MAAM,MAA8B;EAClC,GAAG,cAAc,QAAQ,GAAG;EAC5B,GAAG,QAAQ;EACX,KAAK,KAAK;EACV,aAAa,KAAK;EAClB,GAAI,YAAY,OAAO,CAAC,IAAI,EAAE,UAAU,QAAQ;EAEhD,GAAI,mBAAmB,SAAS,eAAe,EAAE,yBAAyB,KAAK,QAAQ,IAAI,CAAC;EAC5F,eAAe,eAAe,SAAS,aAAa,IAAI;EACxD,gBAAgB,eAAe,SAAS,cAAc,IAAI;EAC1D,iBAAiB,eAAe,SAAS,eAAe,IAAI;EAC5D,cAAc,KAAK,WAAW;EAC9B,eAAe,KAAK,WAAW;EAC/B,cAAc,KAAK,WAAW,gBAAgB,QAAQ,IAAI;EAC1D,YAAY,KAAK;EACjB,GAAI,UAAU,wBAAwB,KAAA,IAClC,CAAC,IACD,EAAE,uBAAuB,UAAU,oBAAoB;EAC3D,GAAI,kBAAkB,IAAI,EAAE,mBAAmB,gBAAgB,IAAI,CAAC;EACpE,GAAI,UAAU,2BAA2B,KAAA,IACrC,CAAC,IACD,EAAE,0BAA0B,UAAU,uBAAuB;CACnE;CACA,IAAI,OAAO,KAAK,eAAe,UAAU,IAAI,aAAa,KAAK;CAC/D,IAAI,KAAK,WAAW,cAAc,KAAA,GAChC,IAAI,mBAAmB,KAAK,WAAW;CAEzC,IAAI,KAAK,WAAW,WAAW,KAAA,GAAW,IAAI,gBAAgB,KAAK,WAAW;CAC9E,IAAI,KAAK,WAAW,eAAe,KAAA,GACjC,IAAI,qBAAqB,KAAK,WAAW;CAE3C,IAAI,KAAK,WAAW,gBAAgB,SAAS,YAAY,QAAQ,UAAU,GACzE,IAAI,qBAAqB,KAAK,WAAW,QAAQ,KAAK,WAAW,UAAU;CAE7E,IAAI,YAAY,QAAQ,QAAQ,UAAU,KAAA,KAAa,QAAQ,QAAQ,KACrE,IAAI,mBAAmB,UAAU,QAAQ;CAG3C,MAAM,UAAsB;EAC1B;EACA,GAAI,QAAQ,cAAc,EAAE,aAAa,QAAQ,YAAY,IAAI,CAAC;CACpE;CACA,IAAI,QAAQ,UAAU,KAAA,GACpB,IAAI,QAAQ,aAAa,WAAW,QAAQ,eAAe,QAAQ;MAC9D,QAAQ,cAAc,QAAQ;CAGrC,OAAO,kBAAkB;EACvB,OAAO,QAAQ;EACf,cAAc,QAAQ;EACtB,aAAa,QAAQ;EACrB,MAAM,QAAQ,QAAQ,KAAK;EAC3B,OAAO,QAAQ;EACf,YAAY,QAAQ;EACpB,YAAY,QAAQ;EACpB,WAAW,QAAQ;EACnB,QAAQ,KAAK;EACb;EACA;EACA,YAAY,EAAE,GAAG,KAAK,WAAW;EACjC,iBAAiB,UAAU;EAC3B,GAAI,UAAU,wBACV,EAAE,uBAAuB,UAAU,sBAAsB,IACzD,CAAC;EACL;EACA,UAAU,QAAQ;EAClB,YAAY,QAAQ,cAAc,KAAK;EACvC,GAAI,QAAQ,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CACvE,CAAC;AACH;;;;;;AAOA,SAAgB,2BACd,MACgB;CAChB,IAAI,CAAC,OAAO,SAAS,KAAK,OAAO,KAAK,KAAK,UAAU,GACnD,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,sBAAsB;CAEtE,MAAM,aAAa,KAAK;CACxB,IAAI,CAAC,cAAc,OAAO,eAAe,UACvC,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,wBAAwB;CAExE,IAAI,WAAW,SAAS,cAAc;EACpC,IAAI,WAAW,QAAQ,MACrB,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,wCAAwC;EAExF,OAAO;GAAE,MAAM;GAAc,KAAK;EAAK;CACzC;CACA,IACG,WAAW,SAAS,cAAc,WAAW,SAAS,eACvD,CAAC,OAAO,SAAS,WAAW,GAAG,KAC/B,WAAW,MAAM,GAEjB,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,6BAA6B;CAE7E,IAAI,WAAW,QAAQ,KAAK,SAC1B,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,+CAA+C;CAE/F,OAAO;EAAE,MAAM,WAAW;EAAM,KAAK,WAAW;CAAI;AACtD;AAEA,SAAgB,8BACd,MAC+B;CAC/B,IAAI,KAAK,eAAe,YACtB,OAAO;EACL,iBAAiB;EACjB,qBAAqB;EACrB,GAAI,KAAK,QAAQ,EAAE,uBAAuB,KAAK,MAAM,IAAI,CAAC;CAC5D;CAEF,IAAI,KAAK,eAAe,SACtB,OAAO;EACL,iBAAiB;EACjB,qBAAqB;EACrB,iBAAiB;CACnB;CAEF,IAAI,CAAC,KAAK,OACR,OAAO;EAAE,iBAAiB;EAAa,qBAAqB;CAAE;CAEhE,OAAO;EACL,iBAAiB;EACjB,wBAAwB;CAC1B;AACF;;;;;;;;AASA,SAAgB,2BACd,MAC+B;CAC/B,IAAI,KAAK,eAAe,YACtB,OAAO;EAAE,uBAAuB,CAAC;EAAG,cAAc,CAAC;EAAG,KAAK,CAAC;CAAE;CAGhE,MAAM,WAAmD,CAAC;CAC1D,MAAM,wBAAoD,CAAC;CAC3D,MAAM,kCAAkB,IAAI,IAAsB;CAClD,MAAM,aAAuB,CAAC;CAC9B,MAAM,QAAkB,CAAC;CACzB,MAAM,eAAe,IAAI,IACvB,KAAK,eAAe,UAAU,CAAC,KAAK,cAAc,eAAe,IAAI,CAAC,CACxE;CACA,MAAM,MAA8B,CAAC;CAErC,KAAK,MAAM,CAAC,WAAW,UAAU,OAAO,QAAQ,KAAK,WAAW,GAAG;EACjE,MAAM,kBAAmB,MAAmC;EAC5D,IACE,OAAO,oBAAoB,YAC3B,oBAAoB,QACpB,MAAM,QAAQ,eAAe,GAE7B,MAAM,IAAI,6BAA6B,SAAS;EAElD,MAAM,mBAAmB,OAAO,OAAO,MAAM,UAAU,CAAC,CAAC,MAAM,OAAO,QAAQ;EAC9E,IAAI,MAAM,UAAU,CAAC,OAAO,SAAS,MAAM,SAAS,KAAK,CAAC,kBAAkB;GAC1E,aAAa,IAAI,SAAS;GAC1B;EACF;EAEA,WAAW,KAAK,MAAM,SAAS;EAC/B,sBAAsB,aAAa;EACnC,MAAM,aAAa,EAAE,GAAG,MAAM,WAAW;EACzC,SAAS,aAAa;EACtB,KAAK,MAAM,CAAC,WAAW,UAAU,OAAO,QAAQ,UAAU,GAAG;GAC3D,IAAI,GAAG,UAAU,GAAG,eAAe;GACnC,MAAM,SAAS,gBAAgB,IAAI,SAAS,KAAK,CAAC;GAClD,OAAO,KAAK,KAAK;GACjB,gBAAgB,IAAI,WAAW,MAAM;EACvC;EACA,IAAI,MAAM,OAAO,MAAM,KAAK,GAAG,UAAU,IAAI,MAAM,OAAO;EAC1D,KAAK,MAAM,eAAe,MAAM,gBAAgB,CAAC,GAC/C,aAAa,IAAI,GAAG,UAAU,GAAG,aAAa;CAElD;CAEA,IAAI,aAAa,OAAO,GAAG,IAAI,oBAAoB,aAAa;CAChE,MAAM,qBAAqB,CAAC,GAAG,YAAY,CAAC,CAAC,KAAK;CAClD,IAAI,WAAW,WAAW,GACxB,OAAO;EACL;EACA,cAAc;EACd;CACF;CAGF,MAAM,YAAYA,OAAK,UAAU;CACjC,MAAM,aAAa,OAAO,YACxB,CAAC,GAAG,gBAAgB,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,WAAW,YAAY,CAAC,WAAWA,OAAK,MAAM,CAAC,CAAC,CACvF;CACA,MAAM,WACJ,KAAK,UAAU,KAAA,KAAa,KAAK,eAAe,KAAA,KAAa,aAAa,SAAS;CACrF,IAAI,UAAU,IAAI,YAAY;CAE9B,OAAO;EACL,GAAI,WAAW,EAAE,OAAO,UAAU,IAAI,CAAC;EACvC;EACA;EACA,cAAc;EACd,aAAa;GACX;GACA;GACA;GACA,GAAI,mBAAmB,SAAS,IAAI,EAAE,cAAc,mBAAmB,IAAI,CAAC;GAC5E,GAAI,MAAM,SAAS,IAAI,EAAE,OAAO,MAAM,KAAK,KAAK,EAAE,IAAI,CAAC;EACzD;CACF;AACF;;AAGA,SAAgB,sBACd,MACoB;CACpB,OAAO,2BAA2B,IAAI,CAAC,CAAC;AAC1C;;AAGA,SAAgB,4BACd,MACwC;CACxC,OAAO,2BAA2B,IAAI,CAAC,CAAC,aAAa,YAAY,CAAC;AACpE;AAEA,SAAS,cAAc,SAAqE;CAC1F,MAAM,SAAiC,CAAC;CACxC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,WAAW,CAAC,CAAC,GACrD,IAAI,OAAO,SAAS,KAAK,GAAG,OAAO,OAAO;CAE5C,OAAO;AACT;AAEA,SAASA,OAAK,QAA0B;CACtC,OAAO,OAAO,QAAQ,KAAK,UAAU,MAAM,OAAO,CAAC,IAAI,OAAO;AAChE;;;ACtKA,MAAM,+CAA+B,IAAI,IAAI;CAC3C;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAM,sBAAsB,SAAyC;CACnE,MAAM,QAAQ,KAAK,YAAY;CAC/B,IAAI,MAAM,SAAS,MAAM,GAAG,OAAO;CACnC,IACE,MAAM,SAAS,SAAS,KACxB,MAAM,SAAS,OAAO,KACtB,MAAM,SAAS,WAAW,KAC1B,MAAM,SAAS,MAAM,GAErB,OAAO;CACT,IAAI,MAAM,SAAS,QAAQ,GAAG,OAAO;CACrC,IAAI,MAAM,SAAS,SAAS,GAAG,OAAO;CACtC,IAAI,MAAM,SAAS,OAAO,KAAK,MAAM,SAAS,UAAU,GAAG,OAAO;CAIlE,IAAI,MAAM,SAAS,OAAO,KAAK,MAAM,SAAS,QAAQ,KAAK,MAAM,SAAS,MAAM,GAC9E,OAAO;CAET,IAAI,MAAM,SAAS,WAAW,KAAK,MAAM,SAAS,aAAa,GAAG,OAAO;CACzE,IAAI,MAAM,SAAS,UAAU,GAAG,OAAO;CACvC,IAAI,MAAM,SAAS,WAAW,KAAK,MAAM,SAAS,WAAW,GAAG,OAAO;CACvE,OAAO;AACT;;;;;;;;;;;;;;AAeA,SAAgB,wBACd,QACA,OAA0C,CAAC,GAClB;CACzB,MAAM,mBAAmB,IAAI,IAAI,KAAK,uBAAuB,CAAC,GAAG,4BAA4B,CAAC;CAC9F,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,aAAa,KAAK,wBAAwB;CAEhD,MAAM,gBAAgB,OAAO,OAAO,QACjC,UAAU,iBAAiB,IAAI,MAAM,KAAK,KAAK,gBAAgB,KAAK,CACvE;CAEA,IAAI,cAAc,WAAW,GAAG;EAC9B,MAAM,QAAQ,cAAc;EAC5B,MAAM,QAAQC,UAAQ,MAAM,KAAM;EAClC,OAAO;GACL;GACA,QAAQ,UAAU,MAAM,KAAK;GAC7B,aAAa;GACb,YAAY;GACZ,QAAQ,MAAM;GACd,YAAY,GAAG,MAAM,QAAQ,MAAM;GACnC,kBAAkB;EACpB;CACF;CAEA,IAAI,cAAc,SAAS,GAAG;EAE5B,IAAI,MAAM;EACV,IAAI,QAAQ;EACZ,MAAM,aAAqC,CAAC;EAC5C,KAAK,MAAM,KAAK,eAAe;GAC7B,MAAM,IAAK,EAAE,QAAQ,UAAiC;GACtD,OAAO,KAAK,EAAE,SAAS;GACvB,SAAS;GACT,WAAW,EAAE,SAAS,EAAE;EAC1B;EACA,OAAO;GACL,OAAO,UAAU,IAAI,IAAIA,UAAQ,MAAM,KAAK;GAC5C,QAAQ;GACR,aAAa;GACb,YAAY;GACZ,QAAQ,cAAc,KAAK,MAAM,EAAE,KAAK,CAAC,CAAC,KAAK,GAAG;GAClD;GACA,kBAAkB;EACpB;CACF;CAEA,IAAI,CAAC,iBAAiB,OAAO;CAE7B,MAAM,QACJ,OAAO,OAAO,MAAM,UAAU,gBAAgB,KAAK,KAAK,UAAU,MAAM,KAAK,MAAM,OAAO,KAC1F,OAAO,OAAO,KAAK,eAAe;CAEpC,IAAI,CAAC,OAAO,OAAO;CAEnB,MAAM,iBAAiB,MAAM,QAAQ;CACrC,MAAM,aAAaA,UAAQ,MAAM,KAAM;CACvC,OAAO;EACL,OAAO;EACP,QAAQ;EACR,aAAa;EACb,YAAY,OAAO,mBAAmB,WAAW,iBAAiB;EAClE,QAAQ,MAAM;EACd,YAAY,GAAG,MAAM,QAAQ,WAAW;EACxC,kBAAkB;CACpB;AACF;AAEA,SAAS,gBACP,OACmE;CACnE,QACG,MAAM,WAAW,UAAU,MAAM,WAAW,WAC7C,OAAO,MAAM,UAAU,YACvB,OAAO,SAAS,MAAM,KAAK,KAC3B,MAAM,SAAS,KACf,MAAM,SAAS;AAEnB;;;;;;;;;;;;;;;;;;;;;;;AAwBA,SAAgB,oCACd,MACA,OAA0C,CAAC,GACgB;CAC3D,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,mBAAmB,IAAI,IAAI,KAAK,uBAAuB,CAAC,GAAG,4BAA4B,CAAC;CAC9F,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,aAAa,KAAK,wBAAwB;CAChD,MAAM,YAAY,KAAK,qBAAqB;CAE5C,OAAO,KAAK,KAAK,QAAQ;EACvB,MAAM,UAAU,gBAAgB,GAAG;EAInC,MAAM,WAAW,IAAI,QAAQ,aAAa,KAAA,IAAY,CAAC,IAAK,EAAE,kBAAkB,KAAK;EAMrF,MAAM,QAAQ,UAA2B,aAAa,UAAU,IAAI;EAEpE,MAAM,cAAsD,CAAC;EAC7D,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,IAAI,QAAQ,GAAG,GACjD,IACE,EAAE,WAAW,QAAQ,KACrB,CAAC,EAAE,SAAS,KAAK,CAAC,KAClB,OAAO,MAAM,YACb,OAAO,SAAS,CAAC,GAEjB,YAAY,KAAK;GAAE,MAAM,EAAE,MAAM,CAAe;GAAG,OAAO;EAAE,CAAC;EAGjE,MAAM,MAAM,YAAY,QAAQ,MAAM,iBAAiB,IAAI,EAAE,IAAI,CAAC;EAElE,IAAI,IAAI,WAAW,GAAG;GACpB,MAAM,QAAQ,IAAI;GAClB,MAAM,QAAQ,KAAKA,UAAQ,MAAM,KAAK,CAAC;GACvC,OAAO;IACL,OAAO,IAAI;IACX,QAAQ;KACN;KACA,QAAQ,UAAU,MAAM,IAAI;KAC5B,aAAa;KACb,YAAY;KACZ,QAAQ,MAAM;KACd,YAAY,GAAG,MAAM,OAAO,MAAM;KAClC,eAAe;KACf,GAAG;IACL;GACF;EACF;EACA,IAAI,IAAI,SAAS,GAAG;GAClB,MAAM,QAAQ,KAAKA,UAAQ,IAAI,QAAQ,GAAG,MAAM,IAAI,EAAE,OAAO,CAAC,IAAI,IAAI,MAAM,CAAC;GAG7E,MAAM,aAAqC,OAAO,YAChD,IAAI,KAAK,MAAM,CAAC,EAAE,MAAM,KAAKA,UAAQ,EAAE,KAAK,CAAC,CAAC,CAAC,CACjD;GACA,OAAO;IACL,OAAO,IAAI;IACX,QAAQ;KACN;KACA,QAAQ;KACR,aAAa;KACb,YAAY;KACZ,QAAQ,IAAI,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,GAAG;KACvC;KACA,eAAe;KACf,GAAG;IACL;GACF;EACF;EACA,IAAI,CAAC,iBAAiB,OAAO;GAAE,OAAO,IAAI;GAAO,QAAQ;EAAK;EAO9D,MAAM,UAAU,YAAY,cAAc,GAAG,IAAI,cAAc,GAAG;EAClE,IAAI,OAAO,YAAY,YAAY,CAAC,OAAO,SAAS,OAAO,GACzD,OAAO;GAAE,OAAO,IAAI;GAAO,QAAQ;EAAK;EAE1C,MAAM,eAAeA,UAAQ,OAAO;EACpC,OAAO;GACL,OAAO,IAAI;GACX,QAAQ;IACN,OAAO;IACP,QAAQ;IACR,aAAa;IACb,YAAY;IACZ,QAAQ;IACR,YAAY,EAAE,qBAAqB,aAAa;IAChD,eAAe;IACf,GAAG;GACL;EACF;CACF,CAAC;AACH;;;;;;;;;;AAWA,SAAgB,gCACd,MACA,OAA0C,CAAC,GACU;CACrD,MAAM,WAAW,oCAAoC,MAAM;EAAE,GAAG;EAAM,iBAAiB;CAAM,CAAC;CAC9F,MAAM,MAA2D,CAAC;CAClE,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,IAAI,SAAS;EACnB,IAAI,EAAE,UAAU,EAAE,OAAO,gBAAgB,iBACvC,IAAI,KAAK;GAAE,KAAK,KAAK;GAAK,QAAQ,EAAE;EAAO,CAAC;CAEhD;CACA,OAAO;AACT;AAEA,SAASA,UAAQ,GAAmB;CAClC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACnVA,MAAM,iBAAiB,MAAgC;CASrD,MAAM,IAAI,cAAc,CAAC;CACzB,OAAO,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,IAAI,IAAI;AAC3D;AAEA,SAAgB,oBAAoB,OAAsD;CACxF,MAAM,UAAU,MAAM,WAAW;CACjC,MAAM,UAAU,MAAM;CACtB,MAAM,MAAM,MAAM,YAAY,WAAW;CACzC,MAAM,MAAM,MAAM,YAAY,UAAU;CAExC,MAAM,OAAO,MAAM,KAAK,QAAQ,QAAQ,aAAa,QAAQ,GAAG,CAAC,CAAC;CAClE,MAAM,IAAI,KAAK;CACf,IAAI,IAAI,GACN,OAAO;EACL,UAAU,CAAC;EACX,kBAAkB,CAAC;EACnB,SAAS;EACT;EACA,WAAW,CAAC,0CAA0C,EAAE,yBAAyB;CACnF;CAEF,MAAM,aAAa,KAAK,IAAI,GAAG,MAAM,cAAc,KAAK,IAAI,IAAI,KAAK,MAAM,IAAI,CAAC,CAAC,CAAC;CAClF,MAAM,SAAS,KAAK,MAAM,GAAG,IAAI,UAAU;CAC3C,MAAM,QAAQ,KAAK,MAAM,IAAI,UAAU;CAEvC,MAAM,WAAmC,CAAC;CAG1C,IAAI,SAAS;EACX,MAAM,cAAc,OAAO,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EAC3D,MAAM,aAAa,MAAM,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EACzD,MAAM,cAAc,OAAO,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EAC3D,MAAM,aAAa,MAAM,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EACzD,IACE,YAAY,UAAU,KACtB,WAAW,UAAU,KACrB,YAAY,UAAU,KACtB,WAAW,UAAU,GACrB;GACA,MAAM,aAAa,KAAK,UAAU,IAAI,KAAK,WAAW;GACtD,MAAM,aAAa,KAAK,UAAU,IAAI,KAAK,WAAW;GAGtD,MAAM,MAAM,KAAK,IAAI,GAAG,aAAa,UAAU;GAC/C,MAAM,WAAW,QAAQ,MAAM,CAAC;GAChC,SAAS,KAAK;IACZ,QAAQ;IACR;IACA,SACE,YAAY,MACR,wBAAwB,WAAW,QAAQ,CAAC,EAAE,0BAA0B,WAAW,QAAQ,CAAC,EAAE,yBAC9F,yCAAyC,WAAW,QAAQ,CAAC,EAAE,UAAU,WAAW,QAAQ,CAAC,EAAE;IACrG,QAAQ;KACN;KACA;KACA;KACA,SAAS,YAAY;KACrB,QAAQ,WAAW;IACrB;GACF,CAAC;EACH;CACF;CAGA;EACE,MAAM,UAAU,OAAO,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EACvD,MAAM,SAAS,MAAM,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY;EACrD,IAAI,QAAQ,UAAU,KAAK,OAAO,UAAU,GAAG;GAC7C,MAAM,KAAK,YAAY,SAAS,MAAM;GAItC,MAAM,WAAW,QAAQ,KAAK,EAAG;GACjC,SAAS,KAAK;IACZ,QAAQ;IACR;IACA,SACE,YAAY,MACR,MAAM,GAAG,QAAQ,CAAC,EAAE,8DACpB,MAAM,GAAG,QAAQ,CAAC,EAAE;IAC1B,QAAQ;KAAE;KAAI,SAAS,QAAQ;KAAQ,QAAQ,OAAO;IAAO;GAC/D,CAAC;EACH;CACF;CAGA;EACE,MAAM,cAAc,MAAM,qBAAqB,iBAAiB,MAAM,uBAAuB;EAC7F,MAAM,UAAU,KACb,KAAK,OAAO;GAAE,GAAG,QAAQ,CAAC;GAAG,GAAG,YAAY,CAAC;EAAE,EAAE,CAAC,CAClD,QAAQ,MAAqC,aAAa,EAAE,CAAC,KAAK,aAAa,EAAE,CAAC,CAAC;EACtF,IAAI,QAAQ,UAAU,GAAG;GAGvB,MAAM,IAAI,SAFC,QAAQ,KAAK,MAAM,EAAE,CAEZ,GADT,QAAQ,KAAK,MAAM,EAAE,CACR,CAAC;GAGzB,MAAM,WAAW,QAAQ,KAAM,KAAK,IAAI,GAAG,CAAC,CAAC;GAC7C,SAAS,KAAK;IACZ,QAAQ;IACR;IACA,SACE,YAAY,MACR,sDAAsD,EAAE,QAAQ,CAAC,EAAE,6BACnE,0CAA0C,EAAE,QAAQ,CAAC;IAC3D,QAAQ;KAAE,SAAS;KAAG,GAAG,QAAQ;IAAO;GAC1C,CAAC;EACH;CACF;CAGA;EAME,MAAM,UAAU,gCAAgC,MAAM;GACpD,GAAI,MAAM,2BAA2B,CAAC;GACtC,mBAAmB;EACrB,CAAC;EACD,IAAI,QAAQ,UAAU,GAAG;GACvB,MAAM,YAAY,QAAQ,MAAM,GAAG,KAAK,MAAM,QAAQ,SAAS,CAAC,CAAC;GAEjE,MAAM,WACJ,KAFe,QAAQ,MAAM,KAAK,MAAM,QAAQ,SAAS,CAAC,CAE9C,CAAC,CAAC,KAAK,MAAM,EAAE,OAAO,KAAK,CAAC,IAAI,KAAK,UAAU,KAAK,MAAM,EAAE,OAAO,KAAK,CAAC;GACvF,MAAM,aACJ,KAAK,MAAM,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY,CAAC,IAC5C,KAAK,OAAO,IAAI,OAAO,CAAC,CAAC,OAAO,YAAY,CAAC;GAC/C,MAAM,WAAW,KAAK,IAAI,GAAG,aAAa,QAAQ;GAClD,MAAM,WAAW,QAAQ,WAAW,CAAC;GACrC,SAAS,KAAK;IACZ,QAAQ;IACR;IACA,SACE,YAAY,MACR,gBAAgB,WAAW,QAAQ,CAAC,EAAE,+BAA+B,SAAS,QAAQ,CAAC,EAAE,mDACzF,uDAAuD,WAAW,QAAQ,CAAC,EAAE,QAAQ,SAAS,QAAQ,CAAC,EAAE;IAC/G,QAAQ;KAAE;KAAY;KAAU;KAAU,GAAG,QAAQ;IAAO;GAC9D,CAAC;EACH;CACF;CAEA,MAAM,SAAS,SAAS,QAAQ,GAAG,MAAM,KAAK,IAAI,GAAG,EAAE,QAAQ,GAAG,CAAC;CACnE,IAAI,SAAS,WAAW,GACtB,OAAO;EACL;EACA,kBAAkB,CAAC;EACnB,SAAS;EACT,WAAW,CAAC,0DAA0D,EAAE,EAAE;EAC1E;CACF;CAEF,MAAM,UACJ,UAAU,MAAM,WAAW,UAAU,MAAM,YAAY;CACzD,MAAM,YAAY,SACf,QAAQ,MAAM,EAAE,YAAY,GAAG,CAAC,CAChC,KAAK,MAAM,GAAG,EAAE,OAAO,aAAa,EAAE,SAAS,QAAQ,CAAC,EAAE,KAAK,EAAE,SAAS;CAC7E,IAAI,UAAU,WAAW,GAAG,UAAU,KAAK,0CAA0C;CAErF,OAAO;EACL;EACA,kBAAkB,SAAS,KAAK,YAAY,QAAQ,MAAM;EAC1D;EACA;EACA;CACF;AACF;AAIA,SAAS,KAAK,IAAsB;CAClC,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,OAAO,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,GAAG;AAC5C;AAEA,SAAS,aAAa,OAAuC;CAC3D,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK;AAC3D;AAEA,SAAS,QAAQ,GAAmB;CAClC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;AAEA,SAAS,YAAY,GAAa,GAAqB;CAErD,MAAM,UAAU,CAAC,GAAG,CAAC,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,MAAM,UAAU,CAAC,GAAG,CAAC,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,MAAM,MAAM,CAAC,mBAAG,IAAI,IAAI,CAAC,GAAG,SAAS,GAAG,OAAO,CAAC,CAAC,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CACvE,IAAI,MAAM;CACV,KAAK,MAAM,KAAK,KAAK;EACnB,MAAM,KAAK,QAAQ,QAAQ,MAAM,KAAK,CAAC,CAAC,CAAC,SAAS,QAAQ;EAC1D,MAAM,KAAK,QAAQ,QAAQ,MAAM,KAAK,CAAC,CAAC,CAAC,SAAS,QAAQ;EAC1D,MAAM,KAAK,IAAI,KAAK,KAAK,IAAI,KAAK,EAAE,CAAC;CACvC;CACA,OAAO;AACT;AAEA,SAAS,iBACP,gBACmC;CACnC,QAAQ,QAAmB;EAIzB,MAAM,WAAW,gCAAgC,CAAC,GAAG,GAAG;GACtD,GAAI,kBAAkB,CAAC;GACvB,mBAAmB;EACrB,CAAC;EACD,OAAO,SAAS,WAAW,IAAI,SAAS,EAAE,CAAE,OAAO,QAAQ;CAC7D;AACF"}
|