@tangle-network/agent-eval 0.179.0 → 0.181.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +119 -146
- package/dist/adapters/http.d.ts +2 -2
- package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
- package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
- package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
- package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
- package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
- package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +4 -4
- package/dist/ast-CP9ae9B0.js +557 -0
- package/dist/ast-CP9ae9B0.js.map +1 -0
- package/dist/ast-hI-vjW6J.d.ts +457 -0
- package/dist/ast-hI-vjW6J.d.ts.map +1 -0
- package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
- package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
- package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
- package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
- package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
- package/dist/cli.js +5 -8
- package/dist/cli.js.map +1 -1
- package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
- package/dist/client-CXE-U1SA.js.map +1 -0
- package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
- package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +11 -10
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
- package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
- package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
- package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
- package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
- package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
- package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
- package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
- package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
- package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
- package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
- package/dist/engine-DS1cysJy.d.ts.map +1 -0
- package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
- package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
- package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
- package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +27 -477
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +95 -559
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
- package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
- package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
- package/dist/index-Bp_6sj3x.d.ts.map +1 -0
- package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
- package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
- package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
- package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
- package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
- package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
- package/dist/index.d.ts +28 -28
- package/dist/index.js +25 -16
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
- package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
- package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
- package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
- package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
- package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
- package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
- package/dist/journal-Cs9f7385.js.map +1 -0
- package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
- package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
- package/dist/llm-judge-DEFZeSiu.js.map +1 -0
- package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
- package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +138 -7
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +245 -97
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
- package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/outcome-store-BXlkwMPR.js +131 -0
- package/dist/outcome-store-BXlkwMPR.js.map +1 -0
- package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
- package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
- package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
- package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
- package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
- package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
- package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
- package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
- package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
- package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
- package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
- package/dist/report-command-V1ecVgAv.js.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +3 -3
- package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
- package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
- package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
- package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
- package/dist/rl.d.ts +53 -99
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +182 -169
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
- package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
- package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
- package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
- package/dist/run-record-Br-Yzt_k.js +464 -0
- package/dist/run-record-Br-Yzt_k.js.map +1 -0
- package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
- package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
- package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
- package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
- package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
- package/dist/sequential-DAsyV2T9.js.map +1 -0
- package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
- package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
- package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
- package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
- package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
- package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
- package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
- package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
- package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
- package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
- package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +4 -2
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +3 -3
- package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
- package/dist/terminal-record-BtPwKTSr.js.map +1 -0
- package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
- package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
- package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
- package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
- package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
- package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
- package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
- package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
- package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
- package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
- package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
- package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
- package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/docs/adapters-observability.md +14 -0
- package/docs/campaign-proposers.md +86 -128
- package/docs/charter.md +108 -112
- package/docs/concepts.md +157 -69
- package/docs/design/mlbenchmarks-book-review.md +440 -0
- package/docs/design/mlbenchmarks-review/observations.json +713 -0
- package/docs/design/mlbenchmarks-review/probes.mts +476 -0
- package/docs/design/mlbenchmarks-review/sources.json +200 -0
- package/docs/design/self-improvement-evidence-audit.md +263 -0
- package/docs/design.md +2 -1
- package/docs/eval-surface-map.md +95 -42
- package/docs/evaluation-integrity.md +220 -0
- package/docs/experiment.md +111 -55
- package/docs/feature-guide.md +5 -6
- package/docs/hosted-ingest-spec.md +4 -11
- package/docs/insight-report.md +187 -455
- package/docs/outcome-validity.md +182 -0
- package/docs/product-eval-adoption.md +1 -2
- package/docs/research-report-methodology.md +7 -7
- package/docs/search-history-receipts.md +8 -0
- package/docs/statistical-evidence.md +129 -0
- package/docs/verdicts.md +76 -49
- package/package.json +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
- package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
- package/dist/client-BlLY6o2w.js.map +0 -1
- package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
- package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
- package/dist/engine-CX8ReXkn.d.ts.map +0 -1
- package/dist/index-BxWvILU8.d.ts.map +0 -1
- package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
- package/dist/ledger-core-Cs9f7385.js.map +0 -1
- package/dist/llm-judge-v80Kmu9g.js.map +0 -1
- package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
- package/dist/outcome-store-ChBKlTd_.js +0 -75
- package/dist/outcome-store-ChBKlTd_.js.map +0 -1
- package/dist/promotion-policy-DWOm70gx.js.map +0 -1
- package/dist/report-command-DKlXfU5r.js.map +0 -1
- package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
- package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
- package/dist/run-record-CR63CpHK.js +0 -216
- package/dist/run-record-CR63CpHK.js.map +0 -1
- package/dist/sequential-B5gXgcyp.js.map +0 -1
- package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
- package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
- package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","names":[],"sources":["../../src/meta-eval/calibration.ts","../../src/meta-eval/correlation-study.ts","../../src/meta-eval/plants.ts","../../src/meta-eval/sentinel.ts"],"sourcesContent":["/**\n * Calibration curve — binned \"if eval says X, what does reality show?\"\n *\n * Companion to correlationStudy. Raw correlation is a single number;\n * the calibration curve shows *where* the eval is well-calibrated vs\n * overconfident / underconfident. Buckets the eval metric, computes\n * mean outcome per bucket, reports expected-calibration-error (ECE).\n */\n\nimport { runMetricExtractor } from '../trace/query'\nimport type { TraceStore } from '../trace/store'\nimport type { EvalMetricSpec } from './correlation-study'\nimport type { DeploymentOutcome, OutcomeStore } from './outcome-store'\n\nexport interface CalibrationBin {\n lower: number\n upper: number\n n: number\n evalMean: number\n outcomeMean: number\n /** |outcomeMean − evalMean|; contributes to ECE weighted by n/total. */\n gap: number\n}\n\nexport interface CalibrationReport {\n evalMetric: string\n outcomeMetric: string\n n: number\n bins: CalibrationBin[]\n /** Expected Calibration Error — Σ (n_i/N) × |outcomeMean_i − evalMean_i|. */\n ece: number\n /** Max bin gap — upper bound on miscalibration. */\n maxGap: number\n}\n\nexport interface CalibrationOptions {\n bins?: number\n /** Equal-width (fixed bin edges) or equal-frequency (quantile bins). */\n binning?: 'equal-width' | 'equal-frequency'\n /** Clip eval values to [lo, hi] before binning. */\n range?: { lo: number; hi: number }\n}\n\nexport interface CalibrationPair {\n evalScore: number\n outcome: number\n}\n\nexport async function calibrationCurve(\n traceStore: TraceStore,\n outcomeStore: OutcomeStore,\n evalMetric: EvalMetricSpec,\n outcomeMetric: string,\n options: CalibrationOptions = {},\n): Promise<CalibrationReport | null> {\n const runs = await traceStore.listRuns()\n const outcomes = await outcomeStore.list()\n const byRun = new Map<string, DeploymentOutcome[]>()\n for (const o of outcomes) {\n const arr = byRun.get(o.runId) ?? []\n arr.push(o)\n byRun.set(o.runId, arr)\n }\n\n const extract = evalMetric.extract ?? runMetricExtractor(evalMetric.id)\n const pairs: Array<{ x: number; y: number }> = []\n for (const run of runs) {\n const os = byRun.get(run.runId)\n if (!os?.length) continue\n const x = await extract(run, traceStore)\n if (x === null || !Number.isFinite(x)) continue\n const latest = [...os].sort((a, b) => b.capturedAt - a.capturedAt)[0]!\n const y = latest.metrics[outcomeMetric]\n if (typeof y !== 'number' || !Number.isFinite(y)) continue\n pairs.push({ x, y })\n }\n if (pairs.length < 2) return null\n\n return calibrationFromPairs(\n pairs.map((p) => ({ evalScore: p.x, outcome: p.y })),\n evalMetric.id,\n outcomeMetric,\n options,\n )\n}\n\nfunction calibrationFromPairs(\n inputPairs: CalibrationPair[],\n evalMetric: string,\n outcomeMetric: string,\n options: CalibrationOptions = {},\n): CalibrationReport | null {\n const pairs = inputPairs.filter(\n (pair) => Number.isFinite(pair.evalScore) && Number.isFinite(pair.outcome),\n )\n if (pairs.length < 2) return null\n\n const numBins = options.bins ?? 10\n const binning = options.binning ?? 'equal-width'\n const xs = pairs.map((p) => p.evalScore)\n const lo = options.range?.lo ?? Math.min(...xs)\n const hi = options.range?.hi ?? Math.max(...xs)\n\n const bins: CalibrationBin[] = []\n if (binning === 'equal-frequency') {\n const sorted = [...pairs].sort((a, b) => a.evalScore - b.evalScore)\n const perBin = Math.max(1, Math.floor(sorted.length / numBins))\n for (let i = 0; i < sorted.length; i += perBin) {\n const chunk = sorted.slice(i, i + perBin)\n if (chunk.length === 0) continue\n bins.push(toBin(chunk))\n }\n } else {\n const width = (hi - lo) / numBins\n if (width === 0) return null\n for (let i = 0; i < numBins; i++) {\n const binLo = lo + i * width\n const binHi = i === numBins - 1 ? hi + 1e-9 : lo + (i + 1) * width\n const chunk = pairs.filter((p) => p.evalScore >= binLo && p.evalScore < binHi)\n if (chunk.length === 0) continue\n bins.push(toBin(chunk, binLo, binHi))\n }\n }\n\n const total = bins.reduce((a, b) => a + b.n, 0)\n const ece = bins.reduce((a, b) => a + (b.n / total) * b.gap, 0)\n const maxGap = bins.reduce((a, b) => Math.max(a, b.gap), 0)\n\n return { evalMetric, outcomeMetric, n: pairs.length, bins, ece, maxGap }\n}\n\nfunction toBin(chunk: CalibrationPair[], lower?: number, upper?: number): CalibrationBin {\n const xs = chunk.map((c) => c.evalScore)\n const ys = chunk.map((c) => c.outcome)\n const evalMean = mean(xs)\n const outcomeMean = mean(ys)\n return {\n lower: lower ?? Math.min(...xs),\n upper: upper ?? Math.max(...xs),\n n: chunk.length,\n evalMean,\n outcomeMean,\n gap: Math.abs(outcomeMean - evalMean),\n }\n}\n\nfunction mean(xs: number[]): number {\n return xs.reduce((a, b) => a + b, 0) / xs.length\n}\n","/**\n * Correlation study — \"does our eval score predict real-world outcomes?\"\n *\n * This is the load-bearing signal. Takes a TraceStore + OutcomeStore,\n * joins on runId, computes Pearson + Spearman + bootstrap CI for every\n * (evalMetric, outcomeMetric) pair the caller declares.\n *\n * Without this number the framework is ornamental. With it and r > 0.6\n * the framework is a moat — no other agent-eval tool publishes one.\n */\n\nimport { pearsonR, spearmanR } from '../statistics'\nimport { makeRng } from '../statistics/internal'\nimport { runMetricExtractor } from '../trace/query'\nimport type { Run } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport type { DeploymentOutcome, OutcomeFilter, OutcomeStore } from './outcome-store'\n\nexport interface EvalMetricSpec {\n id: string\n /** Extract a scalar from a run. Omit it and `id` must name one of\n * `RUN_METRICS`; any other `id` is refused. */\n extract?: (run: Run, store: TraceStore) => Promise<number | null>\n}\n\nexport interface OutcomePair {\n evalMetric: string\n outcomeMetric: string\n}\n\nexport interface CorrelationResult {\n evalMetric: string\n outcomeMetric: string\n n: number\n pearson: number\n spearman: number\n /** 95% bootstrap CI for Pearson. */\n pearsonCi95: { lower: number; upper: number }\n /** Rough verdict: 'strong' ≥ 0.7, 'moderate' ≥ 0.4, else 'weak'. */\n verdict: 'strong' | 'moderate' | 'weak'\n}\n\nexport interface CorrelationStudyResult {\n pairs: CorrelationResult[]\n joinedSamples: number\n skippedRuns: number\n}\n\nexport interface CorrelationStudyOptions {\n /** Only join outcomes captured within this window after run.startedAt. */\n maxCaptureLagMs?: number\n /** Restrict to a subset of outcomes (cohort, region, source). */\n outcomeFilter?: OutcomeFilter\n /** Which outcome per run to use when multiple exist. Default 'latest'. */\n reduction?: 'latest' | 'mean' | 'max'\n /** Bootstrap iterations for the CI. Default 500. */\n bootstrapIterations?: number\n /** Seed for the bootstrap resampler. Absent, the seed is derived from the\n * paired observations, so the same study reproduces the same interval. */\n seed?: number\n}\n\nexport async function correlationStudy(\n traceStore: TraceStore,\n outcomeStore: OutcomeStore,\n evalMetrics: EvalMetricSpec[],\n outcomeMetricNames: string[],\n options: CorrelationStudyOptions = {},\n): Promise<CorrelationStudyResult> {\n const runs = await traceStore.listRuns()\n const outcomes = await outcomeStore.list(options.outcomeFilter)\n const outcomesByRun = new Map<string, DeploymentOutcome[]>()\n for (const o of outcomes) {\n const arr = outcomesByRun.get(o.runId) ?? []\n arr.push(o)\n outcomesByRun.set(o.runId, arr)\n }\n\n const reduction = options.reduction ?? 'latest'\n const maxLag = options.maxCaptureLagMs ?? Infinity\n\n const pairs: Array<{ evalMetric: string; outcomeMetric: string; xs: number[]; ys: number[] }> = []\n for (const em of evalMetrics) {\n for (const om of outcomeMetricNames) {\n pairs.push({ evalMetric: em.id, outcomeMetric: om, xs: [], ys: [] })\n }\n }\n\n let joined = 0\n let skipped = 0\n for (const run of runs) {\n const os = outcomesByRun.get(run.runId)\n if (!os || os.length === 0) {\n skipped++\n continue\n }\n const eligible = os.filter((o) => o.capturedAt - run.startedAt <= maxLag)\n if (eligible.length === 0) {\n skipped++\n continue\n }\n\n for (const em of evalMetrics) {\n const extract = em.extract ?? runMetricExtractor(em.id)\n const x = await extract(run, traceStore)\n if (x === null || !Number.isFinite(x)) continue\n\n for (const om of outcomeMetricNames) {\n const values = eligible\n .map((o) => o.metrics[om])\n .filter((v): v is number => typeof v === 'number' && Number.isFinite(v))\n if (values.length === 0) continue\n const y = reduce(values, reduction, eligible)\n if (y === null) continue\n const pair = pairs.find((p) => p.evalMetric === em.id && p.outcomeMetric === om)!\n pair.xs.push(x)\n pair.ys.push(y)\n }\n }\n joined++\n }\n\n const results: CorrelationResult[] = pairs\n .filter((p) => p.xs.length >= 3)\n .map((p) => {\n const pearson = pearsonR(p.xs, p.ys)\n const spearman = spearmanR(p.xs, p.ys)\n const pearsonCi95 = bootstrapPearsonCi(\n p.xs,\n p.ys,\n options.bootstrapIterations ?? 500,\n options.seed,\n )\n const verdict: CorrelationResult['verdict'] =\n Math.abs(pearson) >= 0.7 ? 'strong' : Math.abs(pearson) >= 0.4 ? 'moderate' : 'weak'\n return {\n evalMetric: p.evalMetric,\n outcomeMetric: p.outcomeMetric,\n n: p.xs.length,\n pearson,\n spearman,\n pearsonCi95,\n verdict,\n }\n })\n\n return { pairs: results, joinedSamples: joined, skippedRuns: skipped }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────\n\nfunction reduce(\n values: number[],\n kind: 'latest' | 'mean' | 'max',\n outcomes: DeploymentOutcome[],\n): number | null {\n if (values.length === 0) return null\n if (kind === 'mean') return values.reduce((a, b) => a + b, 0) / values.length\n if (kind === 'max') return Math.max(...values)\n // 'latest': pick the outcome captured last, then lookup its metric\n const latest = [...outcomes].sort((a, b) => b.capturedAt - a.capturedAt)[0]\n if (!latest) return null\n const latestKey = Object.keys(latest.metrics)[0]\n const v = latestKey !== undefined ? latest.metrics[latestKey] : undefined\n // For 'latest' we already have `values` aligned; use the last-captured one\n const paired = outcomes\n .map((o) => {\n const k = Object.keys(o.metrics)[0]\n return {\n at: o.capturedAt,\n v: k !== undefined ? values.find((x) => o.metrics[k] === x) : undefined,\n }\n })\n .filter((p) => p.v !== undefined)\n if (paired.length === 0) return v ?? null\n return paired.sort((a, b) => b.at - a.at)[0]?.v ?? null\n}\n\nfunction bootstrapPearsonCi(\n xs: number[],\n ys: number[],\n iterations: number,\n seed: number | undefined,\n): { lower: number; upper: number } {\n const n = xs.length\n if (n < 3) return { lower: NaN, upper: NaN }\n const rng = makeRng(seed, xs, ys)\n const rs: number[] = []\n for (let b = 0; b < iterations; b++) {\n const rx: number[] = new Array(n)\n const ry: number[] = new Array(n)\n for (let i = 0; i < n; i++) {\n const idx = Math.floor(rng() * n)\n rx[i] = xs[idx]!\n ry[i] = ys[idx]!\n }\n const r = pearsonR(rx, ry)\n if (Number.isFinite(r)) rs.push(r)\n }\n rs.sort((a, b) => a - b)\n if (rs.length === 0) return { lower: NaN, upper: NaN }\n return {\n lower: rs[Math.floor(0.025 * rs.length)]!,\n upper: rs[Math.min(rs.length - 1, Math.floor(0.975 * rs.length))]!,\n }\n}\n","/**\n * Plants — seeded known-wrong items that measure the grader, not the work.\n *\n * A grading run reports how the work scored. It cannot report whether the\n * grader would have noticed a wrong answer, because every item it saw was\n * authored in good faith. A plant closes that hole: an item authored wrong by\n * construction is mixed into the live set, graded by the same path as\n * everything else, and the share of plants the grader refused is the catch\n * rate.\n *\n * Measured motive: a sibling lab ran a deliverable gate that accepted any\n * non-empty submission. It produced six false certifications in seventeen\n * deliveries, and no agent lied — the gate never asked a question the format\n * could fail. A catch rate is the number that would have shown it on day one.\n *\n * This module composes existing primitives rather than adding parallel ones:\n *\n * - A plant IS a {@link GoldenItem} from `../judge-calibration`. Its\n * `humanScore` is the grade a working grader owes the item, so the same\n * array feeds `calibrateJudge` unchanged.\n * - The grader's output is `CandidateScore[]`, the array `calibrateJudge` and\n * `snapshotFromSentinelSet` already consume.\n * - \"Caught\" is `snapshotFromSentinelSet`'s join with the labels inverted:\n * the grade lands on the side of `acceptThreshold` the seed demands.\n * - The manifest is sealed with `hashCanonical` from `../ledger-core/canonical`,\n * the digest the sealed-experiment path uses, so the answer key cannot be\n * revised once the results are in.\n *\n * Authoring the wrong item is the step before all of that, and it is where the\n * measurement silently breaks. {@link plantByPerturbation} takes a claim a\n * grader verified and alters exactly one load-bearing value in its evidence,\n * so the grader is asked a question it must fail. Hand-rolled perturbation\n * bumps a number that is part of an identity — `python3`, `file42.txt`, `1.5`,\n * `utf-8` — and the check stops EXECUTING; the plant then measures the\n * environment rather than the grader. {@link perturbEvidence} refuses those\n * digit runs by construction.\n *\n * Blindness has two halves, and this module owns one. It never puts a plant\n * flag on a graded item: `seedPlants` returns the mixed set and a manifest,\n * and only the manifest knows which ids are seeded. Keeping the manifest out\n * of the graded workspace and publishing its `seal` before grading is the\n * caller's half; {@link catchRate} refuses a manifest whose contents no longer\n * match its seal.\n *\n * Refusals, because a catch rate that cannot refuse is not a measurement:\n *\n * - a seeded id with no result makes the report `incomplete`, never a rate\n * over the results that did come back;\n * - zero seeded plants makes it `not_evaluated`, never 1.0;\n * - a result for an id the manifest never handed out is refused outright.\n */\n\nimport { CaptureIntegrityError, ValidationError } from '../errors'\nimport type { GoldenItem } from '../judge-calibration'\nimport { hashCanonical, type LedgerHash } from '../ledger-core/canonical'\nimport { mulberry32 } from '../statistics/random'\n\n/**\n * How a plant item was authored wrong. The class is reported separately in\n * {@link CatchRateReport.byKind} because a grader is routinely sharp on one\n * and blind to another.\n */\nexport type PlantKind =\n /** A load-bearing value is altered: a number off by one, a comparison flipped. */\n | 'wrong-value'\n /** The item carries its own check, and that check passes without testing the claim. */\n | 'self-certifying'\n /** The check names an input that does not exist, so it cannot run at all. */\n | 'unreachable-input'\n /** A copy of an item already in the set, which is owed a duplicate flag rather than a second grade. */\n | 'duplicate'\n\nconst PLANT_KINDS: readonly PlantKind[] = [\n 'wrong-value',\n 'self-certifying',\n 'unreachable-input',\n 'duplicate',\n]\n\n/** What a working grader owes a seeded item. */\nexport type PlantExpectation = 'reject' | 'accept'\n\nconst PLANT_EXPECTATIONS: readonly PlantExpectation[] = ['reject', 'accept']\n\n/** The label boundary a plant record is checked against at definition time. */\nconst RECORD_LABEL_BOUNDARY = 0.5\n\nexport interface Plant {\n /** Name of the plant record. Reported in `missedIds` and `missingIds`. */\n id: string\n kind: PlantKind\n /** The seeded item, indistinguishable from a real one once mixed. */\n item: GoldenItem\n /** The verdict a working grader owes this item. */\n expectedVerdict: PlantExpectation\n}\n\n/**\n * Build one plant record and refuse an incoherent one.\n *\n * The refusal that matters is the last: a record whose `expectedVerdict`\n * disagrees with `item.humanScore` inverts the measurement silently, because\n * the same item then reads as wrong here and as correct to every calibration\n * instrument that joins on the id.\n *\n * `item` is copied field by field so a later mutation of the caller's object\n * cannot change what the manifest sealed, and `group` is dropped when it is\n * absent so the record always has a canonical JSON form.\n */\nexport function definePlant(input: {\n id: string\n kind: PlantKind\n item: GoldenItem\n expectedVerdict: PlantExpectation\n}): Plant {\n const { id, kind, item, expectedVerdict } = input\n if (typeof id !== 'string' || id.trim() === '') {\n throw new ValidationError('definePlant: id must be a non-empty string')\n }\n if (!PLANT_KINDS.includes(kind)) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has kind ${JSON.stringify(kind)}; expected one of ${PLANT_KINDS.join(', ')}`,\n )\n }\n if (!PLANT_EXPECTATIONS.includes(expectedVerdict)) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has expectedVerdict ${JSON.stringify(expectedVerdict)}; expected one of ${PLANT_EXPECTATIONS.join(', ')}`,\n )\n }\n if (typeof item.itemId !== 'string' || item.itemId.trim() === '') {\n throw new ValidationError(`definePlant: plant \"${id}\" has an empty item.itemId`)\n }\n if (!Number.isFinite(item.humanScore) || item.humanScore < 0 || item.humanScore > 1) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has humanScore ${item.humanScore}; expected a finite number in [0, 1]`,\n )\n }\n if (item.group !== undefined && typeof item.group !== 'string') {\n throw new ValidationError(`definePlant: plant \"${id}\" has a non-string item.group`)\n }\n if (item.humanScore === RECORD_LABEL_BOUNDARY) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has humanScore ${RECORD_LABEL_BOUNDARY}, which states neither a rejection nor an acceptance`,\n )\n }\n const labelSays: PlantExpectation = item.humanScore < RECORD_LABEL_BOUNDARY ? 'reject' : 'accept'\n if (labelSays !== expectedVerdict) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" expects the grader to ${expectedVerdict} it, but humanScore ${item.humanScore} says ${labelSays}`,\n )\n }\n return {\n id,\n kind,\n item: {\n itemId: item.itemId,\n humanScore: item.humanScore,\n ...(item.group === undefined ? {} : { group: item.group }),\n },\n expectedVerdict,\n }\n}\n\n/**\n * A claim's executable evidence: the command a grader runs and the output it\n * must produce.\n */\nexport interface PlantEvidence {\n /** The command a grader runs to test the claim. */\n check?: string\n /** The output that command must produce for the claim to stand. */\n expect?: string\n}\n\n/** Which evidence field {@link perturbEvidence} altered. */\nexport type PerturbedField = 'check' | 'expect'\n\n/** How {@link perturbEvidence} altered it. */\nexport type PerturbationKind = 'number-off-by-one' | 'comparison-flipped'\n\nexport interface EvidencePerturbation {\n /** The evidence with exactly one value altered. */\n evidence: PlantEvidence\n /** Which field changed. */\n field: PerturbedField\n /** How it changed. */\n how: PerturbationKind\n /**\n * The exact substring that was replaced. A comparison token carries its\n * surrounding spaces, because the spaces are what make ` -ge ` a comparison\n * and not part of a word.\n */\n original: string\n /** What replaced it. */\n perturbed: string\n}\n\n/**\n * Characters that make a digit run part of something larger than a number.\n *\n * A run that touches one of these is an identity — `python3`, `file42.txt`,\n * `1.5`, `utf-8` — and bumping it renames a program, a file, a version, or an\n * encoding instead of falsifying the claim. The check then stops EXECUTING,\n * and a plant that cannot run measures the environment rather than the grader:\n * it reads as caught, for a reason that has nothing to do with the seeded\n * defect.\n */\nconst NUMBER_GLUE = /[A-Za-z0-9_./-]/\n\n/**\n * Comparison tokens that mean one thing only, each with its inversion.\n *\n * A bare `>` or `<` is absent on purpose: in a shell check it is a\n * redirection, so flipping it rewrites where output goes rather than what the\n * test asks, and the check stops testing the claim.\n */\nconst COMPARISON_FLIPS: readonly (readonly [string, string])[] = [\n [' -ge ', ' -lt '],\n [' -gt ', ' -le '],\n [' -le ', ' -gt '],\n [' -lt ', ' -ge '],\n [' -eq ', ' -ne '],\n [' -ne ', ' -eq '],\n ['>=', '<'],\n ['<=', '>'],\n ['==', '!='],\n ['!=', '=='],\n]\n\ninterface Span {\n start: number\n end: number\n}\n\n/**\n * The last digit run in `text` that is a number and nothing else, or null.\n *\n * The last is taken because a measured value trails its label — `count=42`,\n * `n>=5 OK cells=8` — and because the choice has to be deterministic: the same\n * claim must always yield the same plant, or two runs of the same authoring\n * step seed two different answer keys for one item.\n */\nfunction lastStandaloneNumber(text: string): Span | null {\n const digits = /\\d+/g\n let found: Span | null = null\n let match = digits.exec(text)\n while (match !== null) {\n const start = match.index\n const end = start + match[0].length\n // charAt returns '' outside the string, and NUMBER_GLUE never matches ''.\n if (!NUMBER_GLUE.test(text.charAt(start - 1)) && !NUMBER_GLUE.test(text.charAt(end))) {\n found = { start, end }\n }\n match = digits.exec(text)\n }\n return found\n}\n\n/**\n * The first flippable comparison in `text`, or null.\n *\n * Any single flip inverts the test wherever it sits, so unlike the number rule\n * there is no safety ordering among the matches. The first is taken so the\n * result is fixed by the head of the command and does not move when a later\n * comparison is added.\n */\nfunction firstComparison(text: string): { span: Span; flipped: string } | null {\n for (let index = 0; index < text.length; index += 1) {\n for (const [token, flipped] of COMPARISON_FLIPS) {\n if (text.startsWith(token, index)) {\n return { span: { start: index, end: index + token.length }, flipped }\n }\n }\n }\n return null\n}\n\nfunction replaceSpan(text: string, span: Span, value: string): string {\n return text.slice(0, span.start) + value + text.slice(span.end)\n}\n\n/**\n * Rebuild the evidence with one field replaced. An absent field is dropped\n * rather than set to undefined, so the record always has a canonical JSON form.\n */\nfunction withField(\n check: string | undefined,\n expect: string | undefined,\n field: PerturbedField,\n value: string,\n): PlantEvidence {\n const nextCheck = field === 'check' ? value : check\n const nextExpect = field === 'expect' ? value : expect\n return {\n ...(nextCheck === undefined ? {} : { check: nextCheck }),\n ...(nextExpect === undefined ? {} : { expect: nextExpect }),\n }\n}\n\nfunction offByOne(\n check: string | undefined,\n expect: string | undefined,\n field: PerturbedField,\n text: string,\n span: Span,\n): EvidencePerturbation {\n const original = text.slice(span.start, span.end)\n const perturbed = (BigInt(original) + 1n).toString()\n return {\n evidence: withField(check, expect, field, replaceSpan(text, span, perturbed)),\n field,\n how: 'number-off-by-one',\n original,\n perturbed,\n }\n}\n\n/**\n * Derive a known-wrong version of a claim's evidence by altering exactly one\n * value, or return null when no value can be altered safely.\n *\n * Exactly one value changes. A perturbation that moves two does not name the\n * defect it seeded, so a grader that catches it says nothing about which defect\n * the grader can see.\n *\n * The preference order IS the safety order:\n *\n * 1. the last standalone number in `expect` — it falsifies the claim while the\n * command still runs exactly as it did;\n * 2. a flipped comparison in `check` — it inverts a test that was passing;\n * 3. the last standalone number in `check` — last because a number in a command\n * can name an input (a port, a width, a version) rather than a threshold,\n * and renaming an input breaks execution, not the claim.\n *\n * \"Standalone\" excludes a digit run glued to a word, a dot, a slash, or a\n * hyphen: `python3`, `file42.txt`, `1.5`, `utf-8`. That exclusion is why this\n * helper exists. Hand-rolled perturbation bumps one of those, the check stops\n * executing, and the plant measures the environment instead of the grader.\n *\n * A number is bumped with BigInt, so a value wider than\n * `Number.MAX_SAFE_INTEGER` moves by exactly one. Bumped as a `number` it\n * rounds back to itself, and the plant is then a correct claim the grader is\n * right to verify.\n */\nexport function perturbEvidence(evidence: PlantEvidence): EvidencePerturbation | null {\n const check = typeof evidence.check === 'string' ? evidence.check : undefined\n const expect = typeof evidence.expect === 'string' ? evidence.expect : undefined\n\n if (expect !== undefined) {\n const span = lastStandaloneNumber(expect)\n if (span !== null) return offByOne(check, expect, 'expect', expect, span)\n }\n if (check !== undefined) {\n const comparison = firstComparison(check)\n if (comparison !== null) {\n return {\n evidence: withField(\n check,\n expect,\n 'check',\n replaceSpan(check, comparison.span, comparison.flipped),\n ),\n field: 'check',\n how: 'comparison-flipped',\n original: check.slice(comparison.span.start, comparison.span.end),\n perturbed: comparison.flipped,\n }\n }\n const span = lastStandaloneNumber(check)\n if (span !== null) return offByOne(check, expect, 'check', check, span)\n }\n return null\n}\n\nexport interface PerturbedPlant {\n plant: Plant\n /** The perturbed evidence to write into the item the grader is handed. */\n evidence: PlantEvidence\n field: PerturbedField\n how: PerturbationKind\n original: string\n perturbed: string\n}\n\n/**\n * Derive a reject-direction plant from a claim a grader verified, by perturbing\n * one value.\n *\n * {@link definePlant} takes a {@link GoldenItem} and never authors a wrong item\n * from a right one, so until now every caller hand-rolled the perturbation.\n * This is that step, done once and the same way every time: the claim's\n * evidence is altered by {@link perturbEvidence}, the result is a `wrong-value`\n * plant the grader owes a `reject`, and the record itself is built by\n * `definePlant` so every refusal there still applies — in particular the one\n * that catches an `expectedVerdict` its `humanScore` contradicts.\n *\n * Returns null when nothing in the evidence can be perturbed. That is not a\n * failure: a claim with no executable evidence cannot be made wrong in a way a\n * grader could execute, and seeding it would measure prose.\n *\n * Refuses an empty `id` or `itemId`. Both name the record — one in the manifest\n * and in `missedIds`, the other in the join with the results — and a miss\n * nobody can look up is not a measurement.\n */\nexport function plantByPerturbation(input: {\n id: string\n itemId: string\n evidence: PlantEvidence\n /** Score for the perturbed item. Must sit below the run's accept threshold. Default 0. */\n humanScore?: number\n group?: string\n}): PerturbedPlant | null {\n const { id, itemId, evidence, humanScore = 0, group } = input\n if (typeof id !== 'string' || id.trim() === '') {\n throw new ValidationError('plantByPerturbation: id must be a non-empty string')\n }\n if (typeof itemId !== 'string' || itemId.trim() === '') {\n throw new ValidationError(`plantByPerturbation: plant \"${id}\" has an empty itemId`)\n }\n const perturbation = perturbEvidence(evidence)\n if (perturbation === null) return null\n const plant = definePlant({\n id,\n kind: 'wrong-value',\n item: { itemId, humanScore, ...(group === undefined ? {} : { group }) },\n expectedVerdict: 'reject',\n })\n return {\n plant,\n evidence: perturbation.evidence,\n field: perturbation.field,\n how: perturbation.how,\n original: perturbation.original,\n perturbed: perturbation.perturbed,\n }\n}\n\nexport interface PlantManifest {\n /**\n * Digest over the seeded order, the plants, and the threshold. Publish it\n * before the grading run: a manifest edited afterwards to match the results\n * no longer matches its seal, and {@link catchRate} refuses it.\n */\n seal: LedgerHash\n /** The seed that fixed the mix order. */\n seed: number\n /** The grade at or above which the graded policy's own gate accepts an item. */\n acceptThreshold: number\n /** Every item id in the seeded set, in the order handed out. */\n itemIds: string[]\n /** The seeded plants. This is the answer key; keep it out of the graded workspace. */\n plants: Plant[]\n}\n\nexport interface SeededGradingSet {\n /**\n * The mixed set in seeded order. Operator-side: it still carries every\n * item's `humanScore`, so hand the graded policy the payload each `itemId`\n * names, never this array.\n */\n items: GoldenItem[]\n manifest: PlantManifest\n}\n\nexport interface SeedPlantsOptions {\n /**\n * Fixes the mix order. The same dataset, plants, threshold, and seed always\n * produce the same seeded order and the same seal.\n */\n seed?: number\n /**\n * The grade at or above which the graded policy's own gate accepts an item.\n * Supply the threshold your gate uses; the default suits a judge scoring in\n * [0, 1] with a pass at the midpoint.\n */\n acceptThreshold?: number\n}\n\n/**\n * Mix plants into a grading set and seal which items they are.\n *\n * Refusals: a duplicate id anywhere in the mixed set (the join is by id, so a\n * repeat makes one of the two unscoreable), a plant whose item id collides\n * with a dataset item (it would shadow real work and grade it as a plant), a\n * dataset item with no usable label, and a plant whose expectation the run's\n * `acceptThreshold` contradicts.\n */\nexport function seedPlants(\n dataset: readonly GoldenItem[],\n plants: readonly Plant[],\n options: SeedPlantsOptions = {},\n): SeededGradingSet {\n const seed = options.seed ?? 7\n const acceptThreshold = options.acceptThreshold ?? 0.5\n if (!Number.isFinite(seed)) {\n throw new ValidationError(`seedPlants: seed must be a finite number, got ${seed}`)\n }\n if (!Number.isFinite(acceptThreshold) || acceptThreshold <= 0 || acceptThreshold > 1) {\n throw new ValidationError(\n `seedPlants: acceptThreshold must be a finite number in (0, 1], got ${acceptThreshold}`,\n )\n }\n\n const datasetItems: GoldenItem[] = []\n const seen = new Set<string>()\n for (const item of dataset) {\n if (typeof item.itemId !== 'string' || item.itemId.trim() === '') {\n throw new ValidationError('seedPlants: a dataset item has an empty itemId')\n }\n if (!Number.isFinite(item.humanScore)) {\n throw new ValidationError(\n `seedPlants: dataset item \"${item.itemId}\" has a non-finite humanScore`,\n )\n }\n if (seen.has(item.itemId)) {\n throw new ValidationError(`seedPlants: duplicate dataset itemId \"${item.itemId}\"`)\n }\n seen.add(item.itemId)\n datasetItems.push({\n itemId: item.itemId,\n humanScore: item.humanScore,\n ...(item.group === undefined ? {} : { group: item.group }),\n })\n }\n\n const sealedPlants: Plant[] = []\n const plantIds = new Set<string>()\n for (const plant of plants) {\n const record = definePlant(plant)\n if (plantIds.has(record.id)) {\n throw new ValidationError(`seedPlants: duplicate plant id \"${record.id}\"`)\n }\n if (seen.has(record.item.itemId)) {\n throw new ValidationError(\n `seedPlants: plant \"${record.id}\" reuses itemId \"${record.item.itemId}\", which is already in the set`,\n )\n }\n const thresholdSays: PlantExpectation =\n record.item.humanScore >= acceptThreshold ? 'accept' : 'reject'\n if (thresholdSays !== record.expectedVerdict) {\n throw new ValidationError(\n `seedPlants: plant \"${record.id}\" expects the grader to ${record.expectedVerdict} it, but humanScore ${record.item.humanScore} is on the ${thresholdSays} side of acceptThreshold ${acceptThreshold}`,\n )\n }\n plantIds.add(record.id)\n seen.add(record.item.itemId)\n sealedPlants.push(record)\n }\n\n const items = shuffled(\n [...datasetItems, ...sealedPlants.map((plant) => plant.item)],\n mulberry32(seed),\n )\n const itemIds = items.map((item) => item.itemId)\n const manifest: PlantManifest = {\n seal: sealManifest({ seed, acceptThreshold, itemIds, plants: sealedPlants }),\n seed,\n acceptThreshold,\n itemIds,\n plants: sealedPlants,\n }\n return { items, manifest }\n}\n\n/**\n * Order by one independent uniform key per item, which is a uniform\n * permutation and a pure function of the seed. A comparator that returns a\n * fresh random sign instead is neither: it is not a consistent ordering, so\n * the permutation it produces is biased and depends on the sort algorithm.\n */\nfunction shuffled<T>(items: readonly T[], random: () => number): T[] {\n return items\n .map((item) => ({ item, key: random() }))\n .sort((left, right) => left.key - right.key)\n .map((entry) => entry.item)\n}\n\nfunction sealManifest(contents: Omit<PlantManifest, 'seal'>): LedgerHash {\n return hashCanonical({\n scheme: 'agent-eval.plant-manifest.v1',\n seed: contents.seed,\n acceptThreshold: contents.acceptThreshold,\n itemIds: contents.itemIds,\n plants: contents.plants,\n })\n}\n\n/**\n * One grader outcome for one item. `score: null` says the grader ran and\n * declined to decide — the check never tested the item's defect. Any\n * `CandidateScore` from a judge run is already a valid outcome.\n */\nexport interface PlantOutcome {\n itemId: string\n score: number | null\n}\n\n/**\n * `evaluated` — a rate stands. `incomplete` — a seeded id had no result.\n * `not_evaluated` — nothing was seeded, or nothing seeded was decided.\n */\nexport type CatchRateStatus = 'evaluated' | 'incomplete' | 'not_evaluated'\n\nexport interface PlantKindCounts {\n seeded: number\n caught: number\n missed: number\n indecisive: number\n /** caught / (caught + missed), or null when the report is not `evaluated`. */\n rate: number | null\n}\n\n/**\n * How the same grader treated the unseeded items of the same set. No labels\n * are needed to read it: a grader that refuses everything scores `rate` 1.0\n * on reject-plants, and a `rejectionRate` of 1.0 here is what separates that\n * reflex from discrimination.\n */\nexport interface UnseededRejection {\n n: number\n decided: number\n rejected: number\n /** rejected / decided, or null when nothing unseeded was decided. */\n rejectionRate: number | null\n}\n\nexport interface CatchRateReport {\n status: CatchRateStatus\n /** Why the status is not `evaluated`. Absent when it is. */\n reason?: string\n seeded: number\n caught: number\n missed: number\n /**\n * The grader returned a result and declined to decide. Counted apart: it\n * enters neither side of `rate`.\n */\n indecisive: number\n /** caught / (caught + missed), or null unless the status is `evaluated`. */\n rate: number | null\n /** One entry per kind actually seeded. A kind nobody seeded is absent, never zero. */\n byKind: Partial<Record<PlantKind, PlantKindCounts>>\n /** Plant ids the grader graded as the seed says it must not. */\n missedIds: string[]\n /** Plant ids with no result at all — the reason a status is `incomplete`. */\n missingIds: string[]\n unseeded: UnseededRejection\n}\n\n/**\n * Score a grading run against its sealed manifest.\n *\n * Refusals: a manifest whose contents no longer match its seal, a result for\n * an id the manifest never handed out, and a repeated result id. Each says\n * the results and the manifest describe different runs, and a rate computed\n * across two runs is a fabrication.\n */\nexport function catchRate(\n results: readonly PlantOutcome[],\n manifest: PlantManifest,\n): CatchRateReport {\n const expectedSeal = sealManifest({\n seed: manifest.seed,\n acceptThreshold: manifest.acceptThreshold,\n itemIds: manifest.itemIds,\n plants: manifest.plants,\n })\n if (expectedSeal !== manifest.seal) {\n throw new CaptureIntegrityError(\n `catchRate: manifest contents hash to ${expectedSeal} but the manifest carries seal ${manifest.seal} — the plant set changed after it was sealed`,\n )\n }\n\n const handedOut = new Set(manifest.itemIds)\n const scoreByItemId = new Map<string, number | null>()\n for (const result of results) {\n if (!handedOut.has(result.itemId)) {\n throw new ValidationError(\n `catchRate: result for \"${result.itemId}\", which this manifest never handed out`,\n )\n }\n if (scoreByItemId.has(result.itemId)) {\n throw new ValidationError(`catchRate: duplicate result for \"${result.itemId}\"`)\n }\n if (result.score !== null && !Number.isFinite(result.score)) {\n throw new ValidationError(\n `catchRate: result for \"${result.itemId}\" has score ${result.score}; expected a finite number or null`,\n )\n }\n scoreByItemId.set(result.itemId, result.score)\n }\n\n const byKind: Partial<Record<PlantKind, PlantKindCounts>> = {}\n const missedIds: string[] = []\n const missingIds: string[] = []\n let caught = 0\n let missed = 0\n let indecisive = 0\n\n for (const plant of manifest.plants) {\n let counts = byKind[plant.kind]\n if (counts === undefined) {\n counts = { seeded: 0, caught: 0, missed: 0, indecisive: 0, rate: null }\n byKind[plant.kind] = counts\n }\n counts.seeded += 1\n if (!scoreByItemId.has(plant.item.itemId)) {\n missingIds.push(plant.id)\n continue\n }\n const score = scoreByItemId.get(plant.item.itemId) ?? null\n if (score === null) {\n indecisive += 1\n counts.indecisive += 1\n continue\n }\n const graderSaid: PlantExpectation = score >= manifest.acceptThreshold ? 'accept' : 'reject'\n if (graderSaid === plant.expectedVerdict) {\n caught += 1\n counts.caught += 1\n } else {\n missed += 1\n counts.missed += 1\n missedIds.push(plant.id)\n }\n }\n\n const seeded = manifest.plants.length\n const decided = caught + missed\n const report: CatchRateReport = {\n status: 'evaluated',\n seeded,\n caught,\n missed,\n indecisive,\n rate: null,\n byKind,\n missedIds,\n missingIds,\n unseeded: unseededRejection(manifest, scoreByItemId),\n }\n\n if (seeded === 0) {\n return {\n ...report,\n status: 'not_evaluated',\n reason: 'no plant was seeded, so the grader was never asked a question it could fail',\n }\n }\n if (missingIds.length > 0) {\n return {\n ...report,\n status: 'incomplete',\n reason: `${missingIds.length} of ${seeded} seeded plants have no result: ${missingIds.join(', ')}`,\n }\n }\n if (decided === 0) {\n return {\n ...report,\n status: 'not_evaluated',\n reason: `all ${seeded} seeded plants are indecisive: no check tested the seeded defect`,\n }\n }\n\n for (const counts of Object.values(byKind)) {\n const kindDecided = counts.caught + counts.missed\n counts.rate = kindDecided === 0 ? null : counts.caught / kindDecided\n }\n report.rate = caught / decided\n return report\n}\n\nfunction unseededRejection(\n manifest: PlantManifest,\n scoreByItemId: ReadonlyMap<string, number | null>,\n): UnseededRejection {\n const plantItemIds = new Set(manifest.plants.map((plant) => plant.item.itemId))\n let n = 0\n let decided = 0\n let rejected = 0\n for (const itemId of manifest.itemIds) {\n if (plantItemIds.has(itemId)) continue\n n += 1\n const score = scoreByItemId.get(itemId)\n if (score === undefined || score === null) continue\n decided += 1\n if (score < manifest.acceptThreshold) rejected += 1\n }\n return { n, decided, rejected, rejectionRate: decided === 0 ? null : rejected / decided }\n}\n","/**\n * Judge sentinel — eval trustworthiness as a continuously measured,\n * alarmed trend.\n *\n * Judges are models; models change underneath us; calibration decays.\n * This module composes three existing instruments into one loop:\n *\n * snapshot — adapters turn real instrument outputs (`calibrateJudge` /\n * `calibrateJudgeContinuous`, `corpusInterRaterAgreement` /\n * `continuousAgreement`, a rerun sentinel golden set) into\n * `SentinelSnapshot`s\n * series — `analyzeSeries` (src/series-convergence) classifies each\n * judge×metric history as stabilized / drifting / noisy\n * alarm — `judgeSentinelReport` turns trends + thresholds into named\n * alarms; `evalHealthStamp` is the tiny object a campaign\n * attaches to its verdicts\n *\n * Alarm conditions:\n * - convergence state `drifting-down` on any tracked metric\n * - irr below `minIrr`\n * - calibrationKappa / sentinelPassRate dropped vs the series baseline\n * beyond `maxKappaDrop` / `maxSentinelDrop`\n * - judge model changed with no post-change golden-grounded snapshot\n * (the silent-upgrade trap: judges agreeing with each other after a\n * model swap proves nothing — only re-measuring against gold does)\n * - newest snapshot older than `staleAfterDays` relative to `asOf`\n *\n * Wire-in: run the snapshot adapters from a post-campaign hook or a\n * nightly job, append to a `SentinelStore`, then compute\n * `judgeSentinelReport(await store.history(), { asOf })` with `asOf` =\n * the campaign timestamp. Attach `evalHealthStamp(report)` to campaign\n * verdicts: an alarmed sentinel means every downstream verdict carries\n * an integrity warning until the judge is recalibrated.\n *\n * The module is clock-free — every timestamp (`at`, `asOf`) is\n * caller-supplied ISO-8601, so reports are reproducible.\n */\n\nimport { ValidationError } from '../errors'\nimport type {\n CalibrationResult,\n CandidateScore,\n ContinuousAgreement,\n ContinuousCalibrationResult,\n GoldenItem,\n} from '../judge-calibration'\nimport {\n analyzeSeries,\n type SeriesConvergenceOptions,\n type SeriesConvergenceResult,\n} from '../series-convergence'\nimport type { CorpusAgreementReport } from '../statistics'\n\nconst SENTINEL_METRIC_NAMES = ['irr', 'calibrationKappa', 'sentinelPassRate'] as const\n\nexport type SentinelMetricName = (typeof SENTINEL_METRIC_NAMES)[number]\n\nexport interface SentinelMetrics {\n /** Inter-rater reliability — ICC(2,1) from agreement instruments. */\n irr?: number\n /** Weighted κ vs the human golden set (`calibrateJudge*`). */\n calibrationKappa?: number\n /** Fraction of a fixed sentinel golden set the judge still scores within tolerance. */\n sentinelPassRate?: number\n}\n\nexport interface SentinelSnapshot {\n /** Caller-supplied ISO-8601 timestamp — the module never reads a clock. */\n at: string\n judgeId: string\n /** Exact model identity behind the judge (pin the full version string). */\n judgeModel: string\n /**\n * Metrics measured at `at`. May be empty: an empty-metrics snapshot is a\n * model-change marker — it records the new `judgeModel` immediately and\n * the silent-upgrade alarm stays raised until a golden-grounded snapshot\n * (calibrationKappa or sentinelPassRate) follows.\n */\n metrics: SentinelMetrics\n}\n\n/** Identity + timestamp the caller supplies alongside an instrument output. */\nexport interface SnapshotMeta {\n at: string\n judgeId: string\n judgeModel: string\n}\n\nfunction parseIso(value: string, label: string): number {\n const ms = typeof value === 'string' ? Date.parse(value) : NaN\n if (!Number.isFinite(ms)) {\n throw new ValidationError(\n `judge-sentinel: ${label} is not a parseable ISO timestamp: ${JSON.stringify(value)}`,\n )\n }\n return ms\n}\n\n/** Shape guard for snapshots crossing a boundary (store append, JSONL load, report input). */\nexport function validateSentinelSnapshot(snapshot: SentinelSnapshot, source = 'snapshot'): void {\n parseIso(snapshot.at, `${source}.at`)\n if (typeof snapshot.judgeId !== 'string' || snapshot.judgeId.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeId must be a non-empty string`)\n }\n if (typeof snapshot.judgeModel !== 'string' || snapshot.judgeModel.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeModel must be a non-empty string`)\n }\n if (snapshot.metrics === null || typeof snapshot.metrics !== 'object') {\n throw new ValidationError(`judge-sentinel: ${source}.metrics must be an object (may be empty)`)\n }\n for (const [key, value] of Object.entries(snapshot.metrics)) {\n if (!(SENTINEL_METRIC_NAMES as readonly string[]).includes(key)) {\n // A typo'd key would silently never trend — reject instead.\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics carries unknown key \"${key}\" — known: ${SENTINEL_METRIC_NAMES.join(', ')}`,\n )\n }\n if (value !== undefined && !Number.isFinite(value)) {\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics.${key} is not a finite number: ${String(value)}`,\n )\n }\n }\n}\n\n// ── Snapshot adapters — real instrument output → SentinelSnapshot ───\n\n/**\n * From `calibrateJudge` / `calibrateJudgeContinuous` output. Prefers the\n * un-rounded continuous κ_w when the report carries one (integer κ\n * discards information for [0,1] judges).\n */\nexport function snapshotFromCalibration(\n report: CalibrationResult | ContinuousCalibrationResult,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const kappa = 'weightedKappaContinuous' in report ? report.weightedKappaContinuous : report.kappa\n if (!Number.isFinite(kappa)) {\n throw new ValidationError(\n `snapshotFromCalibration: κ is not finite (n=${report.n}) — calibrate against ≥2 joined golden items before snapshotting`,\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { calibrationKappa: kappa } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n/**\n * From `corpusInterRaterAgreement` (statistics.ts) or `continuousAgreement`\n * (judge-calibration.ts) output. IRR = ICC(2,1) — the reliability\n * coefficient both instruments compute. For ensemble-level agreement,\n * `meta.judgeId` names the ensemble and `meta.judgeModel` pins its\n * composition (e.g. a joined list of member model strings).\n */\nexport function snapshotFromAgreement(\n report: CorpusAgreementReport | ContinuousAgreement,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const irr = 'overallIcc' in report ? report.overallIcc : report.icc\n if (!Number.isFinite(irr)) {\n throw new ValidationError(\n 'snapshotFromAgreement: ICC is not finite — agreement needs ≥2 raters and ≥2 complete items',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { irr } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\nexport interface SentinelSetOptions {\n /** Max |judge − human| for an item to count as passed. Default 0.1 (tuned for [0,1] scores). */\n tolerance?: number\n}\n\n/**\n * From a rerun of a fixed sentinel golden set: the same `GoldenItem`s the\n * judge was originally calibrated on, re-scored periodically. Pass rate =\n * fraction of joined items where the judge stays within `tolerance` of the\n * human score. Items the judge didn't score are excluded from the\n * denominator; zero joined items is an error, not a 0% pass rate.\n */\nexport function snapshotFromSentinelSet(\n scores: CandidateScore[],\n golden: GoldenItem[],\n meta: SnapshotMeta,\n options: SentinelSetOptions = {},\n): SentinelSnapshot {\n const tolerance = options.tolerance ?? 0.1\n if (!Number.isFinite(tolerance) || tolerance < 0) {\n throw new ValidationError(\n `snapshotFromSentinelSet: tolerance must be a finite number ≥ 0, got ${tolerance}`,\n )\n }\n const goldenById = new Map<string, number>()\n for (const item of golden) {\n if (!Number.isFinite(item.humanScore)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: golden item \"${item.itemId}\" has a non-finite humanScore`,\n )\n }\n if (goldenById.has(item.itemId)) {\n throw new ValidationError(`snapshotFromSentinelSet: duplicate golden itemId \"${item.itemId}\"`)\n }\n goldenById.set(item.itemId, item.humanScore)\n }\n let joined = 0\n let passed = 0\n for (const s of scores) {\n const human = goldenById.get(s.itemId)\n if (human === undefined) continue\n if (!Number.isFinite(s.score)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: judge score for item \"${s.itemId}\" is not finite`,\n )\n }\n joined++\n if (Math.abs(s.score - human) <= tolerance) passed++\n }\n if (joined === 0) {\n throw new ValidationError(\n 'snapshotFromSentinelSet: no judge scores joined the golden set — itemIds mismatch or empty sentinel run',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { sentinelPassRate: passed / joined } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n// ── Persistence ──────────────────────────────────────────────────────\n\nexport interface SentinelStore {\n append(snapshot: SentinelSnapshot): Promise<void>\n /** All snapshots (optionally for one judge), in append order. */\n history(judgeId?: string): Promise<SentinelSnapshot[]>\n}\n\nexport function inMemorySentinelStore(initial: SentinelSnapshot[] = []): SentinelStore {\n for (const s of initial) validateSentinelSnapshot(s)\n const items = initial.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n items.push({ ...snapshot, metrics: { ...snapshot.metrics } })\n },\n async history(judgeId) {\n const all = items.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return judgeId === undefined ? all : all.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n/**\n * JSONL store — one snapshot per line, appended atomically per call.\n * A corrupt or shape-invalid line is a loud error naming the file and\n * line number: a sentinel history that silently drops records would\n * defeat the drift detection it exists to provide.\n */\nexport function fileSentinelStore(path: string): SentinelStore {\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n const fs = await import('node:fs/promises')\n const pathMod = await import('node:path')\n await fs.mkdir(pathMod.dirname(path), { recursive: true })\n await fs.appendFile(path, `${JSON.stringify(snapshot)}\\n`, 'utf8')\n },\n async history(judgeId) {\n const fs = await import('node:fs/promises')\n let raw: string\n try {\n raw = await fs.readFile(path, 'utf8')\n } catch (err) {\n if ((err as NodeJS.ErrnoException).code === 'ENOENT') return []\n throw err\n }\n const lines = raw.split('\\n')\n const snapshots: SentinelSnapshot[] = []\n for (let i = 0; i < lines.length; i++) {\n const line = lines[i]!\n if (line.trim().length === 0) continue\n let parsed: unknown\n try {\n parsed = JSON.parse(line)\n } catch (err) {\n throw new ValidationError(\n `judge-sentinel: store at ${path} has a corrupt JSONL line ${i + 1}: ${(err as Error).message}`,\n )\n }\n const snapshot = parsed as SentinelSnapshot\n validateSentinelSnapshot(snapshot, `${path}:${i + 1}`)\n snapshots.push(snapshot)\n }\n return judgeId === undefined ? snapshots : snapshots.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n// ── Trend report ─────────────────────────────────────────────────────\n\nexport interface SentinelThresholds {\n /** Floor for irr — below it the judge ensemble is unreliable regardless of trend. Default 0.6. */\n minIrr?: number\n /** Max tolerated calibrationKappa drop vs the series baseline. Default 0.15. */\n maxKappaDrop?: number\n /** Max tolerated sentinelPassRate drop vs the series baseline. Default 0.1. */\n maxSentinelDrop?: number\n /** Newest snapshot older than this (days, vs asOf) = the sentinel itself went blind. Default 30. */\n staleAfterDays?: number\n}\n\nexport interface JudgeSentinelOptions {\n /** Caller-supplied ISO timestamp staleness is measured against. */\n asOf: string\n thresholds?: SentinelThresholds\n /** Forwarded to `analyzeSeries` (window, stableCv, driftRun). */\n convergence?: SeriesConvergenceOptions\n}\n\nexport interface SentinelTrend {\n judgeId: string\n metric: SentinelMetricName\n /** Verbatim `analyzeSeries` state for this judge×metric history. */\n state: SeriesConvergenceResult['state']\n /** Newest value in the series. */\n current: number\n /** Oldest value in the series — the reference the drop thresholds compare against. */\n baseline: number\n /** current − baseline (signed; negative = decayed). */\n drift: number\n alarmed: boolean\n reason?: string\n}\n\nexport interface SentinelReport {\n perJudge: SentinelTrend[]\n alarms: string[]\n healthy: boolean\n /**\n * `judgeId:metric` pairs with too few snapshots for the convergence\n * machine. Named, not counted — a blind spot you can't see is a blind\n * spot you won't fix. Floor/drop checks still apply to thin series, so\n * insufficient history alone never masks an alarm.\n */\n insufficientHistory: string[]\n}\n\nconst MS_PER_DAY = 86_400_000\n\nexport function judgeSentinelReport(\n history: SentinelSnapshot[],\n opts: JudgeSentinelOptions,\n): SentinelReport {\n const asOfMs = parseIso(opts.asOf, 'asOf')\n const minIrr = opts.thresholds?.minIrr ?? 0.6\n const maxKappaDrop = opts.thresholds?.maxKappaDrop ?? 0.15\n const maxSentinelDrop = opts.thresholds?.maxSentinelDrop ?? 0.1\n const staleAfterDays = opts.thresholds?.staleAfterDays ?? 30\n\n const byJudge = new Map<string, Array<SentinelSnapshot & { atMs: number }>>()\n for (const snapshot of history) {\n validateSentinelSnapshot(snapshot)\n const atMs = parseIso(snapshot.at, `snapshot.at (judge \"${snapshot.judgeId}\")`)\n const arr = byJudge.get(snapshot.judgeId) ?? []\n arr.push({ ...snapshot, atMs })\n byJudge.set(snapshot.judgeId, arr)\n }\n\n const perJudge: SentinelTrend[] = []\n const alarms: string[] = []\n const insufficientHistory: string[] = []\n\n for (const judgeId of [...byJudge.keys()].sort()) {\n const snaps = byJudge.get(judgeId)!.sort((a, b) => a.atMs - b.atMs)\n const newest = snaps[snaps.length - 1]!\n\n const ageDays = (asOfMs - newest.atMs) / MS_PER_DAY\n if (ageDays > staleAfterDays) {\n alarms.push(\n `judge \"${judgeId}\": stale — newest snapshot ${newest.at} is ${ageDays.toFixed(1)}d old vs asOf (limit ${staleAfterDays}d)`,\n )\n }\n\n // Silent-upgrade trap: only the latest model change matters for current\n // trust, and only golden-grounded metrics clear it — irr alone proves\n // judges still agree with each other, not that they're still right.\n let changeIdx = -1\n for (let i = 1; i < snaps.length; i++) {\n if (snaps[i]!.judgeModel !== snaps[i - 1]!.judgeModel) changeIdx = i\n }\n if (changeIdx >= 0) {\n const recalibrated = snaps\n .slice(changeIdx)\n .some(\n (s) =>\n Number.isFinite(s.metrics.calibrationKappa) ||\n Number.isFinite(s.metrics.sentinelPassRate),\n )\n if (!recalibrated) {\n alarms.push(\n `judge \"${judgeId}\": model changed \"${snaps[changeIdx - 1]!.judgeModel}\" → \"${snaps[changeIdx]!.judgeModel}\" at ${snaps[changeIdx]!.at} with no post-change calibration snapshot — silent-upgrade trap; rerun calibrateJudge or the sentinel set against the new model`,\n )\n }\n }\n\n for (const metric of SENTINEL_METRIC_NAMES) {\n const values = snaps\n .filter((s) => typeof s.metrics[metric] === 'number')\n .map((s) => s.metrics[metric]!)\n if (values.length === 0) continue\n\n const convergence = analyzeSeries(values, opts.convergence)\n if (convergence.state === 'insufficient-data')\n insufficientHistory.push(`${judgeId}:${metric}`)\n\n const current = values[values.length - 1]!\n const baseline = values[0]!\n const drift = current - baseline\n const reasons: string[] = []\n if (convergence.state === 'drifting-down') {\n reasons.push(\n `convergence state drifting-down (tail run ${convergence.tailRun}, window mean ${convergence.windowMean.toFixed(3)})`,\n )\n }\n if (metric === 'irr' && current < minIrr) {\n reasons.push(`irr ${current.toFixed(3)} below floor ${minIrr}`)\n }\n if (metric === 'calibrationKappa' && baseline - current > maxKappaDrop) {\n reasons.push(\n `calibrationKappa dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxKappaDrop})`,\n )\n }\n if (metric === 'sentinelPassRate' && baseline - current > maxSentinelDrop) {\n reasons.push(\n `sentinelPassRate dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxSentinelDrop})`,\n )\n }\n\n const alarmed = reasons.length > 0\n const trend: SentinelTrend = {\n judgeId,\n metric,\n state: convergence.state,\n current,\n baseline,\n drift,\n alarmed,\n }\n if (alarmed) {\n trend.reason = reasons.join('; ')\n alarms.push(`judge \"${judgeId}\" ${metric}: ${trend.reason}`)\n }\n perJudge.push(trend)\n }\n }\n\n return { perJudge, alarms, healthy: alarms.length === 0, insufficientHistory }\n}\n\n// ── Verdict stamp ────────────────────────────────────────────────────\n\nexport interface EvalHealthStamp {\n healthy: boolean\n alarms: string[]\n}\n\n/**\n * The object a campaign attaches to its verdicts. Compute it from the\n * sentinel report in a post-campaign hook (or a nightly job feeding the\n * next day's campaigns) and store it alongside the verdict payload.\n * `healthy: false` means the judges that produced those verdicts have an\n * unresolved drift / decay / silent-upgrade / staleness alarm — treat the\n * verdicts as carrying an integrity warning until recalibration clears it.\n */\nexport function evalHealthStamp(report: SentinelReport): EvalHealthStamp {\n return { healthy: report.healthy, alarms: [...report.alarms] }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;AAgDA,eAAsB,iBACpB,YACA,cACA,YACA,eACA,UAA8B,CAAC,GACI;CACnC,MAAM,OAAO,MAAM,WAAW,SAAS;CACvC,MAAM,WAAW,MAAM,aAAa,KAAK;CACzC,MAAM,wBAAQ,IAAI,IAAiC;CACnD,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,MAAM,MAAM,IAAI,EAAE,KAAK,KAAK,CAAC;EACnC,IAAI,KAAK,CAAC;EACV,MAAM,IAAI,EAAE,OAAO,GAAG;CACxB;CAEA,MAAM,UAAU,WAAW,WAAW,mBAAmB,WAAW,EAAE;CACtE,MAAM,QAAyC,CAAC;CAChD,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,KAAK,MAAM,IAAI,IAAI,KAAK;EAC9B,IAAI,CAAC,IAAI,QAAQ;EACjB,MAAM,IAAI,MAAM,QAAQ,KAAK,UAAU;EACvC,IAAI,MAAM,QAAQ,CAAC,OAAO,SAAS,CAAC,GAAG;EAEvC,MAAM,IADS,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,aAAa,EAAE,UAAU,CAAC,CAAC,EACnD,CAAC,QAAQ;EACzB,IAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAAG;EAClD,MAAM,KAAK;GAAE;GAAG;EAAE,CAAC;CACrB;CACA,IAAI,MAAM,SAAS,GAAG,OAAO;CAE7B,OAAO,qBACL,MAAM,KAAK,OAAO;EAAE,WAAW,EAAE;EAAG,SAAS,EAAE;CAAE,EAAE,GACnD,WAAW,IACX,eACA,OACF;AACF;AAEA,SAAS,qBACP,YACA,YACA,eACA,UAA8B,CAAC,GACL;CAC1B,MAAM,QAAQ,WAAW,QACtB,SAAS,OAAO,SAAS,KAAK,SAAS,KAAK,OAAO,SAAS,KAAK,OAAO,CAC3E;CACA,IAAI,MAAM,SAAS,GAAG,OAAO;CAE7B,MAAM,UAAU,QAAQ,QAAQ;CAChC,MAAM,UAAU,QAAQ,WAAW;CACnC,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,SAAS;CACvC,MAAM,KAAK,QAAQ,OAAO,MAAM,KAAK,IAAI,GAAG,EAAE;CAC9C,MAAM,KAAK,QAAQ,OAAO,MAAM,KAAK,IAAI,GAAG,EAAE;CAE9C,MAAM,OAAyB,CAAC;CAChC,IAAI,YAAY,mBAAmB;EACjC,MAAM,SAAS,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;EAClE,MAAM,SAAS,KAAK,IAAI,GAAG,KAAK,MAAM,OAAO,SAAS,OAAO,CAAC;EAC9D,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK,QAAQ;GAC9C,MAAM,QAAQ,OAAO,MAAM,GAAG,IAAI,MAAM;GACxC,IAAI,MAAM,WAAW,GAAG;GACxB,KAAK,KAAK,MAAM,KAAK,CAAC;EACxB;CACF,OAAO;EACL,MAAM,SAAS,KAAK,MAAM;EAC1B,IAAI,UAAU,GAAG,OAAO;EACxB,KAAK,IAAI,IAAI,GAAG,IAAI,SAAS,KAAK;GAChC,MAAM,QAAQ,KAAK,IAAI;GACvB,MAAM,QAAQ,MAAM,UAAU,IAAI,KAAK,OAAO,MAAM,IAAI,KAAK;GAC7D,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,aAAa,SAAS,EAAE,YAAY,KAAK;GAC7E,IAAI,MAAM,WAAW,GAAG;GACxB,KAAK,KAAK,MAAM,OAAO,OAAO,KAAK,CAAC;EACtC;CACF;CAEA,MAAM,QAAQ,KAAK,QAAQ,GAAG,MAAM,IAAI,EAAE,GAAG,CAAC;CAC9C,MAAM,MAAM,KAAK,QAAQ,GAAG,MAAM,IAAK,EAAE,IAAI,QAAS,EAAE,KAAK,CAAC;CAC9D,MAAM,SAAS,KAAK,QAAQ,GAAG,MAAM,KAAK,IAAI,GAAG,EAAE,GAAG,GAAG,CAAC;CAE1D,OAAO;EAAE;EAAY;EAAe,GAAG,MAAM;EAAQ;EAAM;EAAK;CAAO;AACzE;AAEA,SAAS,MAAM,OAA0B,OAAgB,OAAgC;CACvF,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,SAAS;CACvC,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,OAAO;CACrC,MAAM,WAAW,KAAK,EAAE;CACxB,MAAM,cAAc,KAAK,EAAE;CAC3B,OAAO;EACL,OAAO,SAAS,KAAK,IAAI,GAAG,EAAE;EAC9B,OAAO,SAAS,KAAK,IAAI,GAAG,EAAE;EAC9B,GAAG,MAAM;EACT;EACA;EACA,KAAK,KAAK,IAAI,cAAc,QAAQ;CACtC;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,OAAO,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,GAAG;AAC5C;;;;;;;;;;;;;ACtFA,eAAsB,iBACpB,YACA,cACA,aACA,oBACA,UAAmC,CAAC,GACH;CACjC,MAAM,OAAO,MAAM,WAAW,SAAS;CACvC,MAAM,WAAW,MAAM,aAAa,KAAK,QAAQ,aAAa;CAC9D,MAAM,gCAAgB,IAAI,IAAiC;CAC3D,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,MAAM,cAAc,IAAI,EAAE,KAAK,KAAK,CAAC;EAC3C,IAAI,KAAK,CAAC;EACV,cAAc,IAAI,EAAE,OAAO,GAAG;CAChC;CAEA,MAAM,YAAY,QAAQ,aAAa;CACvC,MAAM,SAAS,QAAQ,mBAAmB;CAE1C,MAAM,QAA0F,CAAC;CACjG,KAAK,MAAM,MAAM,aACf,KAAK,MAAM,MAAM,oBACf,MAAM,KAAK;EAAE,YAAY,GAAG;EAAI,eAAe;EAAI,IAAI,CAAC;EAAG,IAAI,CAAC;CAAE,CAAC;CAIvE,IAAI,SAAS;CACb,IAAI,UAAU;CACd,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,KAAK,cAAc,IAAI,IAAI,KAAK;EACtC,IAAI,CAAC,MAAM,GAAG,WAAW,GAAG;GAC1B;GACA;EACF;EACA,MAAM,WAAW,GAAG,QAAQ,MAAM,EAAE,aAAa,IAAI,aAAa,MAAM;EACxE,IAAI,SAAS,WAAW,GAAG;GACzB;GACA;EACF;EAEA,KAAK,MAAM,MAAM,aAAa;GAE5B,MAAM,IAAI,OADM,GAAG,WAAW,mBAAmB,GAAG,EAAE,EAAA,CAC9B,KAAK,UAAU;GACvC,IAAI,MAAM,QAAQ,CAAC,OAAO,SAAS,CAAC,GAAG;GAEvC,KAAK,MAAM,MAAM,oBAAoB;IACnC,MAAM,SAAS,SACZ,KAAK,MAAM,EAAE,QAAQ,GAAG,CAAC,CACzB,QAAQ,MAAmB,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,CAAC;IACzE,IAAI,OAAO,WAAW,GAAG;IACzB,MAAM,IAAI,OAAO,QAAQ,WAAW,QAAQ;IAC5C,IAAI,MAAM,MAAM;IAChB,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,eAAe,GAAG,MAAM,EAAE,kBAAkB,EAAE;IAC/E,KAAK,GAAG,KAAK,CAAC;IACd,KAAK,GAAG,KAAK,CAAC;GAChB;EACF;EACA;CACF;CA0BA,OAAO;EAAE,OAxB4B,MAClC,QAAQ,MAAM,EAAE,GAAG,UAAU,CAAC,CAAC,CAC/B,KAAK,MAAM;GACV,MAAM,UAAU,SAAS,EAAE,IAAI,EAAE,EAAE;GACnC,MAAM,WAAW,UAAU,EAAE,IAAI,EAAE,EAAE;GACrC,MAAM,cAAc,mBAClB,EAAE,IACF,EAAE,IACF,QAAQ,uBAAuB,KAC/B,QAAQ,IACV;GACA,MAAM,UACJ,KAAK,IAAI,OAAO,KAAK,KAAM,WAAW,KAAK,IAAI,OAAO,KAAK,KAAM,aAAa;GAChF,OAAO;IACL,YAAY,EAAE;IACd,eAAe,EAAE;IACjB,GAAG,EAAE,GAAG;IACR;IACA;IACA;IACA;GACF;EACF,CAEoB;EAAG,eAAe;EAAQ,aAAa;CAAQ;AACvE;AAIA,SAAS,OACP,QACA,MACA,UACe;CACf,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,IAAI,SAAS,QAAQ,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;CACvE,IAAI,SAAS,OAAO,OAAO,KAAK,IAAI,GAAG,MAAM;CAE7C,MAAM,SAAS,CAAC,GAAG,QAAQ,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,aAAa,EAAE,UAAU,CAAC,CAAC;CACzE,IAAI,CAAC,QAAQ,OAAO;CACpB,MAAM,YAAY,OAAO,KAAK,OAAO,OAAO,CAAC,CAAC;CAC9C,MAAM,IAAI,cAAc,KAAA,IAAY,OAAO,QAAQ,aAAa,KAAA;CAEhE,MAAM,SAAS,SACZ,KAAK,MAAM;EACV,MAAM,IAAI,OAAO,KAAK,EAAE,OAAO,CAAC,CAAC;EACjC,OAAO;GACL,IAAI,EAAE;GACN,GAAG,MAAM,KAAA,IAAY,OAAO,MAAM,MAAM,EAAE,QAAQ,OAAO,CAAC,IAAI,KAAA;EAChE;CACF,CAAC,CAAC,CACD,QAAQ,MAAM,EAAE,MAAM,KAAA,CAAS;CAClC,IAAI,OAAO,WAAW,GAAG,OAAO,KAAK;CACrC,OAAO,OAAO,MAAM,GAAG,MAAM,EAAE,KAAK,EAAE,EAAE,CAAC,CAAC,EAAE,EAAE,KAAK;AACrD;AAEA,SAAS,mBACP,IACA,IACA,YACA,MACkC;CAClC,MAAM,IAAI,GAAG;CACb,IAAI,IAAI,GAAG,OAAO;EAAE,OAAO;EAAK,OAAO;CAAI;CAC3C,MAAM,MAAM,QAAQ,MAAM,IAAI,EAAE;CAChC,MAAM,KAAe,CAAC;CACtB,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,KAAe,IAAI,MAAM,CAAC;EAChC,MAAM,KAAe,IAAI,MAAM,CAAC;EAChC,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;GAC1B,MAAM,MAAM,KAAK,MAAM,IAAI,IAAI,CAAC;GAChC,GAAG,KAAK,GAAG;GACX,GAAG,KAAK,GAAG;EACb;EACA,MAAM,IAAI,SAAS,IAAI,EAAE;EACzB,IAAI,OAAO,SAAS,CAAC,GAAG,GAAG,KAAK,CAAC;CACnC;CACA,GAAG,MAAM,GAAG,MAAM,IAAI,CAAC;CACvB,IAAI,GAAG,WAAW,GAAG,OAAO;EAAE,OAAO;EAAK,OAAO;CAAI;CACrD,OAAO;EACL,OAAO,GAAG,KAAK,MAAM,OAAQ,GAAG,MAAM;EACtC,OAAO,GAAG,KAAK,IAAI,GAAG,SAAS,GAAG,KAAK,MAAM,OAAQ,GAAG,MAAM,CAAC;CACjE;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACrIA,MAAM,cAAoC;CACxC;CACA;CACA;CACA;AACF;AAKA,MAAM,qBAAkD,CAAC,UAAU,QAAQ;;AAG3E,MAAM,wBAAwB;;;;;;;;;;;;;AAwB9B,SAAgB,YAAY,OAKlB;CACR,MAAM,EAAE,IAAI,MAAM,MAAM,oBAAoB;CAC5C,IAAI,OAAO,OAAO,YAAY,GAAG,KAAK,MAAM,IAC1C,MAAM,IAAI,gBAAgB,4CAA4C;CAExE,IAAI,CAAC,YAAY,SAAS,IAAI,GAC5B,MAAM,IAAI,gBACR,uBAAuB,GAAG,aAAa,KAAK,UAAU,IAAI,EAAE,oBAAoB,YAAY,KAAK,IAAI,GACvG;CAEF,IAAI,CAAC,mBAAmB,SAAS,eAAe,GAC9C,MAAM,IAAI,gBACR,uBAAuB,GAAG,wBAAwB,KAAK,UAAU,eAAe,EAAE,oBAAoB,mBAAmB,KAAK,IAAI,GACpI;CAEF,IAAI,OAAO,KAAK,WAAW,YAAY,KAAK,OAAO,KAAK,MAAM,IAC5D,MAAM,IAAI,gBAAgB,uBAAuB,GAAG,2BAA2B;CAEjF,IAAI,CAAC,OAAO,SAAS,KAAK,UAAU,KAAK,KAAK,aAAa,KAAK,KAAK,aAAa,GAChF,MAAM,IAAI,gBACR,uBAAuB,GAAG,mBAAmB,KAAK,WAAW,qCAC/D;CAEF,IAAI,KAAK,UAAU,KAAA,KAAa,OAAO,KAAK,UAAU,UACpD,MAAM,IAAI,gBAAgB,uBAAuB,GAAG,8BAA8B;CAEpF,IAAI,KAAK,eAAe,uBACtB,MAAM,IAAI,gBACR,uBAAuB,GAAG,mBAAmB,sBAAsB,qDACrE;CAEF,MAAM,YAA8B,KAAK,aAAa,wBAAwB,WAAW;CACzF,IAAI,cAAc,iBAChB,MAAM,IAAI,gBACR,uBAAuB,GAAG,0BAA0B,gBAAgB,sBAAsB,KAAK,WAAW,QAAQ,WACpH;CAEF,OAAO;EACL;EACA;EACA,MAAM;GACJ,QAAQ,KAAK;GACb,YAAY,KAAK;GACjB,GAAI,KAAK,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,KAAK,MAAM;EAC1D;EACA;CACF;AACF;;;;;;;;;;;AA8CA,MAAM,cAAc;;;;;;;;AASpB,MAAM,mBAA2D;CAC/D,CAAC,SAAS,OAAO;CACjB,CAAC,SAAS,OAAO;CACjB,CAAC,SAAS,OAAO;CACjB,CAAC,SAAS,OAAO;CACjB,CAAC,SAAS,OAAO;CACjB,CAAC,SAAS,OAAO;CACjB,CAAC,MAAM,GAAG;CACV,CAAC,MAAM,GAAG;CACV,CAAC,MAAM,IAAI;CACX,CAAC,MAAM,IAAI;AACb;;;;;;;;;AAeA,SAAS,qBAAqB,MAA2B;CACvD,MAAM,SAAS;CACf,IAAI,QAAqB;CACzB,IAAI,QAAQ,OAAO,KAAK,IAAI;CAC5B,OAAO,UAAU,MAAM;EACrB,MAAM,QAAQ,MAAM;EACpB,MAAM,MAAM,QAAQ,MAAM,EAAE,CAAC;EAE7B,IAAI,CAAC,YAAY,KAAK,KAAK,OAAO,QAAQ,CAAC,CAAC,KAAK,CAAC,YAAY,KAAK,KAAK,OAAO,GAAG,CAAC,GACjF,QAAQ;GAAE;GAAO;EAAI;EAEvB,QAAQ,OAAO,KAAK,IAAI;CAC1B;CACA,OAAO;AACT;;;;;;;;;AAUA,SAAS,gBAAgB,MAAsD;CAC7E,KAAK,IAAI,QAAQ,GAAG,QAAQ,KAAK,QAAQ,SAAS,GAChD,KAAK,MAAM,CAAC,OAAO,YAAY,kBAC7B,IAAI,KAAK,WAAW,OAAO,KAAK,GAC9B,OAAO;EAAE,MAAM;GAAE,OAAO;GAAO,KAAK,QAAQ,MAAM;EAAO;EAAG;CAAQ;CAI1E,OAAO;AACT;AAEA,SAAS,YAAY,MAAc,MAAY,OAAuB;CACpE,OAAO,KAAK,MAAM,GAAG,KAAK,KAAK,IAAI,QAAQ,KAAK,MAAM,KAAK,GAAG;AAChE;;;;;AAMA,SAAS,UACP,OACA,QACA,OACA,OACe;CACf,MAAM,YAAY,UAAU,UAAU,QAAQ;CAC9C,MAAM,aAAa,UAAU,WAAW,QAAQ;CAChD,OAAO;EACL,GAAI,cAAc,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,UAAU;EACtD,GAAI,eAAe,KAAA,IAAY,CAAC,IAAI,EAAE,QAAQ,WAAW;CAC3D;AACF;AAEA,SAAS,SACP,OACA,QACA,OACA,MACA,MACsB;CACtB,MAAM,WAAW,KAAK,MAAM,KAAK,OAAO,KAAK,GAAG;CAChD,MAAM,aAAa,OAAO,QAAQ,IAAI,GAAA,CAAI,SAAS;CACnD,OAAO;EACL,UAAU,UAAU,OAAO,QAAQ,OAAO,YAAY,MAAM,MAAM,SAAS,CAAC;EAC5E;EACA,KAAK;EACL;EACA;CACF;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA6BA,SAAgB,gBAAgB,UAAsD;CACpF,MAAM,QAAQ,OAAO,SAAS,UAAU,WAAW,SAAS,QAAQ,KAAA;CACpE,MAAM,SAAS,OAAO,SAAS,WAAW,WAAW,SAAS,SAAS,KAAA;CAEvE,IAAI,WAAW,KAAA,GAAW;EACxB,MAAM,OAAO,qBAAqB,MAAM;EACxC,IAAI,SAAS,MAAM,OAAO,SAAS,OAAO,QAAQ,UAAU,QAAQ,IAAI;CAC1E;CACA,IAAI,UAAU,KAAA,GAAW;EACvB,MAAM,aAAa,gBAAgB,KAAK;EACxC,IAAI,eAAe,MACjB,OAAO;GACL,UAAU,UACR,OACA,QACA,SACA,YAAY,OAAO,WAAW,MAAM,WAAW,OAAO,CACxD;GACA,OAAO;GACP,KAAK;GACL,UAAU,MAAM,MAAM,WAAW,KAAK,OAAO,WAAW,KAAK,GAAG;GAChE,WAAW,WAAW;EACxB;EAEF,MAAM,OAAO,qBAAqB,KAAK;EACvC,IAAI,SAAS,MAAM,OAAO,SAAS,OAAO,QAAQ,SAAS,OAAO,IAAI;CACxE;CACA,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;AAgCA,SAAgB,oBAAoB,OAOV;CACxB,MAAM,EAAE,IAAI,QAAQ,UAAU,aAAa,GAAG,UAAU;CACxD,IAAI,OAAO,OAAO,YAAY,GAAG,KAAK,MAAM,IAC1C,MAAM,IAAI,gBAAgB,oDAAoD;CAEhF,IAAI,OAAO,WAAW,YAAY,OAAO,KAAK,MAAM,IAClD,MAAM,IAAI,gBAAgB,+BAA+B,GAAG,sBAAsB;CAEpF,MAAM,eAAe,gBAAgB,QAAQ;CAC7C,IAAI,iBAAiB,MAAM,OAAO;CAOlC,OAAO;EACL,OAPY,YAAY;GACxB;GACA,MAAM;GACN,MAAM;IAAE;IAAQ;IAAY,GAAI,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM;GAAG;GACtE,iBAAiB;EACnB,CAEM;EACJ,UAAU,aAAa;EACvB,OAAO,aAAa;EACpB,KAAK,aAAa;EAClB,UAAU,aAAa;EACvB,WAAW,aAAa;CAC1B;AACF;;;;;;;;;;AAoDA,SAAgB,WACd,SACA,QACA,UAA6B,CAAC,GACZ;CAClB,MAAM,OAAO,QAAQ,QAAQ;CAC7B,MAAM,kBAAkB,QAAQ,mBAAmB;CACnD,IAAI,CAAC,OAAO,SAAS,IAAI,GACvB,MAAM,IAAI,gBAAgB,iDAAiD,MAAM;CAEnF,IAAI,CAAC,OAAO,SAAS,eAAe,KAAK,mBAAmB,KAAK,kBAAkB,GACjF,MAAM,IAAI,gBACR,sEAAsE,iBACxE;CAGF,MAAM,eAA6B,CAAC;CACpC,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,QAAQ,SAAS;EAC1B,IAAI,OAAO,KAAK,WAAW,YAAY,KAAK,OAAO,KAAK,MAAM,IAC5D,MAAM,IAAI,gBAAgB,gDAAgD;EAE5E,IAAI,CAAC,OAAO,SAAS,KAAK,UAAU,GAClC,MAAM,IAAI,gBACR,6BAA6B,KAAK,OAAO,8BAC3C;EAEF,IAAI,KAAK,IAAI,KAAK,MAAM,GACtB,MAAM,IAAI,gBAAgB,yCAAyC,KAAK,OAAO,EAAE;EAEnF,KAAK,IAAI,KAAK,MAAM;EACpB,aAAa,KAAK;GAChB,QAAQ,KAAK;GACb,YAAY,KAAK;GACjB,GAAI,KAAK,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,KAAK,MAAM;EAC1D,CAAC;CACH;CAEA,MAAM,eAAwB,CAAC;CAC/B,MAAM,2BAAW,IAAI,IAAY;CACjC,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,SAAS,YAAY,KAAK;EAChC,IAAI,SAAS,IAAI,OAAO,EAAE,GACxB,MAAM,IAAI,gBAAgB,mCAAmC,OAAO,GAAG,EAAE;EAE3E,IAAI,KAAK,IAAI,OAAO,KAAK,MAAM,GAC7B,MAAM,IAAI,gBACR,sBAAsB,OAAO,GAAG,mBAAmB,OAAO,KAAK,OAAO,+BACxE;EAEF,MAAM,gBACJ,OAAO,KAAK,cAAc,kBAAkB,WAAW;EACzD,IAAI,kBAAkB,OAAO,iBAC3B,MAAM,IAAI,gBACR,sBAAsB,OAAO,GAAG,0BAA0B,OAAO,gBAAgB,sBAAsB,OAAO,KAAK,WAAW,aAAa,cAAc,2BAA2B,iBACtL;EAEF,SAAS,IAAI,OAAO,EAAE;EACtB,KAAK,IAAI,OAAO,KAAK,MAAM;EAC3B,aAAa,KAAK,MAAM;CAC1B;CAEA,MAAM,QAAQ,SACZ,CAAC,GAAG,cAAc,GAAG,aAAa,KAAK,UAAU,MAAM,IAAI,CAAC,GAC5D,WAAW,IAAI,CACjB;CACA,MAAM,UAAU,MAAM,KAAK,SAAS,KAAK,MAAM;CAQ/C,OAAO;EAAE;EAAO,UAAA;GANd,MAAM,aAAa;IAAE;IAAM;IAAiB;IAAS,QAAQ;GAAa,CAAC;GAC3E;GACA;GACA;GACA,QAAQ;EAEa;CAAE;AAC3B;;;;;;;AAQA,SAAS,SAAY,OAAqB,QAA2B;CACnE,OAAO,MACJ,KAAK,UAAU;EAAE;EAAM,KAAK,OAAO;CAAE,EAAE,CAAC,CACxC,MAAM,MAAM,UAAU,KAAK,MAAM,MAAM,GAAG,CAAC,CAC3C,KAAK,UAAU,MAAM,IAAI;AAC9B;AAEA,SAAS,aAAa,UAAmD;CACvE,OAAO,cAAc;EACnB,QAAQ;EACR,MAAM,SAAS;EACf,iBAAiB,SAAS;EAC1B,SAAS,SAAS;EAClB,QAAQ,SAAS;CACnB,CAAC;AACH;;;;;;;;;AAwEA,SAAgB,UACd,SACA,UACiB;CACjB,MAAM,eAAe,aAAa;EAChC,MAAM,SAAS;EACf,iBAAiB,SAAS;EAC1B,SAAS,SAAS;EAClB,QAAQ,SAAS;CACnB,CAAC;CACD,IAAI,iBAAiB,SAAS,MAC5B,MAAM,IAAI,sBACR,wCAAwC,aAAa,iCAAiC,SAAS,KAAK,6CACtG;CAGF,MAAM,YAAY,IAAI,IAAI,SAAS,OAAO;CAC1C,MAAM,gCAAgB,IAAI,IAA2B;CACrD,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,UAAU,IAAI,OAAO,MAAM,GAC9B,MAAM,IAAI,gBACR,0BAA0B,OAAO,OAAO,wCAC1C;EAEF,IAAI,cAAc,IAAI,OAAO,MAAM,GACjC,MAAM,IAAI,gBAAgB,oCAAoC,OAAO,OAAO,EAAE;EAEhF,IAAI,OAAO,UAAU,QAAQ,CAAC,OAAO,SAAS,OAAO,KAAK,GACxD,MAAM,IAAI,gBACR,0BAA0B,OAAO,OAAO,cAAc,OAAO,MAAM,mCACrE;EAEF,cAAc,IAAI,OAAO,QAAQ,OAAO,KAAK;CAC/C;CAEA,MAAM,SAAsD,CAAC;CAC7D,MAAM,YAAsB,CAAC;CAC7B,MAAM,aAAuB,CAAC;CAC9B,IAAI,SAAS;CACb,IAAI,SAAS;CACb,IAAI,aAAa;CAEjB,KAAK,MAAM,SAAS,SAAS,QAAQ;EACnC,IAAI,SAAS,OAAO,MAAM;EAC1B,IAAI,WAAW,KAAA,GAAW;GACxB,SAAS;IAAE,QAAQ;IAAG,QAAQ;IAAG,QAAQ;IAAG,YAAY;IAAG,MAAM;GAAK;GACtE,OAAO,MAAM,QAAQ;EACvB;EACA,OAAO,UAAU;EACjB,IAAI,CAAC,cAAc,IAAI,MAAM,KAAK,MAAM,GAAG;GACzC,WAAW,KAAK,MAAM,EAAE;GACxB;EACF;EACA,MAAM,QAAQ,cAAc,IAAI,MAAM,KAAK,MAAM,KAAK;EACtD,IAAI,UAAU,MAAM;GAClB,cAAc;GACd,OAAO,cAAc;GACrB;EACF;EAEA,KADqC,SAAS,SAAS,kBAAkB,WAAW,cACjE,MAAM,iBAAiB;GACxC,UAAU;GACV,OAAO,UAAU;EACnB,OAAO;GACL,UAAU;GACV,OAAO,UAAU;GACjB,UAAU,KAAK,MAAM,EAAE;EACzB;CACF;CAEA,MAAM,SAAS,SAAS,OAAO;CAC/B,MAAM,UAAU,SAAS;CACzB,MAAM,SAA0B;EAC9B,QAAQ;EACR;EACA;EACA;EACA;EACA,MAAM;EACN;EACA;EACA;EACA,UAAU,kBAAkB,UAAU,aAAa;CACrD;CAEA,IAAI,WAAW,GACb,OAAO;EACL,GAAG;EACH,QAAQ;EACR,QAAQ;CACV;CAEF,IAAI,WAAW,SAAS,GACtB,OAAO;EACL,GAAG;EACH,QAAQ;EACR,QAAQ,GAAG,WAAW,OAAO,MAAM,OAAO,iCAAiC,WAAW,KAAK,IAAI;CACjG;CAEF,IAAI,YAAY,GACd,OAAO;EACL,GAAG;EACH,QAAQ;EACR,QAAQ,OAAO,OAAO;CACxB;CAGF,KAAK,MAAM,UAAU,OAAO,OAAO,MAAM,GAAG;EAC1C,MAAM,cAAc,OAAO,SAAS,OAAO;EAC3C,OAAO,OAAO,gBAAgB,IAAI,OAAO,OAAO,SAAS;CAC3D;CACA,OAAO,OAAO,SAAS;CACvB,OAAO;AACT;AAEA,SAAS,kBACP,UACA,eACmB;CACnB,MAAM,eAAe,IAAI,IAAI,SAAS,OAAO,KAAK,UAAU,MAAM,KAAK,MAAM,CAAC;CAC9E,IAAI,IAAI;CACR,IAAI,UAAU;CACd,IAAI,WAAW;CACf,KAAK,MAAM,UAAU,SAAS,SAAS;EACrC,IAAI,aAAa,IAAI,MAAM,GAAG;EAC9B,KAAK;EACL,MAAM,QAAQ,cAAc,IAAI,MAAM;EACtC,IAAI,UAAU,KAAA,KAAa,UAAU,MAAM;EAC3C,WAAW;EACX,IAAI,QAAQ,SAAS,iBAAiB,YAAY;CACpD;CACA,OAAO;EAAE;EAAG;EAAS;EAAU,eAAe,YAAY,IAAI,OAAO,WAAW;CAAQ;AAC1F;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AChuBA,MAAM,wBAAwB;CAAC;CAAO;CAAoB;AAAkB;AAmC5E,SAAS,SAAS,OAAe,OAAuB;CACtD,MAAM,KAAK,OAAO,UAAU,WAAW,KAAK,MAAM,KAAK,IAAI;CAC3D,IAAI,CAAC,OAAO,SAAS,EAAE,GACrB,MAAM,IAAI,gBACR,mBAAmB,MAAM,qCAAqC,KAAK,UAAU,KAAK,GACpF;CAEF,OAAO;AACT;;AAGA,SAAgB,yBAAyB,UAA4B,SAAS,YAAkB;CAC9F,SAAS,SAAS,IAAI,GAAG,OAAO,IAAI;CACpC,IAAI,OAAO,SAAS,YAAY,YAAY,SAAS,QAAQ,WAAW,GACtE,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,oCAAoC;CAE1F,IAAI,OAAO,SAAS,eAAe,YAAY,SAAS,WAAW,WAAW,GAC5E,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,uCAAuC;CAE7F,IAAI,SAAS,YAAY,QAAQ,OAAO,SAAS,YAAY,UAC3D,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,0CAA0C;CAEhG,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,SAAS,OAAO,GAAG;EAC3D,IAAI,CAAE,sBAA4C,SAAS,GAAG,GAE5D,MAAM,IAAI,gBACR,mBAAmB,OAAO,gCAAgC,IAAI,aAAa,sBAAsB,KAAK,IAAI,GAC5G;EAEF,IAAI,UAAU,KAAA,KAAa,CAAC,OAAO,SAAS,KAAK,GAC/C,MAAM,IAAI,gBACR,mBAAmB,OAAO,WAAW,IAAI,2BAA2B,OAAO,KAAK,GAClF;CAEJ;AACF;;;;;;AASA,SAAgB,wBACd,QACA,MACkB;CAClB,MAAM,QAAQ,6BAA6B,SAAS,OAAO,0BAA0B,OAAO;CAC5F,IAAI,CAAC,OAAO,SAAS,KAAK,GACxB,MAAM,IAAI,gBACR,+CAA+C,OAAO,EAAE,iEAC1D;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,kBAAkB,MAAM;CAAE;CACnF,yBAAyB,QAAQ;CACjC,OAAO;AACT;;;;;;;;AASA,SAAgB,sBACd,QACA,MACkB;CAClB,MAAM,MAAM,gBAAgB,SAAS,OAAO,aAAa,OAAO;CAChE,IAAI,CAAC,OAAO,SAAS,GAAG,GACtB,MAAM,IAAI,gBACR,4FACF;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,IAAI;CAAE;CAC/D,yBAAyB,QAAQ;CACjC,OAAO;AACT;;;;;;;;AAcA,SAAgB,wBACd,QACA,QACA,MACA,UAA8B,CAAC,GACb;CAClB,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,SAAS,SAAS,KAAK,YAAY,GAC7C,MAAM,IAAI,gBACR,uEAAuE,WACzE;CAEF,MAAM,6BAAa,IAAI,IAAoB;CAC3C,KAAK,MAAM,QAAQ,QAAQ;EACzB,IAAI,CAAC,OAAO,SAAS,KAAK,UAAU,GAClC,MAAM,IAAI,gBACR,yCAAyC,KAAK,OAAO,8BACvD;EAEF,IAAI,WAAW,IAAI,KAAK,MAAM,GAC5B,MAAM,IAAI,gBAAgB,qDAAqD,KAAK,OAAO,EAAE;EAE/F,WAAW,IAAI,KAAK,QAAQ,KAAK,UAAU;CAC7C;CACA,IAAI,SAAS;CACb,IAAI,SAAS;CACb,KAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,QAAQ,WAAW,IAAI,EAAE,MAAM;EACrC,IAAI,UAAU,KAAA,GAAW;EACzB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAC1B,MAAM,IAAI,gBACR,kDAAkD,EAAE,OAAO,gBAC7D;EAEF;EACA,IAAI,KAAK,IAAI,EAAE,QAAQ,KAAK,KAAK,WAAW;CAC9C;CACA,IAAI,WAAW,GACb,MAAM,IAAI,gBACR,yGACF;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,kBAAkB,SAAS,OAAO;CAAE;CAC7F,yBAAyB,QAAQ;CACjC,OAAO;AACT;AAUA,SAAgB,sBAAsB,UAA8B,CAAC,GAAkB;CACrF,KAAK,MAAM,KAAK,SAAS,yBAAyB,CAAC;CACnD,MAAM,QAAQ,QAAQ,KAAK,OAAO;EAAE,GAAG;EAAG,SAAS,EAAE,GAAG,EAAE,QAAQ;CAAE,EAAE;CACtE,OAAO;EACL,MAAM,OAAO,UAAU;GACrB,yBAAyB,QAAQ;GACjC,MAAM,KAAK;IAAE,GAAG;IAAU,SAAS,EAAE,GAAG,SAAS,QAAQ;GAAE,CAAC;EAC9D;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,MAAM,MAAM,KAAK,OAAO;IAAE,GAAG;IAAG,SAAS,EAAE,GAAG,EAAE,QAAQ;GAAE,EAAE;GAClE,OAAO,YAAY,KAAA,IAAY,MAAM,IAAI,QAAQ,MAAM,EAAE,YAAY,OAAO;EAC9E;CACF;AACF;;;;;;;AAQA,SAAgB,kBAAkB,MAA6B;CAC7D,OAAO;EACL,MAAM,OAAO,UAAU;GACrB,yBAAyB,QAAQ;GACjC,MAAM,KAAK,MAAM,OAAO;GACxB,MAAM,UAAU,MAAM,OAAO;GAC7B,MAAM,GAAG,MAAM,QAAQ,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;GACzD,MAAM,GAAG,WAAW,MAAM,GAAG,KAAK,UAAU,QAAQ,EAAE,KAAK,MAAM;EACnE;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,KAAK,MAAM,OAAO;GACxB,IAAI;GACJ,IAAI;IACF,MAAM,MAAM,GAAG,SAAS,MAAM,MAAM;GACtC,SAAS,KAAK;IACZ,IAAK,IAA8B,SAAS,UAAU,OAAO,CAAC;IAC9D,MAAM;GACR;GACA,MAAM,QAAQ,IAAI,MAAM,IAAI;GAC5B,MAAM,YAAgC,CAAC;GACvC,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;IACrC,MAAM,OAAO,MAAM;IACnB,IAAI,KAAK,KAAK,CAAC,CAAC,WAAW,GAAG;IAC9B,IAAI;IACJ,IAAI;KACF,SAAS,KAAK,MAAM,IAAI;IAC1B,SAAS,KAAK;KACZ,MAAM,IAAI,gBACR,4BAA4B,KAAK,4BAA4B,IAAI,EAAE,IAAK,IAAc,SACxF;IACF;IACA,MAAM,WAAW;IACjB,yBAAyB,UAAU,GAAG,KAAK,GAAG,IAAI,GAAG;IACrD,UAAU,KAAK,QAAQ;GACzB;GACA,OAAO,YAAY,KAAA,IAAY,YAAY,UAAU,QAAQ,MAAM,EAAE,YAAY,OAAO;EAC1F;CACF;AACF;AAmDA,MAAM,aAAa;AAEnB,SAAgB,oBACd,SACA,MACgB;CAChB,MAAM,SAAS,SAAS,KAAK,MAAM,MAAM;CACzC,MAAM,SAAS,KAAK,YAAY,UAAU;CAC1C,MAAM,eAAe,KAAK,YAAY,gBAAgB;CACtD,MAAM,kBAAkB,KAAK,YAAY,mBAAmB;CAC5D,MAAM,iBAAiB,KAAK,YAAY,kBAAkB;CAE1D,MAAM,0BAAU,IAAI,IAAwD;CAC5E,KAAK,MAAM,YAAY,SAAS;EAC9B,yBAAyB,QAAQ;EACjC,MAAM,OAAO,SAAS,SAAS,IAAI,uBAAuB,SAAS,QAAQ,GAAG;EAC9E,MAAM,MAAM,QAAQ,IAAI,SAAS,OAAO,KAAK,CAAC;EAC9C,IAAI,KAAK;GAAE,GAAG;GAAU;EAAK,CAAC;EAC9B,QAAQ,IAAI,SAAS,SAAS,GAAG;CACnC;CAEA,MAAM,WAA4B,CAAC;CACnC,MAAM,SAAmB,CAAC;CAC1B,MAAM,sBAAgC,CAAC;CAEvC,KAAK,MAAM,WAAW,CAAC,GAAG,QAAQ,KAAK,CAAC,CAAC,CAAC,KAAK,GAAG;EAChD,MAAM,QAAQ,QAAQ,IAAI,OAAO,CAAC,CAAE,MAAM,GAAG,MAAM,EAAE,OAAO,EAAE,IAAI;EAClE,MAAM,SAAS,MAAM,MAAM,SAAS;EAEpC,MAAM,WAAW,SAAS,OAAO,QAAQ;EACzC,IAAI,UAAU,gBACZ,OAAO,KACL,UAAU,QAAQ,6BAA6B,OAAO,GAAG,MAAM,QAAQ,QAAQ,CAAC,EAAE,uBAAuB,eAAe,GAC1H;EAMF,IAAI,YAAY;EAChB,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAChC,IAAI,MAAM,EAAE,CAAE,eAAe,MAAM,IAAI,EAAE,CAAE,YAAY,YAAY;EAErE,IAAI,aAAa,GAQX;OAAA,CAPiB,MAClB,MAAM,SAAS,CAAC,CAChB,MACE,MACC,OAAO,SAAS,EAAE,QAAQ,gBAAgB,KAC1C,OAAO,SAAS,EAAE,QAAQ,gBAAgB,CAEhC,GACd,OAAO,KACL,UAAU,QAAQ,oBAAoB,MAAM,YAAY,EAAE,CAAE,WAAW,OAAO,MAAM,UAAU,CAAE,WAAW,OAAO,MAAM,UAAU,CAAE,GAAG,gIACzI;EAAA;EAIJ,KAAK,MAAM,UAAU,uBAAuB;GAC1C,MAAM,SAAS,MACZ,QAAQ,MAAM,OAAO,EAAE,QAAQ,YAAY,QAAQ,CAAC,CACpD,KAAK,MAAM,EAAE,QAAQ,OAAQ;GAChC,IAAI,OAAO,WAAW,GAAG;GAEzB,MAAM,cAAc,cAAc,QAAQ,KAAK,WAAW;GAC1D,IAAI,YAAY,UAAU,qBACxB,oBAAoB,KAAK,GAAG,QAAQ,GAAG,QAAQ;GAEjD,MAAM,UAAU,OAAO,OAAO,SAAS;GACvC,MAAM,WAAW,OAAO;GACxB,MAAM,QAAQ,UAAU;GACxB,MAAM,UAAoB,CAAC;GAC3B,IAAI,YAAY,UAAU,iBACxB,QAAQ,KACN,6CAA6C,YAAY,QAAQ,gBAAgB,YAAY,WAAW,QAAQ,CAAC,EAAE,EACrH;GAEF,IAAI,WAAW,SAAS,UAAU,QAChC,QAAQ,KAAK,OAAO,QAAQ,QAAQ,CAAC,EAAE,eAAe,QAAQ;GAEhE,IAAI,WAAW,sBAAsB,WAAW,UAAU,cACxD,QAAQ,KACN,6BAA6B,WAAW,QAAA,CAAS,QAAQ,CAAC,EAAE,iBAAiB,SAAS,QAAQ,CAAC,EAAE,UAAU,aAAa,EAC1H;GAEF,IAAI,WAAW,sBAAsB,WAAW,UAAU,iBACxD,QAAQ,KACN,6BAA6B,WAAW,QAAA,CAAS,QAAQ,CAAC,EAAE,iBAAiB,SAAS,QAAQ,CAAC,EAAE,UAAU,gBAAgB,EAC7H;GAGF,MAAM,UAAU,QAAQ,SAAS;GACjC,MAAM,QAAuB;IAC3B;IACA;IACA,OAAO,YAAY;IACnB;IACA;IACA;IACA;GACF;GACA,IAAI,SAAS;IACX,MAAM,SAAS,QAAQ,KAAK,IAAI;IAChC,OAAO,KAAK,UAAU,QAAQ,IAAI,OAAO,IAAI,MAAM,QAAQ;GAC7D;GACA,SAAS,KAAK,KAAK;EACrB;CACF;CAEA,OAAO;EAAE;EAAU;EAAQ,SAAS,OAAO,WAAW;EAAG;CAAoB;AAC/E;;;;;;;;;AAiBA,SAAgB,gBAAgB,QAAyC;CACvE,OAAO;EAAE,SAAS,OAAO;EAAS,QAAQ,CAAC,GAAG,OAAO,MAAM;CAAE;AAC/D"}
|
|
1
|
+
{"version":3,"file":"index.js","names":[],"sources":["../../src/meta-eval/calibration.ts","../../src/meta-eval/correlation-study.ts","../../src/meta-eval/evaluator-admission.ts","../../src/meta-eval/plants.ts","../../src/meta-eval/sentinel.ts"],"sourcesContent":["/**\n * Calibration curve — binned \"if eval says X, what does reality show?\"\n *\n * Companion to correlationStudy. Raw correlation is a single number;\n * the calibration curve shows *where* the eval is well-calibrated vs\n * overconfident / underconfident. Buckets the eval metric, computes\n * mean outcome per bucket, reports expected-calibration-error (ECE).\n */\n\nimport { runMetricExtractor } from '../trace/query'\nimport type { TraceStore } from '../trace/store'\nimport type { EvalMetricSpec } from './correlation-study'\nimport { assertUniqueObservationIds, reduceOutcomeMetric } from './outcome-observations'\nimport type { DeploymentOutcome, OutcomeStore } from './outcome-store'\n\nexport interface CalibrationBin {\n lower: number\n upper: number\n n: number\n evalMean: number\n outcomeMean: number\n /** |outcomeMean − evalMean|; contributes to ECE weighted by n/total. */\n gap: number\n}\n\nexport interface CalibrationReport {\n evalMetric: string\n outcomeMetric: string\n n: number\n bins: CalibrationBin[]\n /** Expected Calibration Error — Σ (n_i/N) × |outcomeMean_i − evalMean_i|. */\n ece: number\n /** Largest observed difference between a bin's mean score and mean outcome. */\n maxGap: number\n}\n\nexport interface CalibrationOptions {\n /** Positive integer; empty bins are omitted. Default 10. */\n bins?: number\n /** Equal-width (fixed bin edges) or equal-frequency (quantile bins). */\n binning?: 'equal-width' | 'equal-frequency'\n /** Clip eval values to [lo, hi] before binning. */\n range?: { lo: number; hi: number }\n}\n\nexport interface CalibrationPair {\n evalScore: number\n outcome: number\n}\n\nexport async function calibrationCurve(\n traceStore: TraceStore,\n outcomeStore: OutcomeStore,\n evalMetric: EvalMetricSpec,\n outcomeMetric: string,\n options: CalibrationOptions = {},\n): Promise<CalibrationReport | null> {\n const settings = {\n ...options,\n range: options.range === undefined ? undefined : { ...options.range },\n }\n validateCalibrationRequest(evalMetric.id, outcomeMetric, settings)\n const extract = evalMetric.extract ?? runMetricExtractor(evalMetric.id)\n const metricId = evalMetric.id\n const runs = await traceStore.listRuns()\n assertUniqueObservationIds(\n runs.map((run) => run.runId),\n 'runId',\n )\n const outcomes = await outcomeStore.list()\n const byRun = new Map<string, DeploymentOutcome[]>()\n for (const o of outcomes) {\n const arr = byRun.get(o.runId) ?? []\n arr.push(o)\n byRun.set(o.runId, arr)\n }\n\n const pairs: Array<{ x: number; y: number }> = []\n for (const run of runs) {\n const os = byRun.get(run.runId)\n if (!os?.length) continue\n const x = await extract(run, traceStore)\n if (x === null || !Number.isFinite(x)) continue\n const y = reduceOutcomeMetric(os, outcomeMetric, 'latest')\n if (y === null) continue\n pairs.push({ x, y })\n }\n if (pairs.length < 2) return null\n\n return calibrationFromPairs(\n pairs.map((p) => ({ evalScore: p.x, outcome: p.y })),\n metricId,\n outcomeMetric,\n settings,\n )\n}\n\n/** Measure already joined observations without constructing trace and outcome stores. */\nexport function calibrationFromPairs(\n inputPairs: readonly CalibrationPair[],\n evalMetric: string,\n outcomeMetric: string,\n options: CalibrationOptions = {},\n): CalibrationReport | null {\n validateCalibrationRequest(evalMetric, outcomeMetric, options)\n for (const [index, pair] of inputPairs.entries()) {\n if (\n pair === null ||\n typeof pair !== 'object' ||\n !Number.isFinite(pair.evalScore) ||\n !Number.isFinite(pair.outcome)\n ) {\n throw new Error(`calibration pair ${index} must contain finite evalScore and outcome values`)\n }\n }\n const pairs = inputPairs\n if (pairs.length < 2) return null\n\n const numBins = options.bins ?? 10\n const binning = options.binning ?? 'equal-width'\n const xs = pairs.map((p) => p.evalScore)\n const lo = options.range?.lo ?? Math.min(...xs)\n const hi = options.range?.hi ?? Math.max(...xs)\n const span = hi - lo\n if (!Number.isFinite(span)) throw new Error('calibration range span must be finite')\n const clipped = pairs.map((pair) => ({\n ...pair,\n evalScore: Math.min(hi, Math.max(lo, pair.evalScore)),\n }))\n\n const bins: CalibrationBin[] = []\n if (span === 0 || clipped.every((pair) => pair.evalScore === clipped[0]!.evalScore)) {\n bins.push(toBin(clipped))\n } else if (binning === 'equal-frequency') {\n const sorted = [...clipped].sort((a, b) => a.evalScore - b.evalScore)\n const count = Math.min(numBins, sorted.length)\n for (let i = 0; i < count; i++) {\n const start = Math.floor((i * sorted.length) / count)\n const end = Math.floor(((i + 1) * sorted.length) / count)\n bins.push(toBin(sorted.slice(start, end)))\n }\n } else {\n const groups = new Map<number, CalibrationPair[]>()\n for (const pair of clipped) {\n const index = Math.min(numBins - 1, Math.floor(((pair.evalScore - lo) / span) * numBins))\n const group = groups.get(index) ?? []\n group.push(pair)\n groups.set(index, group)\n }\n for (const [index, chunk] of [...groups].sort(([a], [b]) => a - b)) {\n bins.push(toBin(chunk, lo + span * (index / numBins), lo + span * ((index + 1) / numBins)))\n }\n }\n\n const total = bins.reduce((a, b) => a + b.n, 0)\n const ece = bins.reduce((a, b) => a + (b.n / total) * b.gap, 0)\n const maxGap = bins.reduce((a, b) => Math.max(a, b.gap), 0)\n\n return { evalMetric, outcomeMetric, n: pairs.length, bins, ece, maxGap }\n}\n\nfunction toBin(chunk: CalibrationPair[], lower?: number, upper?: number): CalibrationBin {\n const xs = chunk.map((c) => c.evalScore)\n const ys = chunk.map((c) => c.outcome)\n const evalMean = mean(xs)\n const outcomeMean = mean(ys)\n return {\n lower: lower ?? Math.min(...xs),\n upper: upper ?? Math.max(...xs),\n n: chunk.length,\n evalMean,\n outcomeMean,\n gap: Math.abs(outcomeMean - evalMean),\n }\n}\n\nfunction mean(xs: number[]): number {\n return xs.reduce((sum, value) => sum + value / xs.length, 0)\n}\n\nfunction validateCalibrationRequest(\n evalMetric: string,\n outcomeMetric: string,\n options: CalibrationOptions,\n): void {\n assertUniqueObservationIds([evalMetric], 'eval metric')\n assertUniqueObservationIds([outcomeMetric], 'outcome metric')\n if (evalMetric.trim() !== evalMetric || outcomeMetric.trim() !== outcomeMetric) {\n throw new Error('calibration metric identities must not have surrounding whitespace')\n }\n if (options.bins !== undefined && (!Number.isSafeInteger(options.bins) || options.bins < 1)) {\n throw new Error('calibration bins must be a positive safe integer')\n }\n if (\n options.binning !== undefined &&\n !['equal-width', 'equal-frequency'].includes(options.binning)\n ) {\n throw new Error('calibration binning must be equal-width or equal-frequency')\n }\n if (\n options.range !== undefined &&\n (!Number.isFinite(options.range.lo) ||\n !Number.isFinite(options.range.hi) ||\n !Number.isFinite(options.range.hi - options.range.lo) ||\n options.range.hi < options.range.lo)\n ) {\n throw new Error('calibration range must have finite ordered bounds')\n }\n}\n","/**\n * Correlation study — \"does our eval score predict real-world outcomes?\"\n *\n * Joins traces and outcomes by runId and reports descriptive correlations.\n * Independent runs are the bootstrap observation unit.\n * Association alone does not establish causation or held-out predictive performance.\n */\n\nimport { runMetricExtractor } from '../trace/query'\nimport type { Run } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport {\n assertUniqueObservationIds,\n correlationSummary,\n hasVariation,\n reduceOutcomeMetric,\n validateObservationOptions,\n} from './outcome-observations'\nimport type { DeploymentOutcome, OutcomeFilter, OutcomeStore } from './outcome-store'\n\nexport interface EvalMetricSpec {\n id: string\n /** Extract a scalar from a run. Omit it and `id` must name one of\n * `RUN_METRICS`; any other `id` is refused. */\n extract?: (run: Run, store: TraceStore) => Promise<number | null>\n}\n\nexport interface OutcomePair {\n evalMetric: string\n outcomeMetric: string\n}\n\nexport interface CorrelationResult {\n evalMetric: string\n outcomeMetric: string\n n: number\n pearson: number\n spearman: number\n /** 95% bootstrap CI for Pearson. */\n pearsonCi95: { lower: number; upper: number } | null\n /** 95% bootstrap CI for Spearman; null when no resample is estimable. */\n spearmanCi95: { lower: number; upper: number } | null\n /** Rough verdict: 'strong' ≥ 0.7, 'moderate' ≥ 0.4, else 'weak'. */\n verdict: 'strong' | 'moderate' | 'weak'\n}\n\nexport interface CorrelationStudyResult {\n pairs: CorrelationResult[]\n /** Declared pairs without an estimable correlation, including their usable sample count. */\n excludedPairs: Array<\n OutcomePair & {\n n: number\n reason: 'insufficient_samples' | 'constant_eval_metric' | 'constant_outcome'\n }\n >\n joinedSamples: number\n skippedRuns: number\n}\n\nexport interface CorrelationStudyOptions {\n /** Only join outcomes captured within this window after run.startedAt. */\n maxCaptureLagMs?: number\n /** Restrict to a subset of outcomes (cohort, region, source). */\n outcomeFilter?: OutcomeFilter\n /** Which outcome per run to use when multiple exist. Default 'latest'. */\n reduction?: 'latest' | 'mean' | 'max'\n /** Bootstrap iterations for the CI. Default 500. */\n bootstrapIterations?: number\n /** Seed for the bootstrap resampler. Absent, the seed is derived from the\n * paired observations, so the same study reproduces the same interval. */\n seed?: number\n}\n\nexport async function correlationStudy(\n traceStore: TraceStore,\n outcomeStore: OutcomeStore,\n evalMetrics: EvalMetricSpec[],\n outcomeMetricNames: string[],\n options: CorrelationStudyOptions = {},\n): Promise<CorrelationStudyResult> {\n const reduction = options.reduction ?? 'latest'\n const iterations = options.bootstrapIterations ?? 500\n const seed = options.seed\n validateObservationOptions(reduction, iterations, seed)\n assertUniqueObservationIds(\n evalMetrics.map((metric) => metric.id),\n 'eval metric',\n )\n assertUniqueObservationIds(outcomeMetricNames, 'outcome metric')\n const maxLag = options.maxCaptureLagMs ?? Infinity\n if (maxLag < 0 || Number.isNaN(maxLag)) {\n throw new Error('maxCaptureLagMs must be nonnegative')\n }\n const extractors = evalMetrics.map((metric) => ({\n ...metric,\n extract: metric.extract ?? runMetricExtractor(metric.id),\n }))\n const metricNames = [...outcomeMetricNames]\n const runs = await traceStore.listRuns()\n assertUniqueObservationIds(\n runs.map((run) => run.runId),\n 'runId',\n )\n const outcomes = await outcomeStore.list(options.outcomeFilter)\n const outcomesByRun = new Map<string, DeploymentOutcome[]>()\n for (const o of outcomes) {\n const arr = outcomesByRun.get(o.runId) ?? []\n arr.push(o)\n outcomesByRun.set(o.runId, arr)\n }\n\n const pairs: Array<{ evalMetric: string; outcomeMetric: string; xs: number[]; ys: number[] }> = []\n for (const em of extractors) {\n for (const om of metricNames) {\n pairs.push({ evalMetric: em.id, outcomeMetric: om, xs: [], ys: [] })\n }\n }\n\n let joined = 0\n let skipped = 0\n for (const run of runs) {\n const os = outcomesByRun.get(run.runId)\n if (!os || os.length === 0) {\n skipped++\n continue\n }\n const eligible = os.filter((o) => {\n const lag = o.capturedAt - run.startedAt\n return lag >= 0 && lag <= maxLag\n })\n if (eligible.length === 0) {\n skipped++\n continue\n }\n\n let joinedThisRun = false\n for (const em of extractors) {\n const x = await em.extract(run, traceStore)\n if (x === null || !Number.isFinite(x)) continue\n\n for (const om of metricNames) {\n const y = reduceOutcomeMetric(eligible, om, reduction)\n if (y === null) continue\n const pair = pairs.find((p) => p.evalMetric === em.id && p.outcomeMetric === om)!\n pair.xs.push(x)\n pair.ys.push(y)\n joinedThisRun = true\n }\n }\n if (joinedThisRun) joined++\n else skipped++\n }\n\n const excludedPairs: CorrelationStudyResult['excludedPairs'] = []\n const results: CorrelationResult[] = []\n for (const p of pairs) {\n const reason =\n p.xs.length < 3\n ? 'insufficient_samples'\n : !hasVariation(p.xs)\n ? 'constant_eval_metric'\n : !hasVariation(p.ys)\n ? 'constant_outcome'\n : null\n if (reason !== null) {\n excludedPairs.push({\n evalMetric: p.evalMetric,\n outcomeMetric: p.outcomeMetric,\n n: p.xs.length,\n reason,\n })\n continue\n }\n const summary = correlationSummary(p.xs, p.ys, iterations, seed)\n const verdict: CorrelationResult['verdict'] =\n Math.abs(summary.pearson) >= 0.7\n ? 'strong'\n : Math.abs(summary.pearson) >= 0.4\n ? 'moderate'\n : 'weak'\n results.push({\n evalMetric: p.evalMetric,\n outcomeMetric: p.outcomeMetric,\n n: p.xs.length,\n ...summary,\n verdict,\n })\n }\n return { pairs: results, excludedPairs, joinedSamples: joined, skippedRuns: skipped }\n}\n","import { z } from 'zod'\nimport { ValidationError } from '../errors'\nimport { type ComputedInterval, computeInterval } from '../experiment/ast'\nimport { compareCodeUnits, hashCanonical, type LedgerHash } from '../ledger-core/canonical'\n\nconst identity = z\n .string()\n .min(1)\n .refine((value) => value.trim() === value)\nconst policySchema = z\n .object({\n confidence: z.number().finite().gt(0).lt(1),\n maxFalseAcceptanceRate: z.number().finite().min(0).lt(1),\n maxFalseRejectionRate: z.number().finite().min(0).lt(1),\n })\n .strict()\n\n/** Register both error limits before inspecting audit judgments. */\nexport type EvaluatorAdmissionPolicy = z.infer<typeof policySchema>\n\nconst observationSchema = z\n .object({\n id: identity,\n independentUnitId: identity,\n evidenceRef: identity,\n expected: z.enum(['accept', 'reject']),\n observed: z.enum(['accept', 'reject', 'unknown']),\n exposure: z.enum(['fresh', 'development']),\n })\n .strict()\n\n/** Actual control judgments. Variants from one source retain one independent identity. */\nexport type EvaluatorAuditObservation = z.infer<typeof observationSchema>\n\nconst inputSchema = z\n .object({\n evaluatorDigest: z.string().regex(/^sha256:[a-f0-9]{64}$/),\n population: identity,\n samplingFrame: identity,\n authority: z\n .object({\n evaluatorAuthorId: identity,\n auditorId: identity,\n independenceEvidenceRef: identity,\n })\n .strict(),\n policy: policySchema,\n observations: z.array(observationSchema),\n })\n .strict()\n\nexport type EvaluatorAuditInput = z.infer<typeof inputSchema>\n\nexport interface EvaluatorErrorRate {\n /** Eligible control judgments in this class, including unknown judgments. */\n cases: number\n independentUnits: number\n /** Units with at least one observed mistake. */\n errorUnits: number\n /** Units with unknown judgments and no observed mistake yet. */\n unresolvedUnits: number\n unknownCases: number\n /** Missing judgments do not become measured zeros. */\n errorRate: number | null\n /** Exact bounds conservatively include every possible outcome of unknown judgments. */\n interval: ComputedInterval | null\n limit: number\n verdict: 'pass' | 'fail' | 'inconclusive'\n}\n\nexport interface EvaluatorAdmissionReport {\n evaluatorDigest: LedgerHash\n policyDigest: LedgerHash\n inputDigest: LedgerHash\n reportDigest: LedgerHash\n population: string\n samplingFrame: string\n authority: EvaluatorAuditInput['authority']\n policy: EvaluatorAdmissionPolicy\n /** Joint coverage for the two reported error-rate intervals. */\n confidence: number\n intervalConfidence: number\n verdict: 'admit' | 'reject' | 'inconclusive'\n reasons: string[]\n observations: EvaluatorAuditObservation[]\n coverage: {\n cases: number\n independentUnits: number\n eligibleCases: number\n eligibleIndependentUnits: number\n excludedCases: number\n unknownCases: number\n }\n exclusions: Array<{ id: string; independentUnitId: string; reason: 'development-exposure' }>\n falseAcceptance: EvaluatorErrorRate\n falseRejection: EvaluatorErrorRate\n}\n\nfunction errorRate(\n rows: readonly EvaluatorAuditObservation[],\n expected: 'accept' | 'reject',\n limit: number,\n level: number,\n): EvaluatorErrorRate {\n const cases = rows.filter((row) => row.expected === expected)\n const units = new Map<string, { error: boolean; unknown: boolean }>()\n for (const row of cases) {\n const unit = units.get(row.independentUnitId) ?? { error: false, unknown: false }\n unit.unknown ||= row.observed === 'unknown'\n unit.error ||= row.observed !== 'unknown' && row.observed !== expected\n units.set(row.independentUnitId, unit)\n }\n const n = units.size\n const errorUnits = [...units.values()].filter((unit) => unit.error).length\n const unresolvedUnits = [...units.values()].filter((unit) => unit.unknown && !unit.error).length\n const unknownCases = cases.filter((row) => row.observed === 'unknown').length\n // Unknown judgments contribute to n; their most favorable and adverse outcomes bound the rate.\n const interval =\n n === 0\n ? null\n : {\n lower: computeInterval(\n { kind: 'clopper-pearson', level },\n {\n kind: 'binomial',\n successes: errorUnits,\n trials: n,\n },\n ).lower,\n upper: computeInterval(\n { kind: 'clopper-pearson', level },\n {\n kind: 'binomial',\n successes: errorUnits + unresolvedUnits,\n trials: n,\n },\n ).upper,\n level,\n }\n const verdict =\n interval && interval.lower > limit\n ? 'fail'\n : interval && interval.upper <= limit\n ? 'pass'\n : 'inconclusive'\n return {\n cases: cases.length,\n independentUnits: n,\n errorUnits,\n unresolvedUnits,\n unknownCases,\n errorRate: n === 0 || unresolvedUnits > 0 ? null : errorUnits / n,\n interval,\n limit,\n verdict,\n }\n}\n\n/**\n * Audit frozen judgments using independent source units and exact binomial bounds.\n * A unit fails a class when any control in that class is misjudged.\n * The execution owner enforces auditor separation, fresh sampling, and evidence authenticity.\n */\nexport function auditEvaluator(input: EvaluatorAuditInput): EvaluatorAdmissionReport {\n const parsed = inputSchema.safeParse(input)\n if (!parsed.success) throw new ValidationError(`invalid evaluator audit: ${parsed.error.message}`)\n const audit = parsed.data\n if (audit.authority.evaluatorAuthorId === audit.authority.auditorId) {\n throw new ValidationError('evaluator admission requires a separate declared audit authority')\n }\n const ids = new Set<string>()\n for (const row of audit.observations) {\n if (ids.has(row.id))\n throw new ValidationError(`duplicate evaluator audit observation '${row.id}'`)\n ids.add(row.id)\n }\n audit.observations.sort((a, b) => compareCodeUnits(a.id, b.id))\n const developmentUnits = new Set(\n audit.observations\n .filter((row) => row.exposure === 'development')\n .map((row) => row.independentUnitId),\n )\n const eligible = audit.observations.filter((row) => !developmentUnits.has(row.independentUnitId))\n const exclusions = audit.observations\n .filter((row) => developmentUnits.has(row.independentUnitId))\n .map((row) => ({\n id: row.id,\n independentUnitId: row.independentUnitId,\n reason: 'development-exposure' as const,\n }))\n const intervalConfidence = 1 - (1 - audit.policy.confidence) / 2\n if (intervalConfidence >= 1)\n throw new ValidationError(\n 'evaluator audit confidence is too close to one for simultaneous intervals',\n )\n const falseAcceptance = errorRate(\n eligible,\n 'reject',\n audit.policy.maxFalseAcceptanceRate,\n intervalConfidence,\n )\n const falseRejection = errorRate(\n eligible,\n 'accept',\n audit.policy.maxFalseRejectionRate,\n intervalConfidence,\n )\n const rates = [falseAcceptance, falseRejection]\n const verdict = rates.some((rate) => rate.verdict === 'fail')\n ? 'reject'\n : rates.every((rate) => rate.verdict === 'pass')\n ? 'admit'\n : 'inconclusive'\n const reasons: string[] = []\n for (const [name, rate] of [\n ['false acceptance', falseAcceptance],\n ['false rejection', falseRejection],\n ] as const) {\n if (rate.independentUnits === 0) reasons.push(`${name}: no eligible independent units`)\n else if (rate.unknownCases > 0)\n reasons.push(`${name}: ${rate.unknownCases} judgments are unknown`)\n if (rate.verdict === 'fail') reasons.push(`${name}: lower error bound exceeds ${rate.limit}`)\n else if (rate.verdict === 'inconclusive' && rate.interval)\n reasons.push(`${name}: upper error bound does not establish the required limit`)\n }\n if (verdict === 'admit')\n reasons.push(\n 'both error bounds meet the registered limits, including the worst case for unknown judgments',\n )\n const body: Omit<EvaluatorAdmissionReport, 'reportDigest'> = {\n evaluatorDigest: audit.evaluatorDigest as LedgerHash,\n policyDigest: hashCanonical(audit.policy),\n inputDigest: hashCanonical(audit),\n population: audit.population,\n samplingFrame: audit.samplingFrame,\n authority: audit.authority,\n policy: audit.policy,\n confidence: audit.policy.confidence,\n intervalConfidence,\n verdict,\n reasons,\n observations: audit.observations,\n coverage: {\n cases: audit.observations.length,\n independentUnits: new Set(audit.observations.map((row) => row.independentUnitId)).size,\n eligibleCases: eligible.length,\n eligibleIndependentUnits: new Set(eligible.map((row) => row.independentUnitId)).size,\n excludedCases: exclusions.length,\n unknownCases: eligible.filter((row) => row.observed === 'unknown').length,\n },\n exclusions,\n falseAcceptance,\n falseRejection,\n }\n return { ...body, reportDigest: hashCanonical(body) }\n}\n","/**\n * Plants — seeded known-wrong items that measure the grader, not the work.\n *\n * A grading run reports how the work scored. It cannot report whether the\n * grader would have noticed a wrong answer, because every item it saw was\n * authored in good faith. A plant closes that hole: an item authored wrong by\n * construction is mixed into the live set, graded by the same path as\n * everything else, and the share of plants the grader refused is the catch\n * rate.\n *\n * Measured motive: a sibling lab ran a deliverable gate that accepted any\n * non-empty submission. It produced six false certifications in seventeen\n * deliveries, and no agent lied — the gate never asked a question the format\n * could fail. A catch rate is the number that would have shown it on day one.\n *\n * This module composes existing primitives rather than adding parallel ones:\n *\n * - A plant IS a {@link GoldenItem} from `../judge-calibration`. Its\n * `humanScore` is the grade a working grader owes the item, so the same\n * array feeds `calibrateJudge` unchanged.\n * - The grader's output is `CandidateScore[]`, the array `calibrateJudge` and\n * `snapshotFromSentinelSet` already consume.\n * - \"Caught\" is `snapshotFromSentinelSet`'s join with the labels inverted:\n * the grade lands on the side of `acceptThreshold` the seed demands.\n * - The manifest is sealed with `hashCanonical` from `../ledger-core/canonical`,\n * the digest the sealed-experiment path uses, so the answer key cannot be\n * revised once the results are in.\n *\n * Authoring the wrong item is the step before all of that, and it is where the\n * measurement silently breaks. {@link plantByPerturbation} takes a claim a\n * grader verified and alters exactly one load-bearing value in its evidence,\n * so the grader is asked a question it must fail. Hand-rolled perturbation\n * bumps a number that is part of an identity — `python3`, `file42.txt`, `1.5`,\n * `utf-8` — and the check stops EXECUTING; the plant then measures the\n * environment rather than the grader. {@link perturbEvidence} refuses those\n * digit runs by construction.\n *\n * Blindness has two halves, and this module owns one. It never puts a plant\n * flag on a graded item: `seedPlants` returns the mixed set and a manifest,\n * and only the manifest knows which ids are seeded. Keeping the manifest out\n * of the graded workspace and publishing its `seal` before grading is the\n * caller's half; {@link catchRate} refuses a manifest whose contents no longer\n * match its seal.\n *\n * Refusals, because a catch rate that cannot refuse is not a measurement:\n *\n * - a seeded id with no result makes the report `incomplete`, never a rate\n * over the results that did come back;\n * - zero seeded plants makes it `not_evaluated`, never 1.0;\n * - a result for an id the manifest never handed out is refused outright.\n */\n\nimport { CaptureIntegrityError, ValidationError } from '../errors'\nimport type { GoldenItem } from '../judge-calibration'\nimport { hashCanonical, type LedgerHash } from '../ledger-core/canonical'\nimport { mulberry32 } from '../statistics/random'\n\n/**\n * How a plant item was authored wrong. The class is reported separately in\n * {@link CatchRateReport.byKind} because a grader is routinely sharp on one\n * and blind to another.\n */\nexport type PlantKind =\n /** A load-bearing value is altered: a number off by one, a comparison flipped. */\n | 'wrong-value'\n /** The item carries its own check, and that check passes without testing the claim. */\n | 'self-certifying'\n /** The check names an input that does not exist, so it cannot run at all. */\n | 'unreachable-input'\n /** A copy of an item already in the set, which is owed a duplicate flag rather than a second grade. */\n | 'duplicate'\n\nconst PLANT_KINDS: readonly PlantKind[] = [\n 'wrong-value',\n 'self-certifying',\n 'unreachable-input',\n 'duplicate',\n]\n\n/** What a working grader owes a seeded item. */\nexport type PlantExpectation = 'reject' | 'accept'\n\nconst PLANT_EXPECTATIONS: readonly PlantExpectation[] = ['reject', 'accept']\n\n/** The label boundary a plant record is checked against at definition time. */\nconst RECORD_LABEL_BOUNDARY = 0.5\n\nexport interface Plant {\n /** Name of the plant record. Reported in `missedIds` and `missingIds`. */\n id: string\n kind: PlantKind\n /** The seeded item, indistinguishable from a real one once mixed. */\n item: GoldenItem\n /** The verdict a working grader owes this item. */\n expectedVerdict: PlantExpectation\n}\n\n/**\n * Build one plant record and refuse an incoherent one.\n *\n * The refusal that matters is the last: a record whose `expectedVerdict`\n * disagrees with `item.humanScore` inverts the measurement silently, because\n * the same item then reads as wrong here and as correct to every calibration\n * instrument that joins on the id.\n *\n * `item` is copied field by field so a later mutation of the caller's object\n * cannot change what the manifest sealed, and `group` is dropped when it is\n * absent so the record always has a canonical JSON form.\n */\nexport function definePlant(input: {\n id: string\n kind: PlantKind\n item: GoldenItem\n expectedVerdict: PlantExpectation\n}): Plant {\n const { id, kind, item, expectedVerdict } = input\n if (typeof id !== 'string' || id.trim() === '') {\n throw new ValidationError('definePlant: id must be a non-empty string')\n }\n if (!PLANT_KINDS.includes(kind)) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has kind ${JSON.stringify(kind)}; expected one of ${PLANT_KINDS.join(', ')}`,\n )\n }\n if (!PLANT_EXPECTATIONS.includes(expectedVerdict)) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has expectedVerdict ${JSON.stringify(expectedVerdict)}; expected one of ${PLANT_EXPECTATIONS.join(', ')}`,\n )\n }\n if (typeof item.itemId !== 'string' || item.itemId.trim() === '') {\n throw new ValidationError(`definePlant: plant \"${id}\" has an empty item.itemId`)\n }\n if (!Number.isFinite(item.humanScore) || item.humanScore < 0 || item.humanScore > 1) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has humanScore ${item.humanScore}; expected a finite number in [0, 1]`,\n )\n }\n if (item.group !== undefined && typeof item.group !== 'string') {\n throw new ValidationError(`definePlant: plant \"${id}\" has a non-string item.group`)\n }\n if (item.humanScore === RECORD_LABEL_BOUNDARY) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has humanScore ${RECORD_LABEL_BOUNDARY}, which states neither a rejection nor an acceptance`,\n )\n }\n const labelSays: PlantExpectation = item.humanScore < RECORD_LABEL_BOUNDARY ? 'reject' : 'accept'\n if (labelSays !== expectedVerdict) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" expects the grader to ${expectedVerdict} it, but humanScore ${item.humanScore} says ${labelSays}`,\n )\n }\n return {\n id,\n kind,\n item: {\n itemId: item.itemId,\n humanScore: item.humanScore,\n ...(item.group === undefined ? {} : { group: item.group }),\n },\n expectedVerdict,\n }\n}\n\n/**\n * A claim's executable evidence: the command a grader runs and the output it\n * must produce.\n */\nexport interface PlantEvidence {\n /** The command a grader runs to test the claim. */\n check?: string\n /** The output that command must produce for the claim to stand. */\n expect?: string\n}\n\n/** Which evidence field {@link perturbEvidence} altered. */\nexport type PerturbedField = 'check' | 'expect'\n\n/** How {@link perturbEvidence} altered it. */\nexport type PerturbationKind = 'number-off-by-one' | 'comparison-flipped'\n\nexport interface EvidencePerturbation {\n /** The evidence with exactly one value altered. */\n evidence: PlantEvidence\n /** Which field changed. */\n field: PerturbedField\n /** How it changed. */\n how: PerturbationKind\n /**\n * The exact substring that was replaced. A comparison token carries its\n * surrounding spaces, because the spaces are what make ` -ge ` a comparison\n * and not part of a word.\n */\n original: string\n /** What replaced it. */\n perturbed: string\n}\n\n/**\n * Characters that make a digit run part of something larger than a number.\n *\n * A run that touches one of these is an identity — `python3`, `file42.txt`,\n * `1.5`, `utf-8` — and bumping it renames a program, a file, a version, or an\n * encoding instead of falsifying the claim. The check then stops EXECUTING,\n * and a plant that cannot run measures the environment rather than the grader:\n * it reads as caught, for a reason that has nothing to do with the seeded\n * defect.\n */\nconst NUMBER_GLUE = /[A-Za-z0-9_./-]/\n\n/**\n * Comparison tokens that mean one thing only, each with its inversion.\n *\n * A bare `>` or `<` is absent on purpose: in a shell check it is a\n * redirection, so flipping it rewrites where output goes rather than what the\n * test asks, and the check stops testing the claim.\n */\nconst COMPARISON_FLIPS: readonly (readonly [string, string])[] = [\n [' -ge ', ' -lt '],\n [' -gt ', ' -le '],\n [' -le ', ' -gt '],\n [' -lt ', ' -ge '],\n [' -eq ', ' -ne '],\n [' -ne ', ' -eq '],\n ['>=', '<'],\n ['<=', '>'],\n ['==', '!='],\n ['!=', '=='],\n]\n\ninterface Span {\n start: number\n end: number\n}\n\n/**\n * The last digit run in `text` that is a number and nothing else, or null.\n *\n * The last is taken because a measured value trails its label — `count=42`,\n * `n>=5 OK cells=8` — and because the choice has to be deterministic: the same\n * claim must always yield the same plant, or two runs of the same authoring\n * step seed two different answer keys for one item.\n */\nfunction lastStandaloneNumber(text: string): Span | null {\n const digits = /\\d+/g\n let found: Span | null = null\n let match = digits.exec(text)\n while (match !== null) {\n const start = match.index\n const end = start + match[0].length\n // charAt returns '' outside the string, and NUMBER_GLUE never matches ''.\n if (!NUMBER_GLUE.test(text.charAt(start - 1)) && !NUMBER_GLUE.test(text.charAt(end))) {\n found = { start, end }\n }\n match = digits.exec(text)\n }\n return found\n}\n\n/**\n * The first flippable comparison in `text`, or null.\n *\n * Any single flip inverts the test wherever it sits, so unlike the number rule\n * there is no safety ordering among the matches. The first is taken so the\n * result is fixed by the head of the command and does not move when a later\n * comparison is added.\n */\nfunction firstComparison(text: string): { span: Span; flipped: string } | null {\n for (let index = 0; index < text.length; index += 1) {\n for (const [token, flipped] of COMPARISON_FLIPS) {\n if (text.startsWith(token, index)) {\n return { span: { start: index, end: index + token.length }, flipped }\n }\n }\n }\n return null\n}\n\nfunction replaceSpan(text: string, span: Span, value: string): string {\n return text.slice(0, span.start) + value + text.slice(span.end)\n}\n\n/**\n * Rebuild the evidence with one field replaced. An absent field is dropped\n * rather than set to undefined, so the record always has a canonical JSON form.\n */\nfunction withField(\n check: string | undefined,\n expect: string | undefined,\n field: PerturbedField,\n value: string,\n): PlantEvidence {\n const nextCheck = field === 'check' ? value : check\n const nextExpect = field === 'expect' ? value : expect\n return {\n ...(nextCheck === undefined ? {} : { check: nextCheck }),\n ...(nextExpect === undefined ? {} : { expect: nextExpect }),\n }\n}\n\nfunction offByOne(\n check: string | undefined,\n expect: string | undefined,\n field: PerturbedField,\n text: string,\n span: Span,\n): EvidencePerturbation {\n const original = text.slice(span.start, span.end)\n const perturbed = (BigInt(original) + 1n).toString()\n return {\n evidence: withField(check, expect, field, replaceSpan(text, span, perturbed)),\n field,\n how: 'number-off-by-one',\n original,\n perturbed,\n }\n}\n\n/**\n * Derive a known-wrong version of a claim's evidence by altering exactly one\n * value, or return null when no value can be altered safely.\n *\n * Exactly one value changes. A perturbation that moves two does not name the\n * defect it seeded, so a grader that catches it says nothing about which defect\n * the grader can see.\n *\n * The preference order IS the safety order:\n *\n * 1. the last standalone number in `expect` — it falsifies the claim while the\n * command still runs exactly as it did;\n * 2. a flipped comparison in `check` — it inverts a test that was passing;\n * 3. the last standalone number in `check` — last because a number in a command\n * can name an input (a port, a width, a version) rather than a threshold,\n * and renaming an input breaks execution, not the claim.\n *\n * \"Standalone\" excludes a digit run glued to a word, a dot, a slash, or a\n * hyphen: `python3`, `file42.txt`, `1.5`, `utf-8`. That exclusion is why this\n * helper exists. Hand-rolled perturbation bumps one of those, the check stops\n * executing, and the plant measures the environment instead of the grader.\n *\n * A number is bumped with BigInt, so a value wider than\n * `Number.MAX_SAFE_INTEGER` moves by exactly one. Bumped as a `number` it\n * rounds back to itself, and the plant is then a correct claim the grader is\n * right to verify.\n */\nexport function perturbEvidence(evidence: PlantEvidence): EvidencePerturbation | null {\n const check = typeof evidence.check === 'string' ? evidence.check : undefined\n const expect = typeof evidence.expect === 'string' ? evidence.expect : undefined\n\n if (expect !== undefined) {\n const span = lastStandaloneNumber(expect)\n if (span !== null) return offByOne(check, expect, 'expect', expect, span)\n }\n if (check !== undefined) {\n const comparison = firstComparison(check)\n if (comparison !== null) {\n return {\n evidence: withField(\n check,\n expect,\n 'check',\n replaceSpan(check, comparison.span, comparison.flipped),\n ),\n field: 'check',\n how: 'comparison-flipped',\n original: check.slice(comparison.span.start, comparison.span.end),\n perturbed: comparison.flipped,\n }\n }\n const span = lastStandaloneNumber(check)\n if (span !== null) return offByOne(check, expect, 'check', check, span)\n }\n return null\n}\n\nexport interface PerturbedPlant {\n plant: Plant\n /** The perturbed evidence to write into the item the grader is handed. */\n evidence: PlantEvidence\n field: PerturbedField\n how: PerturbationKind\n original: string\n perturbed: string\n}\n\n/**\n * Derive a reject-direction plant from a claim a grader verified, by perturbing\n * one value.\n *\n * {@link definePlant} takes a {@link GoldenItem} and never authors a wrong item\n * from a right one, so until now every caller hand-rolled the perturbation.\n * This is that step, done once and the same way every time: the claim's\n * evidence is altered by {@link perturbEvidence}, the result is a `wrong-value`\n * plant the grader owes a `reject`, and the record itself is built by\n * `definePlant` so every refusal there still applies — in particular the one\n * that catches an `expectedVerdict` its `humanScore` contradicts.\n *\n * Returns null when nothing in the evidence can be perturbed. That is not a\n * failure: a claim with no executable evidence cannot be made wrong in a way a\n * grader could execute, and seeding it would measure prose.\n *\n * Refuses an empty `id` or `itemId`. Both name the record — one in the manifest\n * and in `missedIds`, the other in the join with the results — and a miss\n * nobody can look up is not a measurement.\n */\nexport function plantByPerturbation(input: {\n id: string\n itemId: string\n evidence: PlantEvidence\n /** Score for the perturbed item. Must sit below the run's accept threshold. Default 0. */\n humanScore?: number\n group?: string\n}): PerturbedPlant | null {\n const { id, itemId, evidence, humanScore = 0, group } = input\n if (typeof id !== 'string' || id.trim() === '') {\n throw new ValidationError('plantByPerturbation: id must be a non-empty string')\n }\n if (typeof itemId !== 'string' || itemId.trim() === '') {\n throw new ValidationError(`plantByPerturbation: plant \"${id}\" has an empty itemId`)\n }\n const perturbation = perturbEvidence(evidence)\n if (perturbation === null) return null\n const plant = definePlant({\n id,\n kind: 'wrong-value',\n item: { itemId, humanScore, ...(group === undefined ? {} : { group }) },\n expectedVerdict: 'reject',\n })\n return {\n plant,\n evidence: perturbation.evidence,\n field: perturbation.field,\n how: perturbation.how,\n original: perturbation.original,\n perturbed: perturbation.perturbed,\n }\n}\n\nexport interface PlantManifest {\n /**\n * Digest over the seeded order, the plants, and the threshold. Publish it\n * before the grading run: a manifest edited afterwards to match the results\n * no longer matches its seal, and {@link catchRate} refuses it.\n */\n seal: LedgerHash\n /** The seed that fixed the mix order. */\n seed: number\n /** The grade at or above which the graded policy's own gate accepts an item. */\n acceptThreshold: number\n /** Every item id in the seeded set, in the order handed out. */\n itemIds: string[]\n /** The seeded plants. This is the answer key; keep it out of the graded workspace. */\n plants: Plant[]\n}\n\nexport interface SeededGradingSet {\n /**\n * The mixed set in seeded order. Operator-side: it still carries every\n * item's `humanScore`, so hand the graded policy the payload each `itemId`\n * names, never this array.\n */\n items: GoldenItem[]\n manifest: PlantManifest\n}\n\nexport interface SeedPlantsOptions {\n /**\n * Fixes the mix order. The same dataset, plants, threshold, and seed always\n * produce the same seeded order and the same seal.\n */\n seed?: number\n /**\n * The grade at or above which the graded policy's own gate accepts an item.\n * Supply the threshold your gate uses; the default suits a judge scoring in\n * [0, 1] with a pass at the midpoint.\n */\n acceptThreshold?: number\n}\n\n/**\n * Mix plants into a grading set and seal which items they are.\n *\n * Refusals: a duplicate id anywhere in the mixed set (the join is by id, so a\n * repeat makes one of the two unscoreable), a plant whose item id collides\n * with a dataset item (it would shadow real work and grade it as a plant), a\n * dataset item with no usable label, and a plant whose expectation the run's\n * `acceptThreshold` contradicts.\n */\nexport function seedPlants(\n dataset: readonly GoldenItem[],\n plants: readonly Plant[],\n options: SeedPlantsOptions = {},\n): SeededGradingSet {\n const seed = options.seed ?? 7\n const acceptThreshold = options.acceptThreshold ?? 0.5\n if (!Number.isFinite(seed)) {\n throw new ValidationError(`seedPlants: seed must be a finite number, got ${seed}`)\n }\n if (!Number.isFinite(acceptThreshold) || acceptThreshold <= 0 || acceptThreshold > 1) {\n throw new ValidationError(\n `seedPlants: acceptThreshold must be a finite number in (0, 1], got ${acceptThreshold}`,\n )\n }\n\n const datasetItems: GoldenItem[] = []\n const seen = new Set<string>()\n for (const item of dataset) {\n if (typeof item.itemId !== 'string' || item.itemId.trim() === '') {\n throw new ValidationError('seedPlants: a dataset item has an empty itemId')\n }\n if (!Number.isFinite(item.humanScore)) {\n throw new ValidationError(\n `seedPlants: dataset item \"${item.itemId}\" has a non-finite humanScore`,\n )\n }\n if (seen.has(item.itemId)) {\n throw new ValidationError(`seedPlants: duplicate dataset itemId \"${item.itemId}\"`)\n }\n seen.add(item.itemId)\n datasetItems.push({\n itemId: item.itemId,\n humanScore: item.humanScore,\n ...(item.group === undefined ? {} : { group: item.group }),\n })\n }\n\n const sealedPlants: Plant[] = []\n const plantIds = new Set<string>()\n for (const plant of plants) {\n const record = definePlant(plant)\n if (plantIds.has(record.id)) {\n throw new ValidationError(`seedPlants: duplicate plant id \"${record.id}\"`)\n }\n if (seen.has(record.item.itemId)) {\n throw new ValidationError(\n `seedPlants: plant \"${record.id}\" reuses itemId \"${record.item.itemId}\", which is already in the set`,\n )\n }\n const thresholdSays: PlantExpectation =\n record.item.humanScore >= acceptThreshold ? 'accept' : 'reject'\n if (thresholdSays !== record.expectedVerdict) {\n throw new ValidationError(\n `seedPlants: plant \"${record.id}\" expects the grader to ${record.expectedVerdict} it, but humanScore ${record.item.humanScore} is on the ${thresholdSays} side of acceptThreshold ${acceptThreshold}`,\n )\n }\n plantIds.add(record.id)\n seen.add(record.item.itemId)\n sealedPlants.push(record)\n }\n\n const items = shuffled(\n [...datasetItems, ...sealedPlants.map((plant) => plant.item)],\n mulberry32(seed),\n )\n const itemIds = items.map((item) => item.itemId)\n const manifest: PlantManifest = {\n seal: sealManifest({ seed, acceptThreshold, itemIds, plants: sealedPlants }),\n seed,\n acceptThreshold,\n itemIds,\n plants: sealedPlants,\n }\n return { items, manifest }\n}\n\n/**\n * Order by one independent uniform key per item, which is a uniform\n * permutation and a pure function of the seed. A comparator that returns a\n * fresh random sign instead is neither: it is not a consistent ordering, so\n * the permutation it produces is biased and depends on the sort algorithm.\n */\nfunction shuffled<T>(items: readonly T[], random: () => number): T[] {\n return items\n .map((item) => ({ item, key: random() }))\n .sort((left, right) => left.key - right.key)\n .map((entry) => entry.item)\n}\n\nfunction sealManifest(contents: Omit<PlantManifest, 'seal'>): LedgerHash {\n return hashCanonical({\n scheme: 'agent-eval.plant-manifest.v1',\n seed: contents.seed,\n acceptThreshold: contents.acceptThreshold,\n itemIds: contents.itemIds,\n plants: contents.plants,\n })\n}\n\n/**\n * One grader outcome for one item. `score: null` says the grader ran and\n * declined to decide — the check never tested the item's defect. Any\n * `CandidateScore` from a judge run is already a valid outcome.\n */\nexport interface PlantOutcome {\n itemId: string\n score: number | null\n}\n\n/**\n * `evaluated` — a rate stands. `incomplete` — a seeded id had no result.\n * `not_evaluated` — nothing was seeded, or nothing seeded was decided.\n */\nexport type CatchRateStatus = 'evaluated' | 'incomplete' | 'not_evaluated'\n\nexport interface PlantKindCounts {\n seeded: number\n caught: number\n missed: number\n indecisive: number\n /** caught / (caught + missed), or null when the report is not `evaluated`. */\n rate: number | null\n}\n\n/**\n * How the same grader treated the unseeded items of the same set. No labels\n * are needed to read it: a grader that refuses everything scores `rate` 1.0\n * on reject-plants, and a `rejectionRate` of 1.0 here is what separates that\n * reflex from discrimination.\n */\nexport interface UnseededRejection {\n n: number\n decided: number\n rejected: number\n /** rejected / decided, or null when nothing unseeded was decided. */\n rejectionRate: number | null\n}\n\nexport interface CatchRateReport {\n status: CatchRateStatus\n /** Why the status is not `evaluated`. Absent when it is. */\n reason?: string\n seeded: number\n caught: number\n missed: number\n /**\n * The grader returned a result and declined to decide. Counted apart: it\n * enters neither side of `rate`.\n */\n indecisive: number\n /** caught / (caught + missed), or null unless the status is `evaluated`. */\n rate: number | null\n /** One entry per kind actually seeded. A kind nobody seeded is absent, never zero. */\n byKind: Partial<Record<PlantKind, PlantKindCounts>>\n /** Plant ids the grader graded as the seed says it must not. */\n missedIds: string[]\n /** Plant ids with no result at all — the reason a status is `incomplete`. */\n missingIds: string[]\n unseeded: UnseededRejection\n}\n\n/**\n * Score a grading run against its sealed manifest.\n *\n * Refusals: a manifest whose contents no longer match its seal, a result for\n * an id the manifest never handed out, and a repeated result id. Each says\n * the results and the manifest describe different runs, and a rate computed\n * across two runs is a fabrication.\n */\nexport function catchRate(\n results: readonly PlantOutcome[],\n manifest: PlantManifest,\n): CatchRateReport {\n const expectedSeal = sealManifest({\n seed: manifest.seed,\n acceptThreshold: manifest.acceptThreshold,\n itemIds: manifest.itemIds,\n plants: manifest.plants,\n })\n if (expectedSeal !== manifest.seal) {\n throw new CaptureIntegrityError(\n `catchRate: manifest contents hash to ${expectedSeal} but the manifest carries seal ${manifest.seal} — the plant set changed after it was sealed`,\n )\n }\n\n const handedOut = new Set(manifest.itemIds)\n const scoreByItemId = new Map<string, number | null>()\n for (const result of results) {\n if (!handedOut.has(result.itemId)) {\n throw new ValidationError(\n `catchRate: result for \"${result.itemId}\", which this manifest never handed out`,\n )\n }\n if (scoreByItemId.has(result.itemId)) {\n throw new ValidationError(`catchRate: duplicate result for \"${result.itemId}\"`)\n }\n if (result.score !== null && !Number.isFinite(result.score)) {\n throw new ValidationError(\n `catchRate: result for \"${result.itemId}\" has score ${result.score}; expected a finite number or null`,\n )\n }\n scoreByItemId.set(result.itemId, result.score)\n }\n\n const byKind: Partial<Record<PlantKind, PlantKindCounts>> = {}\n const missedIds: string[] = []\n const missingIds: string[] = []\n let caught = 0\n let missed = 0\n let indecisive = 0\n\n for (const plant of manifest.plants) {\n let counts = byKind[plant.kind]\n if (counts === undefined) {\n counts = { seeded: 0, caught: 0, missed: 0, indecisive: 0, rate: null }\n byKind[plant.kind] = counts\n }\n counts.seeded += 1\n if (!scoreByItemId.has(plant.item.itemId)) {\n missingIds.push(plant.id)\n continue\n }\n const score = scoreByItemId.get(plant.item.itemId) ?? null\n if (score === null) {\n indecisive += 1\n counts.indecisive += 1\n continue\n }\n const graderSaid: PlantExpectation = score >= manifest.acceptThreshold ? 'accept' : 'reject'\n if (graderSaid === plant.expectedVerdict) {\n caught += 1\n counts.caught += 1\n } else {\n missed += 1\n counts.missed += 1\n missedIds.push(plant.id)\n }\n }\n\n const seeded = manifest.plants.length\n const decided = caught + missed\n const report: CatchRateReport = {\n status: 'evaluated',\n seeded,\n caught,\n missed,\n indecisive,\n rate: null,\n byKind,\n missedIds,\n missingIds,\n unseeded: unseededRejection(manifest, scoreByItemId),\n }\n\n if (seeded === 0) {\n return {\n ...report,\n status: 'not_evaluated',\n reason: 'no plant was seeded, so the grader was never asked a question it could fail',\n }\n }\n if (missingIds.length > 0) {\n return {\n ...report,\n status: 'incomplete',\n reason: `${missingIds.length} of ${seeded} seeded plants have no result: ${missingIds.join(', ')}`,\n }\n }\n if (decided === 0) {\n return {\n ...report,\n status: 'not_evaluated',\n reason: `all ${seeded} seeded plants are indecisive: no check tested the seeded defect`,\n }\n }\n\n for (const counts of Object.values(byKind)) {\n const kindDecided = counts.caught + counts.missed\n counts.rate = kindDecided === 0 ? null : counts.caught / kindDecided\n }\n report.rate = caught / decided\n return report\n}\n\nfunction unseededRejection(\n manifest: PlantManifest,\n scoreByItemId: ReadonlyMap<string, number | null>,\n): UnseededRejection {\n const plantItemIds = new Set(manifest.plants.map((plant) => plant.item.itemId))\n let n = 0\n let decided = 0\n let rejected = 0\n for (const itemId of manifest.itemIds) {\n if (plantItemIds.has(itemId)) continue\n n += 1\n const score = scoreByItemId.get(itemId)\n if (score === undefined || score === null) continue\n decided += 1\n if (score < manifest.acceptThreshold) rejected += 1\n }\n return { n, decided, rejected, rejectionRate: decided === 0 ? null : rejected / decided }\n}\n","/**\n * Judge sentinel — eval trustworthiness as a continuously measured,\n * alarmed trend.\n *\n * Judges are models; models change underneath us; calibration decays.\n * This module composes three existing instruments into one loop:\n *\n * snapshot — adapters turn real instrument outputs (`calibrateJudge` /\n * `calibrateJudgeContinuous`, `corpusInterRaterAgreement` /\n * `continuousAgreement`, a rerun sentinel golden set) into\n * `SentinelSnapshot`s\n * series — `analyzeSeries` (src/series-convergence) classifies each\n * judge×metric history as stabilized / drifting / noisy\n * alarm — `judgeSentinelReport` turns trends + thresholds into named\n * alarms; `evalHealthStamp` is the tiny object a campaign\n * attaches to its verdicts\n *\n * Alarm conditions:\n * - convergence state `drifting-down` on any tracked metric\n * - irr below `minIrr`\n * - calibrationKappa / sentinelPassRate dropped vs the series baseline\n * beyond `maxKappaDrop` / `maxSentinelDrop`\n * - judge model changed with no post-change golden-grounded snapshot\n * (the silent-upgrade trap: judges agreeing with each other after a\n * model swap proves nothing — only re-measuring against gold does)\n * - newest snapshot older than `staleAfterDays` relative to `asOf`\n *\n * Wire-in: run the snapshot adapters from a post-campaign hook or a\n * nightly job, append to a `SentinelStore`, then compute\n * `judgeSentinelReport(await store.history(), { asOf })` with `asOf` =\n * the campaign timestamp. Attach `evalHealthStamp(report)` to campaign\n * verdicts: an alarmed sentinel means every downstream verdict carries\n * an integrity warning until the judge is recalibrated.\n *\n * The module is clock-free — every timestamp (`at`, `asOf`) is\n * caller-supplied ISO-8601, so reports are reproducible.\n */\n\nimport { ValidationError } from '../errors'\nimport type {\n CalibrationResult,\n CandidateScore,\n ContinuousAgreement,\n ContinuousCalibrationResult,\n GoldenItem,\n} from '../judge-calibration'\nimport {\n analyzeSeries,\n type SeriesConvergenceOptions,\n type SeriesConvergenceResult,\n} from '../series-convergence'\nimport type { CorpusAgreementReport } from '../statistics'\n\nconst SENTINEL_METRIC_NAMES = ['irr', 'calibrationKappa', 'sentinelPassRate'] as const\n\nexport type SentinelMetricName = (typeof SENTINEL_METRIC_NAMES)[number]\n\nexport interface SentinelMetrics {\n /** Inter-rater reliability — ICC(2,1) from agreement instruments. */\n irr?: number\n /** Weighted κ vs the human golden set (`calibrateJudge*`). */\n calibrationKappa?: number\n /** Fraction of a fixed sentinel golden set the judge still scores within tolerance. */\n sentinelPassRate?: number\n}\n\nexport interface SentinelSnapshot {\n /** Caller-supplied ISO-8601 timestamp — the module never reads a clock. */\n at: string\n judgeId: string\n /** Exact model identity behind the judge (pin the full version string). */\n judgeModel: string\n /**\n * Metrics measured at `at`. May be empty: an empty-metrics snapshot is a\n * model-change marker — it records the new `judgeModel` immediately and\n * the silent-upgrade alarm stays raised until a golden-grounded snapshot\n * (calibrationKappa or sentinelPassRate) follows.\n */\n metrics: SentinelMetrics\n}\n\n/** Identity + timestamp the caller supplies alongside an instrument output. */\nexport interface SnapshotMeta {\n at: string\n judgeId: string\n judgeModel: string\n}\n\nfunction parseIso(value: string, label: string): number {\n const ms = typeof value === 'string' ? Date.parse(value) : NaN\n if (!Number.isFinite(ms)) {\n throw new ValidationError(\n `judge-sentinel: ${label} is not a parseable ISO timestamp: ${JSON.stringify(value)}`,\n )\n }\n return ms\n}\n\n/** Shape guard for snapshots crossing a boundary (store append, JSONL load, report input). */\nexport function validateSentinelSnapshot(snapshot: SentinelSnapshot, source = 'snapshot'): void {\n parseIso(snapshot.at, `${source}.at`)\n if (typeof snapshot.judgeId !== 'string' || snapshot.judgeId.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeId must be a non-empty string`)\n }\n if (typeof snapshot.judgeModel !== 'string' || snapshot.judgeModel.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeModel must be a non-empty string`)\n }\n if (snapshot.metrics === null || typeof snapshot.metrics !== 'object') {\n throw new ValidationError(`judge-sentinel: ${source}.metrics must be an object (may be empty)`)\n }\n for (const [key, value] of Object.entries(snapshot.metrics)) {\n if (!(SENTINEL_METRIC_NAMES as readonly string[]).includes(key)) {\n // A typo'd key would silently never trend — reject instead.\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics carries unknown key \"${key}\" — known: ${SENTINEL_METRIC_NAMES.join(', ')}`,\n )\n }\n if (value !== undefined && !Number.isFinite(value)) {\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics.${key} is not a finite number: ${String(value)}`,\n )\n }\n }\n}\n\n// ── Snapshot adapters — real instrument output → SentinelSnapshot ───\n\n/**\n * From `calibrateJudge` / `calibrateJudgeContinuous` output. Prefers the\n * un-rounded continuous κ_w when the report carries one (integer κ\n * discards information for [0,1] judges).\n */\nexport function snapshotFromCalibration(\n report: CalibrationResult | ContinuousCalibrationResult,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const kappa = 'weightedKappaContinuous' in report ? report.weightedKappaContinuous : report.kappa\n if (!Number.isFinite(kappa)) {\n throw new ValidationError(\n `snapshotFromCalibration: κ is not finite (n=${report.n}) — calibrate against ≥2 joined golden items before snapshotting`,\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { calibrationKappa: kappa } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n/**\n * From `corpusInterRaterAgreement` (statistics.ts) or `continuousAgreement`\n * (judge-calibration.ts) output. IRR = ICC(2,1) — the reliability\n * coefficient both instruments compute. For ensemble-level agreement,\n * `meta.judgeId` names the ensemble and `meta.judgeModel` pins its\n * composition (e.g. a joined list of member model strings).\n */\nexport function snapshotFromAgreement(\n report: CorpusAgreementReport | ContinuousAgreement,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const irr = 'overallIcc' in report ? report.overallIcc : report.icc\n if (!Number.isFinite(irr)) {\n throw new ValidationError(\n 'snapshotFromAgreement: ICC is not finite — agreement needs ≥2 raters and ≥2 complete items',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { irr } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\nexport interface SentinelSetOptions {\n /** Max |judge − human| for an item to count as passed. Default 0.1 (tuned for [0,1] scores). */\n tolerance?: number\n}\n\n/**\n * From a rerun of a fixed sentinel golden set: the same `GoldenItem`s the\n * judge was originally calibrated on, re-scored periodically. Pass rate =\n * fraction of joined items where the judge stays within `tolerance` of the\n * human score. Items the judge didn't score are excluded from the\n * denominator; zero joined items is an error, not a 0% pass rate.\n */\nexport function snapshotFromSentinelSet(\n scores: CandidateScore[],\n golden: GoldenItem[],\n meta: SnapshotMeta,\n options: SentinelSetOptions = {},\n): SentinelSnapshot {\n const tolerance = options.tolerance ?? 0.1\n if (!Number.isFinite(tolerance) || tolerance < 0) {\n throw new ValidationError(\n `snapshotFromSentinelSet: tolerance must be a finite number ≥ 0, got ${tolerance}`,\n )\n }\n const goldenById = new Map<string, number>()\n for (const item of golden) {\n if (!Number.isFinite(item.humanScore)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: golden item \"${item.itemId}\" has a non-finite humanScore`,\n )\n }\n if (goldenById.has(item.itemId)) {\n throw new ValidationError(`snapshotFromSentinelSet: duplicate golden itemId \"${item.itemId}\"`)\n }\n goldenById.set(item.itemId, item.humanScore)\n }\n let joined = 0\n let passed = 0\n for (const s of scores) {\n const human = goldenById.get(s.itemId)\n if (human === undefined) continue\n if (!Number.isFinite(s.score)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: judge score for item \"${s.itemId}\" is not finite`,\n )\n }\n joined++\n if (Math.abs(s.score - human) <= tolerance) passed++\n }\n if (joined === 0) {\n throw new ValidationError(\n 'snapshotFromSentinelSet: no judge scores joined the golden set — itemIds mismatch or empty sentinel run',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { sentinelPassRate: passed / joined } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n// ── Persistence ──────────────────────────────────────────────────────\n\nexport interface SentinelStore {\n append(snapshot: SentinelSnapshot): Promise<void>\n /** All snapshots (optionally for one judge), in append order. */\n history(judgeId?: string): Promise<SentinelSnapshot[]>\n}\n\nexport function inMemorySentinelStore(initial: SentinelSnapshot[] = []): SentinelStore {\n for (const s of initial) validateSentinelSnapshot(s)\n const items = initial.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n items.push({ ...snapshot, metrics: { ...snapshot.metrics } })\n },\n async history(judgeId) {\n const all = items.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return judgeId === undefined ? all : all.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n/**\n * JSONL store — one snapshot per line, appended atomically per call.\n * A corrupt or shape-invalid line is a loud error naming the file and\n * line number: a sentinel history that silently drops records would\n * defeat the drift detection it exists to provide.\n */\nexport function fileSentinelStore(path: string): SentinelStore {\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n const fs = await import('node:fs/promises')\n const pathMod = await import('node:path')\n await fs.mkdir(pathMod.dirname(path), { recursive: true })\n await fs.appendFile(path, `${JSON.stringify(snapshot)}\\n`, 'utf8')\n },\n async history(judgeId) {\n const fs = await import('node:fs/promises')\n let raw: string\n try {\n raw = await fs.readFile(path, 'utf8')\n } catch (err) {\n if ((err as NodeJS.ErrnoException).code === 'ENOENT') return []\n throw err\n }\n const lines = raw.split('\\n')\n const snapshots: SentinelSnapshot[] = []\n for (let i = 0; i < lines.length; i++) {\n const line = lines[i]!\n if (line.trim().length === 0) continue\n let parsed: unknown\n try {\n parsed = JSON.parse(line)\n } catch (err) {\n throw new ValidationError(\n `judge-sentinel: store at ${path} has a corrupt JSONL line ${i + 1}: ${(err as Error).message}`,\n )\n }\n const snapshot = parsed as SentinelSnapshot\n validateSentinelSnapshot(snapshot, `${path}:${i + 1}`)\n snapshots.push(snapshot)\n }\n return judgeId === undefined ? snapshots : snapshots.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n// ── Trend report ─────────────────────────────────────────────────────\n\nexport interface SentinelThresholds {\n /** Floor for irr — below it the judge ensemble is unreliable regardless of trend. Default 0.6. */\n minIrr?: number\n /** Max tolerated calibrationKappa drop vs the series baseline. Default 0.15. */\n maxKappaDrop?: number\n /** Max tolerated sentinelPassRate drop vs the series baseline. Default 0.1. */\n maxSentinelDrop?: number\n /** Newest snapshot older than this (days, vs asOf) = the sentinel itself went blind. Default 30. */\n staleAfterDays?: number\n}\n\nexport interface JudgeSentinelOptions {\n /** Caller-supplied ISO timestamp staleness is measured against. */\n asOf: string\n thresholds?: SentinelThresholds\n /** Forwarded to `analyzeSeries` (window, stableCv, driftRun). */\n convergence?: SeriesConvergenceOptions\n}\n\nexport interface SentinelTrend {\n judgeId: string\n metric: SentinelMetricName\n /** Verbatim `analyzeSeries` state for this judge×metric history. */\n state: SeriesConvergenceResult['state']\n /** Newest value in the series. */\n current: number\n /** Oldest value in the series — the reference the drop thresholds compare against. */\n baseline: number\n /** current − baseline (signed; negative = decayed). */\n drift: number\n alarmed: boolean\n reason?: string\n}\n\nexport interface SentinelReport {\n perJudge: SentinelTrend[]\n alarms: string[]\n healthy: boolean\n /**\n * `judgeId:metric` pairs with too few snapshots for the convergence\n * machine. Named, not counted — a blind spot you can't see is a blind\n * spot you won't fix. Floor/drop checks still apply to thin series, so\n * insufficient history alone never masks an alarm.\n */\n insufficientHistory: string[]\n}\n\nconst MS_PER_DAY = 86_400_000\n\nexport function judgeSentinelReport(\n history: SentinelSnapshot[],\n opts: JudgeSentinelOptions,\n): SentinelReport {\n const asOfMs = parseIso(opts.asOf, 'asOf')\n const minIrr = opts.thresholds?.minIrr ?? 0.6\n const maxKappaDrop = opts.thresholds?.maxKappaDrop ?? 0.15\n const maxSentinelDrop = opts.thresholds?.maxSentinelDrop ?? 0.1\n const staleAfterDays = opts.thresholds?.staleAfterDays ?? 30\n\n const byJudge = new Map<string, Array<SentinelSnapshot & { atMs: number }>>()\n for (const snapshot of history) {\n validateSentinelSnapshot(snapshot)\n const atMs = parseIso(snapshot.at, `snapshot.at (judge \"${snapshot.judgeId}\")`)\n const arr = byJudge.get(snapshot.judgeId) ?? []\n arr.push({ ...snapshot, atMs })\n byJudge.set(snapshot.judgeId, arr)\n }\n\n const perJudge: SentinelTrend[] = []\n const alarms: string[] = []\n const insufficientHistory: string[] = []\n\n for (const judgeId of [...byJudge.keys()].sort()) {\n const snaps = byJudge.get(judgeId)!.sort((a, b) => a.atMs - b.atMs)\n const newest = snaps[snaps.length - 1]!\n\n const ageDays = (asOfMs - newest.atMs) / MS_PER_DAY\n if (ageDays > staleAfterDays) {\n alarms.push(\n `judge \"${judgeId}\": stale — newest snapshot ${newest.at} is ${ageDays.toFixed(1)}d old vs asOf (limit ${staleAfterDays}d)`,\n )\n }\n\n // Silent-upgrade trap: only the latest model change matters for current\n // trust, and only golden-grounded metrics clear it — irr alone proves\n // judges still agree with each other, not that they're still right.\n let changeIdx = -1\n for (let i = 1; i < snaps.length; i++) {\n if (snaps[i]!.judgeModel !== snaps[i - 1]!.judgeModel) changeIdx = i\n }\n if (changeIdx >= 0) {\n const recalibrated = snaps\n .slice(changeIdx)\n .some(\n (s) =>\n Number.isFinite(s.metrics.calibrationKappa) ||\n Number.isFinite(s.metrics.sentinelPassRate),\n )\n if (!recalibrated) {\n alarms.push(\n `judge \"${judgeId}\": model changed \"${snaps[changeIdx - 1]!.judgeModel}\" → \"${snaps[changeIdx]!.judgeModel}\" at ${snaps[changeIdx]!.at} with no post-change calibration snapshot — silent-upgrade trap; rerun calibrateJudge or the sentinel set against the new model`,\n )\n }\n }\n\n for (const metric of SENTINEL_METRIC_NAMES) {\n const values = snaps\n .filter((s) => typeof s.metrics[metric] === 'number')\n .map((s) => s.metrics[metric]!)\n if (values.length === 0) continue\n\n const convergence = analyzeSeries(values, opts.convergence)\n if (convergence.state === 'insufficient-data')\n insufficientHistory.push(`${judgeId}:${metric}`)\n\n const current = values[values.length - 1]!\n const baseline = values[0]!\n const drift = current - baseline\n const reasons: string[] = []\n if (convergence.state === 'drifting-down') {\n reasons.push(\n `convergence state drifting-down (tail run ${convergence.tailRun}, window mean ${convergence.windowMean.toFixed(3)})`,\n )\n }\n if (metric === 'irr' && current < minIrr) {\n reasons.push(`irr ${current.toFixed(3)} below floor ${minIrr}`)\n }\n if (metric === 'calibrationKappa' && baseline - current > maxKappaDrop) {\n reasons.push(\n `calibrationKappa dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxKappaDrop})`,\n )\n }\n if (metric === 'sentinelPassRate' && baseline - current > maxSentinelDrop) {\n reasons.push(\n `sentinelPassRate dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxSentinelDrop})`,\n )\n }\n\n const alarmed = reasons.length > 0\n const trend: SentinelTrend = {\n judgeId,\n metric,\n state: convergence.state,\n current,\n baseline,\n drift,\n alarmed,\n }\n if (alarmed) {\n trend.reason = reasons.join('; ')\n alarms.push(`judge \"${judgeId}\" ${metric}: ${trend.reason}`)\n }\n perJudge.push(trend)\n }\n }\n\n return { perJudge, alarms, healthy: alarms.length === 0, insufficientHistory }\n}\n\n// ── Verdict stamp ────────────────────────────────────────────────────\n\nexport interface EvalHealthStamp {\n healthy: boolean\n alarms: string[]\n}\n\n/**\n * The object a campaign attaches to its verdicts. Compute it from the\n * sentinel report in a post-campaign hook (or a nightly job feeding the\n * next day's campaigns) and store it alongside the verdict payload.\n * `healthy: false` means the judges that produced those verdicts have an\n * unresolved drift / decay / silent-upgrade / staleness alarm — treat the\n * verdicts as carrying an integrity warning until recalibration clears it.\n */\nexport function evalHealthStamp(report: SentinelReport): EvalHealthStamp {\n return { healthy: report.healthy, alarms: [...report.alarms] }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;AAkDA,eAAsB,iBACpB,YACA,cACA,YACA,eACA,UAA8B,CAAC,GACI;CACnC,MAAM,WAAW;EACf,GAAG;EACH,OAAO,QAAQ,UAAU,KAAA,IAAY,KAAA,IAAY,EAAE,GAAG,QAAQ,MAAM;CACtE;CACA,2BAA2B,WAAW,IAAI,eAAe,QAAQ;CACjE,MAAM,UAAU,WAAW,WAAW,mBAAmB,WAAW,EAAE;CACtE,MAAM,WAAW,WAAW;CAC5B,MAAM,OAAO,MAAM,WAAW,SAAS;CACvC,2BACE,KAAK,KAAK,QAAQ,IAAI,KAAK,GAC3B,OACF;CACA,MAAM,WAAW,MAAM,aAAa,KAAK;CACzC,MAAM,wBAAQ,IAAI,IAAiC;CACnD,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,MAAM,MAAM,IAAI,EAAE,KAAK,KAAK,CAAC;EACnC,IAAI,KAAK,CAAC;EACV,MAAM,IAAI,EAAE,OAAO,GAAG;CACxB;CAEA,MAAM,QAAyC,CAAC;CAChD,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,KAAK,MAAM,IAAI,IAAI,KAAK;EAC9B,IAAI,CAAC,IAAI,QAAQ;EACjB,MAAM,IAAI,MAAM,QAAQ,KAAK,UAAU;EACvC,IAAI,MAAM,QAAQ,CAAC,OAAO,SAAS,CAAC,GAAG;EACvC,MAAM,IAAI,oBAAoB,IAAI,eAAe,QAAQ;EACzD,IAAI,MAAM,MAAM;EAChB,MAAM,KAAK;GAAE;GAAG;EAAE,CAAC;CACrB;CACA,IAAI,MAAM,SAAS,GAAG,OAAO;CAE7B,OAAO,qBACL,MAAM,KAAK,OAAO;EAAE,WAAW,EAAE;EAAG,SAAS,EAAE;CAAE,EAAE,GACnD,UACA,eACA,QACF;AACF;;AAGA,SAAgB,qBACd,YACA,YACA,eACA,UAA8B,CAAC,GACL;CAC1B,2BAA2B,YAAY,eAAe,OAAO;CAC7D,KAAK,MAAM,CAAC,OAAO,SAAS,WAAW,QAAQ,GAC7C,IACE,SAAS,QACT,OAAO,SAAS,YAChB,CAAC,OAAO,SAAS,KAAK,SAAS,KAC/B,CAAC,OAAO,SAAS,KAAK,OAAO,GAE7B,MAAM,IAAI,MAAM,oBAAoB,MAAM,kDAAkD;CAGhG,MAAM,QAAQ;CACd,IAAI,MAAM,SAAS,GAAG,OAAO;CAE7B,MAAM,UAAU,QAAQ,QAAQ;CAChC,MAAM,UAAU,QAAQ,WAAW;CACnC,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,SAAS;CACvC,MAAM,KAAK,QAAQ,OAAO,MAAM,KAAK,IAAI,GAAG,EAAE;CAC9C,MAAM,KAAK,QAAQ,OAAO,MAAM,KAAK,IAAI,GAAG,EAAE;CAC9C,MAAM,OAAO,KAAK;CAClB,IAAI,CAAC,OAAO,SAAS,IAAI,GAAG,MAAM,IAAI,MAAM,uCAAuC;CACnF,MAAM,UAAU,MAAM,KAAK,UAAU;EACnC,GAAG;EACH,WAAW,KAAK,IAAI,IAAI,KAAK,IAAI,IAAI,KAAK,SAAS,CAAC;CACtD,EAAE;CAEF,MAAM,OAAyB,CAAC;CAChC,IAAI,SAAS,KAAK,QAAQ,OAAO,SAAS,KAAK,cAAc,QAAQ,EAAE,CAAE,SAAS,GAChF,KAAK,KAAK,MAAM,OAAO,CAAC;MACnB,IAAI,YAAY,mBAAmB;EACxC,MAAM,SAAS,CAAC,GAAG,OAAO,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;EACpE,MAAM,QAAQ,KAAK,IAAI,SAAS,OAAO,MAAM;EAC7C,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,KAAK;GAC9B,MAAM,QAAQ,KAAK,MAAO,IAAI,OAAO,SAAU,KAAK;GACpD,MAAM,MAAM,KAAK,OAAQ,IAAI,KAAK,OAAO,SAAU,KAAK;GACxD,KAAK,KAAK,MAAM,OAAO,MAAM,OAAO,GAAG,CAAC,CAAC;EAC3C;CACF,OAAO;EACL,MAAM,yBAAS,IAAI,IAA+B;EAClD,KAAK,MAAM,QAAQ,SAAS;GAC1B,MAAM,QAAQ,KAAK,IAAI,UAAU,GAAG,KAAK,OAAQ,KAAK,YAAY,MAAM,OAAQ,OAAO,CAAC;GACxF,MAAM,QAAQ,OAAO,IAAI,KAAK,KAAK,CAAC;GACpC,MAAM,KAAK,IAAI;GACf,OAAO,IAAI,OAAO,KAAK;EACzB;EACA,KAAK,MAAM,CAAC,OAAO,UAAU,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,CAAC,IAAI,CAAC,OAAO,IAAI,CAAC,GAC/D,KAAK,KAAK,MAAM,OAAO,KAAK,QAAQ,QAAQ,UAAU,KAAK,SAAS,QAAQ,KAAK,QAAQ,CAAC;CAE9F;CAEA,MAAM,QAAQ,KAAK,QAAQ,GAAG,MAAM,IAAI,EAAE,GAAG,CAAC;CAC9C,MAAM,MAAM,KAAK,QAAQ,GAAG,MAAM,IAAK,EAAE,IAAI,QAAS,EAAE,KAAK,CAAC;CAC9D,MAAM,SAAS,KAAK,QAAQ,GAAG,MAAM,KAAK,IAAI,GAAG,EAAE,GAAG,GAAG,CAAC;CAE1D,OAAO;EAAE;EAAY;EAAe,GAAG,MAAM;EAAQ;EAAM;EAAK;CAAO;AACzE;AAEA,SAAS,MAAM,OAA0B,OAAgB,OAAgC;CACvF,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,SAAS;CACvC,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,OAAO;CACrC,MAAM,WAAW,KAAK,EAAE;CACxB,MAAM,cAAc,KAAK,EAAE;CAC3B,OAAO;EACL,OAAO,SAAS,KAAK,IAAI,GAAG,EAAE;EAC9B,OAAO,SAAS,KAAK,IAAI,GAAG,EAAE;EAC9B,GAAG,MAAM;EACT;EACA;EACA,KAAK,KAAK,IAAI,cAAc,QAAQ;CACtC;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,OAAO,GAAG,QAAQ,KAAK,UAAU,MAAM,QAAQ,GAAG,QAAQ,CAAC;AAC7D;AAEA,SAAS,2BACP,YACA,eACA,SACM;CACN,2BAA2B,CAAC,UAAU,GAAG,aAAa;CACtD,2BAA2B,CAAC,aAAa,GAAG,gBAAgB;CAC5D,IAAI,WAAW,KAAK,MAAM,cAAc,cAAc,KAAK,MAAM,eAC/D,MAAM,IAAI,MAAM,oEAAoE;CAEtF,IAAI,QAAQ,SAAS,KAAA,MAAc,CAAC,OAAO,cAAc,QAAQ,IAAI,KAAK,QAAQ,OAAO,IACvF,MAAM,IAAI,MAAM,kDAAkD;CAEpE,IACE,QAAQ,YAAY,KAAA,KACpB,CAAC,CAAC,eAAe,iBAAiB,CAAC,CAAC,SAAS,QAAQ,OAAO,GAE5D,MAAM,IAAI,MAAM,4DAA4D;CAE9E,IACE,QAAQ,UAAU,KAAA,MACjB,CAAC,OAAO,SAAS,QAAQ,MAAM,EAAE,KAChC,CAAC,OAAO,SAAS,QAAQ,MAAM,EAAE,KACjC,CAAC,OAAO,SAAS,QAAQ,MAAM,KAAK,QAAQ,MAAM,EAAE,KACpD,QAAQ,MAAM,KAAK,QAAQ,MAAM,KAEnC,MAAM,IAAI,MAAM,mDAAmD;AAEvE;;;;;;;;;;ACvIA,eAAsB,iBACpB,YACA,cACA,aACA,oBACA,UAAmC,CAAC,GACH;CACjC,MAAM,YAAY,QAAQ,aAAa;CACvC,MAAM,aAAa,QAAQ,uBAAuB;CAClD,MAAM,OAAO,QAAQ;CACrB,2BAA2B,WAAW,YAAY,IAAI;CACtD,2BACE,YAAY,KAAK,WAAW,OAAO,EAAE,GACrC,aACF;CACA,2BAA2B,oBAAoB,gBAAgB;CAC/D,MAAM,SAAS,QAAQ,mBAAmB;CAC1C,IAAI,SAAS,KAAK,OAAO,MAAM,MAAM,GACnC,MAAM,IAAI,MAAM,qCAAqC;CAEvD,MAAM,aAAa,YAAY,KAAK,YAAY;EAC9C,GAAG;EACH,SAAS,OAAO,WAAW,mBAAmB,OAAO,EAAE;CACzD,EAAE;CACF,MAAM,cAAc,CAAC,GAAG,kBAAkB;CAC1C,MAAM,OAAO,MAAM,WAAW,SAAS;CACvC,2BACE,KAAK,KAAK,QAAQ,IAAI,KAAK,GAC3B,OACF;CACA,MAAM,WAAW,MAAM,aAAa,KAAK,QAAQ,aAAa;CAC9D,MAAM,gCAAgB,IAAI,IAAiC;CAC3D,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,MAAM,cAAc,IAAI,EAAE,KAAK,KAAK,CAAC;EAC3C,IAAI,KAAK,CAAC;EACV,cAAc,IAAI,EAAE,OAAO,GAAG;CAChC;CAEA,MAAM,QAA0F,CAAC;CACjG,KAAK,MAAM,MAAM,YACf,KAAK,MAAM,MAAM,aACf,MAAM,KAAK;EAAE,YAAY,GAAG;EAAI,eAAe;EAAI,IAAI,CAAC;EAAG,IAAI,CAAC;CAAE,CAAC;CAIvE,IAAI,SAAS;CACb,IAAI,UAAU;CACd,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,KAAK,cAAc,IAAI,IAAI,KAAK;EACtC,IAAI,CAAC,MAAM,GAAG,WAAW,GAAG;GAC1B;GACA;EACF;EACA,MAAM,WAAW,GAAG,QAAQ,MAAM;GAChC,MAAM,MAAM,EAAE,aAAa,IAAI;GAC/B,OAAO,OAAO,KAAK,OAAO;EAC5B,CAAC;EACD,IAAI,SAAS,WAAW,GAAG;GACzB;GACA;EACF;EAEA,IAAI,gBAAgB;EACpB,KAAK,MAAM,MAAM,YAAY;GAC3B,MAAM,IAAI,MAAM,GAAG,QAAQ,KAAK,UAAU;GAC1C,IAAI,MAAM,QAAQ,CAAC,OAAO,SAAS,CAAC,GAAG;GAEvC,KAAK,MAAM,MAAM,aAAa;IAC5B,MAAM,IAAI,oBAAoB,UAAU,IAAI,SAAS;IACrD,IAAI,MAAM,MAAM;IAChB,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,eAAe,GAAG,MAAM,EAAE,kBAAkB,EAAE;IAC/E,KAAK,GAAG,KAAK,CAAC;IACd,KAAK,GAAG,KAAK,CAAC;IACd,gBAAgB;GAClB;EACF;EACA,IAAI,eAAe;OACd;CACP;CAEA,MAAM,gBAAyD,CAAC;CAChE,MAAM,UAA+B,CAAC;CACtC,KAAK,MAAM,KAAK,OAAO;EACrB,MAAM,SACJ,EAAE,GAAG,SAAS,IACV,yBACA,CAAC,aAAa,EAAE,EAAE,IAChB,yBACA,CAAC,aAAa,EAAE,EAAE,IAChB,qBACA;EACV,IAAI,WAAW,MAAM;GACnB,cAAc,KAAK;IACjB,YAAY,EAAE;IACd,eAAe,EAAE;IACjB,GAAG,EAAE,GAAG;IACR;GACF,CAAC;GACD;EACF;EACA,MAAM,UAAU,mBAAmB,EAAE,IAAI,EAAE,IAAI,YAAY,IAAI;EAC/D,MAAM,UACJ,KAAK,IAAI,QAAQ,OAAO,KAAK,KACzB,WACA,KAAK,IAAI,QAAQ,OAAO,KAAK,KAC3B,aACA;EACR,QAAQ,KAAK;GACX,YAAY,EAAE;GACd,eAAe,EAAE;GACjB,GAAG,EAAE,GAAG;GACR,GAAG;GACH;EACF,CAAC;CACH;CACA,OAAO;EAAE,OAAO;EAAS;EAAe,eAAe;EAAQ,aAAa;CAAQ;AACtF;;;ACxLA,MAAM,WAAW,EACd,OAAO,CAAC,CACR,IAAI,CAAC,CAAC,CACN,QAAQ,UAAU,MAAM,KAAK,MAAM,KAAK;AAC3C,MAAM,eAAe,EAClB,OAAO;CACN,YAAY,EAAE,OAAO,CAAC,CAAC,OAAO,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,GAAG,CAAC;CAC1C,wBAAwB,EAAE,OAAO,CAAC,CAAC,OAAO,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,GAAG,CAAC;CACvD,uBAAuB,EAAE,OAAO,CAAC,CAAC,OAAO,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,GAAG,CAAC;AACxD,CAAC,CAAC,CACD,OAAO;AAKV,MAAM,oBAAoB,EACvB,OAAO;CACN,IAAI;CACJ,mBAAmB;CACnB,aAAa;CACb,UAAU,EAAE,KAAK,CAAC,UAAU,QAAQ,CAAC;CACrC,UAAU,EAAE,KAAK;EAAC;EAAU;EAAU;CAAS,CAAC;CAChD,UAAU,EAAE,KAAK,CAAC,SAAS,aAAa,CAAC;AAC3C,CAAC,CAAC,CACD,OAAO;AAKV,MAAM,cAAc,EACjB,OAAO;CACN,iBAAiB,EAAE,OAAO,CAAC,CAAC,MAAM,uBAAuB;CACzD,YAAY;CACZ,eAAe;CACf,WAAW,EACR,OAAO;EACN,mBAAmB;EACnB,WAAW;EACX,yBAAyB;CAC3B,CAAC,CAAC,CACD,OAAO;CACV,QAAQ;CACR,cAAc,EAAE,MAAM,iBAAiB;AACzC,CAAC,CAAC,CACD,OAAO;AAiDV,SAAS,UACP,MACA,UACA,OACA,OACoB;CACpB,MAAM,QAAQ,KAAK,QAAQ,QAAQ,IAAI,aAAa,QAAQ;CAC5D,MAAM,wBAAQ,IAAI,IAAkD;CACpE,KAAK,MAAM,OAAO,OAAO;EACvB,MAAM,OAAO,MAAM,IAAI,IAAI,iBAAiB,KAAK;GAAE,OAAO;GAAO,SAAS;EAAM;EAChF,KAAK,YAAY,IAAI,aAAa;EAClC,KAAK,UAAU,IAAI,aAAa,aAAa,IAAI,aAAa;EAC9D,MAAM,IAAI,IAAI,mBAAmB,IAAI;CACvC;CACA,MAAM,IAAI,MAAM;CAChB,MAAM,aAAa,CAAC,GAAG,MAAM,OAAO,CAAC,CAAC,CAAC,QAAQ,SAAS,KAAK,KAAK,CAAC,CAAC;CACpE,MAAM,kBAAkB,CAAC,GAAG,MAAM,OAAO,CAAC,CAAC,CAAC,QAAQ,SAAS,KAAK,WAAW,CAAC,KAAK,KAAK,CAAC,CAAC;CAC1F,MAAM,eAAe,MAAM,QAAQ,QAAQ,IAAI,aAAa,SAAS,CAAC,CAAC;CAEvE,MAAM,WACJ,MAAM,IACF,OACA;EACE,OAAO,gBACL;GAAE,MAAM;GAAmB;EAAM,GACjC;GACE,MAAM;GACN,WAAW;GACX,QAAQ;EACV,CACF,CAAC,CAAC;EACF,OAAO,gBACL;GAAE,MAAM;GAAmB;EAAM,GACjC;GACE,MAAM;GACN,WAAW,aAAa;GACxB,QAAQ;EACV,CACF,CAAC,CAAC;EACF;CACF;CACN,MAAM,UACJ,YAAY,SAAS,QAAQ,QACzB,SACA,YAAY,SAAS,SAAS,QAC5B,SACA;CACR,OAAO;EACL,OAAO,MAAM;EACb,kBAAkB;EAClB;EACA;EACA;EACA,WAAW,MAAM,KAAK,kBAAkB,IAAI,OAAO,aAAa;EAChE;EACA;EACA;CACF;AACF;;;;;;AAOA,SAAgB,eAAe,OAAsD;CACnF,MAAM,SAAS,YAAY,UAAU,KAAK;CAC1C,IAAI,CAAC,OAAO,SAAS,MAAM,IAAI,gBAAgB,4BAA4B,OAAO,MAAM,SAAS;CACjG,MAAM,QAAQ,OAAO;CACrB,IAAI,MAAM,UAAU,sBAAsB,MAAM,UAAU,WACxD,MAAM,IAAI,gBAAgB,kEAAkE;CAE9F,MAAM,sBAAM,IAAI,IAAY;CAC5B,KAAK,MAAM,OAAO,MAAM,cAAc;EACpC,IAAI,IAAI,IAAI,IAAI,EAAE,GAChB,MAAM,IAAI,gBAAgB,0CAA0C,IAAI,GAAG,EAAE;EAC/E,IAAI,IAAI,IAAI,EAAE;CAChB;CACA,MAAM,aAAa,MAAM,GAAG,MAAM,iBAAiB,EAAE,IAAI,EAAE,EAAE,CAAC;CAC9D,MAAM,mBAAmB,IAAI,IAC3B,MAAM,aACH,QAAQ,QAAQ,IAAI,aAAa,aAAa,CAAC,CAC/C,KAAK,QAAQ,IAAI,iBAAiB,CACvC;CACA,MAAM,WAAW,MAAM,aAAa,QAAQ,QAAQ,CAAC,iBAAiB,IAAI,IAAI,iBAAiB,CAAC;CAChG,MAAM,aAAa,MAAM,aACtB,QAAQ,QAAQ,iBAAiB,IAAI,IAAI,iBAAiB,CAAC,CAAC,CAC5D,KAAK,SAAS;EACb,IAAI,IAAI;EACR,mBAAmB,IAAI;EACvB,QAAQ;CACV,EAAE;CACJ,MAAM,qBAAqB,KAAK,IAAI,MAAM,OAAO,cAAc;CAC/D,IAAI,sBAAsB,GACxB,MAAM,IAAI,gBACR,2EACF;CACF,MAAM,kBAAkB,UACtB,UACA,UACA,MAAM,OAAO,wBACb,kBACF;CACA,MAAM,iBAAiB,UACrB,UACA,UACA,MAAM,OAAO,uBACb,kBACF;CACA,MAAM,QAAQ,CAAC,iBAAiB,cAAc;CAC9C,MAAM,UAAU,MAAM,MAAM,SAAS,KAAK,YAAY,MAAM,IACxD,WACA,MAAM,OAAO,SAAS,KAAK,YAAY,MAAM,IAC3C,UACA;CACN,MAAM,UAAoB,CAAC;CAC3B,KAAK,MAAM,CAAC,MAAM,SAAS,CACzB,CAAC,oBAAoB,eAAe,GACpC,CAAC,mBAAmB,cAAc,CACpC,GAAY;EACV,IAAI,KAAK,qBAAqB,GAAG,QAAQ,KAAK,GAAG,KAAK,gCAAgC;OACjF,IAAI,KAAK,eAAe,GAC3B,QAAQ,KAAK,GAAG,KAAK,IAAI,KAAK,aAAa,uBAAuB;EACpE,IAAI,KAAK,YAAY,QAAQ,QAAQ,KAAK,GAAG,KAAK,8BAA8B,KAAK,OAAO;OACvF,IAAI,KAAK,YAAY,kBAAkB,KAAK,UAC/C,QAAQ,KAAK,GAAG,KAAK,0DAA0D;CACnF;CACA,IAAI,YAAY,SACd,QAAQ,KACN,8FACF;CACF,MAAM,OAAuD;EAC3D,iBAAiB,MAAM;EACvB,cAAc,cAAc,MAAM,MAAM;EACxC,aAAa,cAAc,KAAK;EAChC,YAAY,MAAM;EAClB,eAAe,MAAM;EACrB,WAAW,MAAM;EACjB,QAAQ,MAAM;EACd,YAAY,MAAM,OAAO;EACzB;EACA;EACA;EACA,cAAc,MAAM;EACpB,UAAU;GACR,OAAO,MAAM,aAAa;GAC1B,kBAAkB,IAAI,IAAI,MAAM,aAAa,KAAK,QAAQ,IAAI,iBAAiB,CAAC,CAAC,CAAC;GAClF,eAAe,SAAS;GACxB,0BAA0B,IAAI,IAAI,SAAS,KAAK,QAAQ,IAAI,iBAAiB,CAAC,CAAC,CAAC;GAChF,eAAe,WAAW;GAC1B,cAAc,SAAS,QAAQ,QAAQ,IAAI,aAAa,SAAS,CAAC,CAAC;EACrE;EACA;EACA;EACA;CACF;CACA,OAAO;EAAE,GAAG;EAAM,cAAc,cAAc,IAAI;CAAE;AACtD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACvLA,MAAM,cAAoC;CACxC;CACA;CACA;CACA;AACF;AAKA,MAAM,qBAAkD,CAAC,UAAU,QAAQ;;AAG3E,MAAM,wBAAwB;;;;;;;;;;;;;AAwB9B,SAAgB,YAAY,OAKlB;CACR,MAAM,EAAE,IAAI,MAAM,MAAM,oBAAoB;CAC5C,IAAI,OAAO,OAAO,YAAY,GAAG,KAAK,MAAM,IAC1C,MAAM,IAAI,gBAAgB,4CAA4C;CAExE,IAAI,CAAC,YAAY,SAAS,IAAI,GAC5B,MAAM,IAAI,gBACR,uBAAuB,GAAG,aAAa,KAAK,UAAU,IAAI,EAAE,oBAAoB,YAAY,KAAK,IAAI,GACvG;CAEF,IAAI,CAAC,mBAAmB,SAAS,eAAe,GAC9C,MAAM,IAAI,gBACR,uBAAuB,GAAG,wBAAwB,KAAK,UAAU,eAAe,EAAE,oBAAoB,mBAAmB,KAAK,IAAI,GACpI;CAEF,IAAI,OAAO,KAAK,WAAW,YAAY,KAAK,OAAO,KAAK,MAAM,IAC5D,MAAM,IAAI,gBAAgB,uBAAuB,GAAG,2BAA2B;CAEjF,IAAI,CAAC,OAAO,SAAS,KAAK,UAAU,KAAK,KAAK,aAAa,KAAK,KAAK,aAAa,GAChF,MAAM,IAAI,gBACR,uBAAuB,GAAG,mBAAmB,KAAK,WAAW,qCAC/D;CAEF,IAAI,KAAK,UAAU,KAAA,KAAa,OAAO,KAAK,UAAU,UACpD,MAAM,IAAI,gBAAgB,uBAAuB,GAAG,8BAA8B;CAEpF,IAAI,KAAK,eAAe,uBACtB,MAAM,IAAI,gBACR,uBAAuB,GAAG,mBAAmB,sBAAsB,qDACrE;CAEF,MAAM,YAA8B,KAAK,aAAa,wBAAwB,WAAW;CACzF,IAAI,cAAc,iBAChB,MAAM,IAAI,gBACR,uBAAuB,GAAG,0BAA0B,gBAAgB,sBAAsB,KAAK,WAAW,QAAQ,WACpH;CAEF,OAAO;EACL;EACA;EACA,MAAM;GACJ,QAAQ,KAAK;GACb,YAAY,KAAK;GACjB,GAAI,KAAK,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,KAAK,MAAM;EAC1D;EACA;CACF;AACF;;;;;;;;;;;AA8CA,MAAM,cAAc;;;;;;;;AASpB,MAAM,mBAA2D;CAC/D,CAAC,SAAS,OAAO;CACjB,CAAC,SAAS,OAAO;CACjB,CAAC,SAAS,OAAO;CACjB,CAAC,SAAS,OAAO;CACjB,CAAC,SAAS,OAAO;CACjB,CAAC,SAAS,OAAO;CACjB,CAAC,MAAM,GAAG;CACV,CAAC,MAAM,GAAG;CACV,CAAC,MAAM,IAAI;CACX,CAAC,MAAM,IAAI;AACb;;;;;;;;;AAeA,SAAS,qBAAqB,MAA2B;CACvD,MAAM,SAAS;CACf,IAAI,QAAqB;CACzB,IAAI,QAAQ,OAAO,KAAK,IAAI;CAC5B,OAAO,UAAU,MAAM;EACrB,MAAM,QAAQ,MAAM;EACpB,MAAM,MAAM,QAAQ,MAAM,EAAE,CAAC;EAE7B,IAAI,CAAC,YAAY,KAAK,KAAK,OAAO,QAAQ,CAAC,CAAC,KAAK,CAAC,YAAY,KAAK,KAAK,OAAO,GAAG,CAAC,GACjF,QAAQ;GAAE;GAAO;EAAI;EAEvB,QAAQ,OAAO,KAAK,IAAI;CAC1B;CACA,OAAO;AACT;;;;;;;;;AAUA,SAAS,gBAAgB,MAAsD;CAC7E,KAAK,IAAI,QAAQ,GAAG,QAAQ,KAAK,QAAQ,SAAS,GAChD,KAAK,MAAM,CAAC,OAAO,YAAY,kBAC7B,IAAI,KAAK,WAAW,OAAO,KAAK,GAC9B,OAAO;EAAE,MAAM;GAAE,OAAO;GAAO,KAAK,QAAQ,MAAM;EAAO;EAAG;CAAQ;CAI1E,OAAO;AACT;AAEA,SAAS,YAAY,MAAc,MAAY,OAAuB;CACpE,OAAO,KAAK,MAAM,GAAG,KAAK,KAAK,IAAI,QAAQ,KAAK,MAAM,KAAK,GAAG;AAChE;;;;;AAMA,SAAS,UACP,OACA,QACA,OACA,OACe;CACf,MAAM,YAAY,UAAU,UAAU,QAAQ;CAC9C,MAAM,aAAa,UAAU,WAAW,QAAQ;CAChD,OAAO;EACL,GAAI,cAAc,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,UAAU;EACtD,GAAI,eAAe,KAAA,IAAY,CAAC,IAAI,EAAE,QAAQ,WAAW;CAC3D;AACF;AAEA,SAAS,SACP,OACA,QACA,OACA,MACA,MACsB;CACtB,MAAM,WAAW,KAAK,MAAM,KAAK,OAAO,KAAK,GAAG;CAChD,MAAM,aAAa,OAAO,QAAQ,IAAI,GAAA,CAAI,SAAS;CACnD,OAAO;EACL,UAAU,UAAU,OAAO,QAAQ,OAAO,YAAY,MAAM,MAAM,SAAS,CAAC;EAC5E;EACA,KAAK;EACL;EACA;CACF;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA6BA,SAAgB,gBAAgB,UAAsD;CACpF,MAAM,QAAQ,OAAO,SAAS,UAAU,WAAW,SAAS,QAAQ,KAAA;CACpE,MAAM,SAAS,OAAO,SAAS,WAAW,WAAW,SAAS,SAAS,KAAA;CAEvE,IAAI,WAAW,KAAA,GAAW;EACxB,MAAM,OAAO,qBAAqB,MAAM;EACxC,IAAI,SAAS,MAAM,OAAO,SAAS,OAAO,QAAQ,UAAU,QAAQ,IAAI;CAC1E;CACA,IAAI,UAAU,KAAA,GAAW;EACvB,MAAM,aAAa,gBAAgB,KAAK;EACxC,IAAI,eAAe,MACjB,OAAO;GACL,UAAU,UACR,OACA,QACA,SACA,YAAY,OAAO,WAAW,MAAM,WAAW,OAAO,CACxD;GACA,OAAO;GACP,KAAK;GACL,UAAU,MAAM,MAAM,WAAW,KAAK,OAAO,WAAW,KAAK,GAAG;GAChE,WAAW,WAAW;EACxB;EAEF,MAAM,OAAO,qBAAqB,KAAK;EACvC,IAAI,SAAS,MAAM,OAAO,SAAS,OAAO,QAAQ,SAAS,OAAO,IAAI;CACxE;CACA,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;AAgCA,SAAgB,oBAAoB,OAOV;CACxB,MAAM,EAAE,IAAI,QAAQ,UAAU,aAAa,GAAG,UAAU;CACxD,IAAI,OAAO,OAAO,YAAY,GAAG,KAAK,MAAM,IAC1C,MAAM,IAAI,gBAAgB,oDAAoD;CAEhF,IAAI,OAAO,WAAW,YAAY,OAAO,KAAK,MAAM,IAClD,MAAM,IAAI,gBAAgB,+BAA+B,GAAG,sBAAsB;CAEpF,MAAM,eAAe,gBAAgB,QAAQ;CAC7C,IAAI,iBAAiB,MAAM,OAAO;CAOlC,OAAO;EACL,OAPY,YAAY;GACxB;GACA,MAAM;GACN,MAAM;IAAE;IAAQ;IAAY,GAAI,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM;GAAG;GACtE,iBAAiB;EACnB,CAEM;EACJ,UAAU,aAAa;EACvB,OAAO,aAAa;EACpB,KAAK,aAAa;EAClB,UAAU,aAAa;EACvB,WAAW,aAAa;CAC1B;AACF;;;;;;;;;;AAoDA,SAAgB,WACd,SACA,QACA,UAA6B,CAAC,GACZ;CAClB,MAAM,OAAO,QAAQ,QAAQ;CAC7B,MAAM,kBAAkB,QAAQ,mBAAmB;CACnD,IAAI,CAAC,OAAO,SAAS,IAAI,GACvB,MAAM,IAAI,gBAAgB,iDAAiD,MAAM;CAEnF,IAAI,CAAC,OAAO,SAAS,eAAe,KAAK,mBAAmB,KAAK,kBAAkB,GACjF,MAAM,IAAI,gBACR,sEAAsE,iBACxE;CAGF,MAAM,eAA6B,CAAC;CACpC,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,QAAQ,SAAS;EAC1B,IAAI,OAAO,KAAK,WAAW,YAAY,KAAK,OAAO,KAAK,MAAM,IAC5D,MAAM,IAAI,gBAAgB,gDAAgD;EAE5E,IAAI,CAAC,OAAO,SAAS,KAAK,UAAU,GAClC,MAAM,IAAI,gBACR,6BAA6B,KAAK,OAAO,8BAC3C;EAEF,IAAI,KAAK,IAAI,KAAK,MAAM,GACtB,MAAM,IAAI,gBAAgB,yCAAyC,KAAK,OAAO,EAAE;EAEnF,KAAK,IAAI,KAAK,MAAM;EACpB,aAAa,KAAK;GAChB,QAAQ,KAAK;GACb,YAAY,KAAK;GACjB,GAAI,KAAK,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,KAAK,MAAM;EAC1D,CAAC;CACH;CAEA,MAAM,eAAwB,CAAC;CAC/B,MAAM,2BAAW,IAAI,IAAY;CACjC,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,SAAS,YAAY,KAAK;EAChC,IAAI,SAAS,IAAI,OAAO,EAAE,GACxB,MAAM,IAAI,gBAAgB,mCAAmC,OAAO,GAAG,EAAE;EAE3E,IAAI,KAAK,IAAI,OAAO,KAAK,MAAM,GAC7B,MAAM,IAAI,gBACR,sBAAsB,OAAO,GAAG,mBAAmB,OAAO,KAAK,OAAO,+BACxE;EAEF,MAAM,gBACJ,OAAO,KAAK,cAAc,kBAAkB,WAAW;EACzD,IAAI,kBAAkB,OAAO,iBAC3B,MAAM,IAAI,gBACR,sBAAsB,OAAO,GAAG,0BAA0B,OAAO,gBAAgB,sBAAsB,OAAO,KAAK,WAAW,aAAa,cAAc,2BAA2B,iBACtL;EAEF,SAAS,IAAI,OAAO,EAAE;EACtB,KAAK,IAAI,OAAO,KAAK,MAAM;EAC3B,aAAa,KAAK,MAAM;CAC1B;CAEA,MAAM,QAAQ,SACZ,CAAC,GAAG,cAAc,GAAG,aAAa,KAAK,UAAU,MAAM,IAAI,CAAC,GAC5D,WAAW,IAAI,CACjB;CACA,MAAM,UAAU,MAAM,KAAK,SAAS,KAAK,MAAM;CAQ/C,OAAO;EAAE;EAAO,UAAA;GANd,MAAM,aAAa;IAAE;IAAM;IAAiB;IAAS,QAAQ;GAAa,CAAC;GAC3E;GACA;GACA;GACA,QAAQ;EAEa;CAAE;AAC3B;;;;;;;AAQA,SAAS,SAAY,OAAqB,QAA2B;CACnE,OAAO,MACJ,KAAK,UAAU;EAAE;EAAM,KAAK,OAAO;CAAE,EAAE,CAAC,CACxC,MAAM,MAAM,UAAU,KAAK,MAAM,MAAM,GAAG,CAAC,CAC3C,KAAK,UAAU,MAAM,IAAI;AAC9B;AAEA,SAAS,aAAa,UAAmD;CACvE,OAAO,cAAc;EACnB,QAAQ;EACR,MAAM,SAAS;EACf,iBAAiB,SAAS;EAC1B,SAAS,SAAS;EAClB,QAAQ,SAAS;CACnB,CAAC;AACH;;;;;;;;;AAwEA,SAAgB,UACd,SACA,UACiB;CACjB,MAAM,eAAe,aAAa;EAChC,MAAM,SAAS;EACf,iBAAiB,SAAS;EAC1B,SAAS,SAAS;EAClB,QAAQ,SAAS;CACnB,CAAC;CACD,IAAI,iBAAiB,SAAS,MAC5B,MAAM,IAAI,sBACR,wCAAwC,aAAa,iCAAiC,SAAS,KAAK,6CACtG;CAGF,MAAM,YAAY,IAAI,IAAI,SAAS,OAAO;CAC1C,MAAM,gCAAgB,IAAI,IAA2B;CACrD,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,UAAU,IAAI,OAAO,MAAM,GAC9B,MAAM,IAAI,gBACR,0BAA0B,OAAO,OAAO,wCAC1C;EAEF,IAAI,cAAc,IAAI,OAAO,MAAM,GACjC,MAAM,IAAI,gBAAgB,oCAAoC,OAAO,OAAO,EAAE;EAEhF,IAAI,OAAO,UAAU,QAAQ,CAAC,OAAO,SAAS,OAAO,KAAK,GACxD,MAAM,IAAI,gBACR,0BAA0B,OAAO,OAAO,cAAc,OAAO,MAAM,mCACrE;EAEF,cAAc,IAAI,OAAO,QAAQ,OAAO,KAAK;CAC/C;CAEA,MAAM,SAAsD,CAAC;CAC7D,MAAM,YAAsB,CAAC;CAC7B,MAAM,aAAuB,CAAC;CAC9B,IAAI,SAAS;CACb,IAAI,SAAS;CACb,IAAI,aAAa;CAEjB,KAAK,MAAM,SAAS,SAAS,QAAQ;EACnC,IAAI,SAAS,OAAO,MAAM;EAC1B,IAAI,WAAW,KAAA,GAAW;GACxB,SAAS;IAAE,QAAQ;IAAG,QAAQ;IAAG,QAAQ;IAAG,YAAY;IAAG,MAAM;GAAK;GACtE,OAAO,MAAM,QAAQ;EACvB;EACA,OAAO,UAAU;EACjB,IAAI,CAAC,cAAc,IAAI,MAAM,KAAK,MAAM,GAAG;GACzC,WAAW,KAAK,MAAM,EAAE;GACxB;EACF;EACA,MAAM,QAAQ,cAAc,IAAI,MAAM,KAAK,MAAM,KAAK;EACtD,IAAI,UAAU,MAAM;GAClB,cAAc;GACd,OAAO,cAAc;GACrB;EACF;EAEA,KADqC,SAAS,SAAS,kBAAkB,WAAW,cACjE,MAAM,iBAAiB;GACxC,UAAU;GACV,OAAO,UAAU;EACnB,OAAO;GACL,UAAU;GACV,OAAO,UAAU;GACjB,UAAU,KAAK,MAAM,EAAE;EACzB;CACF;CAEA,MAAM,SAAS,SAAS,OAAO;CAC/B,MAAM,UAAU,SAAS;CACzB,MAAM,SAA0B;EAC9B,QAAQ;EACR;EACA;EACA;EACA;EACA,MAAM;EACN;EACA;EACA;EACA,UAAU,kBAAkB,UAAU,aAAa;CACrD;CAEA,IAAI,WAAW,GACb,OAAO;EACL,GAAG;EACH,QAAQ;EACR,QAAQ;CACV;CAEF,IAAI,WAAW,SAAS,GACtB,OAAO;EACL,GAAG;EACH,QAAQ;EACR,QAAQ,GAAG,WAAW,OAAO,MAAM,OAAO,iCAAiC,WAAW,KAAK,IAAI;CACjG;CAEF,IAAI,YAAY,GACd,OAAO;EACL,GAAG;EACH,QAAQ;EACR,QAAQ,OAAO,OAAO;CACxB;CAGF,KAAK,MAAM,UAAU,OAAO,OAAO,MAAM,GAAG;EAC1C,MAAM,cAAc,OAAO,SAAS,OAAO;EAC3C,OAAO,OAAO,gBAAgB,IAAI,OAAO,OAAO,SAAS;CAC3D;CACA,OAAO,OAAO,SAAS;CACvB,OAAO;AACT;AAEA,SAAS,kBACP,UACA,eACmB;CACnB,MAAM,eAAe,IAAI,IAAI,SAAS,OAAO,KAAK,UAAU,MAAM,KAAK,MAAM,CAAC;CAC9E,IAAI,IAAI;CACR,IAAI,UAAU;CACd,IAAI,WAAW;CACf,KAAK,MAAM,UAAU,SAAS,SAAS;EACrC,IAAI,aAAa,IAAI,MAAM,GAAG;EAC9B,KAAK;EACL,MAAM,QAAQ,cAAc,IAAI,MAAM;EACtC,IAAI,UAAU,KAAA,KAAa,UAAU,MAAM;EAC3C,WAAW;EACX,IAAI,QAAQ,SAAS,iBAAiB,YAAY;CACpD;CACA,OAAO;EAAE;EAAG;EAAS;EAAU,eAAe,YAAY,IAAI,OAAO,WAAW;CAAQ;AAC1F;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AChuBA,MAAM,wBAAwB;CAAC;CAAO;CAAoB;AAAkB;AAmC5E,SAAS,SAAS,OAAe,OAAuB;CACtD,MAAM,KAAK,OAAO,UAAU,WAAW,KAAK,MAAM,KAAK,IAAI;CAC3D,IAAI,CAAC,OAAO,SAAS,EAAE,GACrB,MAAM,IAAI,gBACR,mBAAmB,MAAM,qCAAqC,KAAK,UAAU,KAAK,GACpF;CAEF,OAAO;AACT;;AAGA,SAAgB,yBAAyB,UAA4B,SAAS,YAAkB;CAC9F,SAAS,SAAS,IAAI,GAAG,OAAO,IAAI;CACpC,IAAI,OAAO,SAAS,YAAY,YAAY,SAAS,QAAQ,WAAW,GACtE,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,oCAAoC;CAE1F,IAAI,OAAO,SAAS,eAAe,YAAY,SAAS,WAAW,WAAW,GAC5E,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,uCAAuC;CAE7F,IAAI,SAAS,YAAY,QAAQ,OAAO,SAAS,YAAY,UAC3D,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,0CAA0C;CAEhG,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,SAAS,OAAO,GAAG;EAC3D,IAAI,CAAE,sBAA4C,SAAS,GAAG,GAE5D,MAAM,IAAI,gBACR,mBAAmB,OAAO,gCAAgC,IAAI,aAAa,sBAAsB,KAAK,IAAI,GAC5G;EAEF,IAAI,UAAU,KAAA,KAAa,CAAC,OAAO,SAAS,KAAK,GAC/C,MAAM,IAAI,gBACR,mBAAmB,OAAO,WAAW,IAAI,2BAA2B,OAAO,KAAK,GAClF;CAEJ;AACF;;;;;;AASA,SAAgB,wBACd,QACA,MACkB;CAClB,MAAM,QAAQ,6BAA6B,SAAS,OAAO,0BAA0B,OAAO;CAC5F,IAAI,CAAC,OAAO,SAAS,KAAK,GACxB,MAAM,IAAI,gBACR,+CAA+C,OAAO,EAAE,iEAC1D;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,kBAAkB,MAAM;CAAE;CACnF,yBAAyB,QAAQ;CACjC,OAAO;AACT;;;;;;;;AASA,SAAgB,sBACd,QACA,MACkB;CAClB,MAAM,MAAM,gBAAgB,SAAS,OAAO,aAAa,OAAO;CAChE,IAAI,CAAC,OAAO,SAAS,GAAG,GACtB,MAAM,IAAI,gBACR,4FACF;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,IAAI;CAAE;CAC/D,yBAAyB,QAAQ;CACjC,OAAO;AACT;;;;;;;;AAcA,SAAgB,wBACd,QACA,QACA,MACA,UAA8B,CAAC,GACb;CAClB,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,SAAS,SAAS,KAAK,YAAY,GAC7C,MAAM,IAAI,gBACR,uEAAuE,WACzE;CAEF,MAAM,6BAAa,IAAI,IAAoB;CAC3C,KAAK,MAAM,QAAQ,QAAQ;EACzB,IAAI,CAAC,OAAO,SAAS,KAAK,UAAU,GAClC,MAAM,IAAI,gBACR,yCAAyC,KAAK,OAAO,8BACvD;EAEF,IAAI,WAAW,IAAI,KAAK,MAAM,GAC5B,MAAM,IAAI,gBAAgB,qDAAqD,KAAK,OAAO,EAAE;EAE/F,WAAW,IAAI,KAAK,QAAQ,KAAK,UAAU;CAC7C;CACA,IAAI,SAAS;CACb,IAAI,SAAS;CACb,KAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,QAAQ,WAAW,IAAI,EAAE,MAAM;EACrC,IAAI,UAAU,KAAA,GAAW;EACzB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAC1B,MAAM,IAAI,gBACR,kDAAkD,EAAE,OAAO,gBAC7D;EAEF;EACA,IAAI,KAAK,IAAI,EAAE,QAAQ,KAAK,KAAK,WAAW;CAC9C;CACA,IAAI,WAAW,GACb,MAAM,IAAI,gBACR,yGACF;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,kBAAkB,SAAS,OAAO;CAAE;CAC7F,yBAAyB,QAAQ;CACjC,OAAO;AACT;AAUA,SAAgB,sBAAsB,UAA8B,CAAC,GAAkB;CACrF,KAAK,MAAM,KAAK,SAAS,yBAAyB,CAAC;CACnD,MAAM,QAAQ,QAAQ,KAAK,OAAO;EAAE,GAAG;EAAG,SAAS,EAAE,GAAG,EAAE,QAAQ;CAAE,EAAE;CACtE,OAAO;EACL,MAAM,OAAO,UAAU;GACrB,yBAAyB,QAAQ;GACjC,MAAM,KAAK;IAAE,GAAG;IAAU,SAAS,EAAE,GAAG,SAAS,QAAQ;GAAE,CAAC;EAC9D;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,MAAM,MAAM,KAAK,OAAO;IAAE,GAAG;IAAG,SAAS,EAAE,GAAG,EAAE,QAAQ;GAAE,EAAE;GAClE,OAAO,YAAY,KAAA,IAAY,MAAM,IAAI,QAAQ,MAAM,EAAE,YAAY,OAAO;EAC9E;CACF;AACF;;;;;;;AAQA,SAAgB,kBAAkB,MAA6B;CAC7D,OAAO;EACL,MAAM,OAAO,UAAU;GACrB,yBAAyB,QAAQ;GACjC,MAAM,KAAK,MAAM,OAAO;GACxB,MAAM,UAAU,MAAM,OAAO;GAC7B,MAAM,GAAG,MAAM,QAAQ,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;GACzD,MAAM,GAAG,WAAW,MAAM,GAAG,KAAK,UAAU,QAAQ,EAAE,KAAK,MAAM;EACnE;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,KAAK,MAAM,OAAO;GACxB,IAAI;GACJ,IAAI;IACF,MAAM,MAAM,GAAG,SAAS,MAAM,MAAM;GACtC,SAAS,KAAK;IACZ,IAAK,IAA8B,SAAS,UAAU,OAAO,CAAC;IAC9D,MAAM;GACR;GACA,MAAM,QAAQ,IAAI,MAAM,IAAI;GAC5B,MAAM,YAAgC,CAAC;GACvC,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;IACrC,MAAM,OAAO,MAAM;IACnB,IAAI,KAAK,KAAK,CAAC,CAAC,WAAW,GAAG;IAC9B,IAAI;IACJ,IAAI;KACF,SAAS,KAAK,MAAM,IAAI;IAC1B,SAAS,KAAK;KACZ,MAAM,IAAI,gBACR,4BAA4B,KAAK,4BAA4B,IAAI,EAAE,IAAK,IAAc,SACxF;IACF;IACA,MAAM,WAAW;IACjB,yBAAyB,UAAU,GAAG,KAAK,GAAG,IAAI,GAAG;IACrD,UAAU,KAAK,QAAQ;GACzB;GACA,OAAO,YAAY,KAAA,IAAY,YAAY,UAAU,QAAQ,MAAM,EAAE,YAAY,OAAO;EAC1F;CACF;AACF;AAmDA,MAAM,aAAa;AAEnB,SAAgB,oBACd,SACA,MACgB;CAChB,MAAM,SAAS,SAAS,KAAK,MAAM,MAAM;CACzC,MAAM,SAAS,KAAK,YAAY,UAAU;CAC1C,MAAM,eAAe,KAAK,YAAY,gBAAgB;CACtD,MAAM,kBAAkB,KAAK,YAAY,mBAAmB;CAC5D,MAAM,iBAAiB,KAAK,YAAY,kBAAkB;CAE1D,MAAM,0BAAU,IAAI,IAAwD;CAC5E,KAAK,MAAM,YAAY,SAAS;EAC9B,yBAAyB,QAAQ;EACjC,MAAM,OAAO,SAAS,SAAS,IAAI,uBAAuB,SAAS,QAAQ,GAAG;EAC9E,MAAM,MAAM,QAAQ,IAAI,SAAS,OAAO,KAAK,CAAC;EAC9C,IAAI,KAAK;GAAE,GAAG;GAAU;EAAK,CAAC;EAC9B,QAAQ,IAAI,SAAS,SAAS,GAAG;CACnC;CAEA,MAAM,WAA4B,CAAC;CACnC,MAAM,SAAmB,CAAC;CAC1B,MAAM,sBAAgC,CAAC;CAEvC,KAAK,MAAM,WAAW,CAAC,GAAG,QAAQ,KAAK,CAAC,CAAC,CAAC,KAAK,GAAG;EAChD,MAAM,QAAQ,QAAQ,IAAI,OAAO,CAAC,CAAE,MAAM,GAAG,MAAM,EAAE,OAAO,EAAE,IAAI;EAClE,MAAM,SAAS,MAAM,MAAM,SAAS;EAEpC,MAAM,WAAW,SAAS,OAAO,QAAQ;EACzC,IAAI,UAAU,gBACZ,OAAO,KACL,UAAU,QAAQ,6BAA6B,OAAO,GAAG,MAAM,QAAQ,QAAQ,CAAC,EAAE,uBAAuB,eAAe,GAC1H;EAMF,IAAI,YAAY;EAChB,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAChC,IAAI,MAAM,EAAE,CAAE,eAAe,MAAM,IAAI,EAAE,CAAE,YAAY,YAAY;EAErE,IAAI,aAAa,GAQX;OAAA,CAPiB,MAClB,MAAM,SAAS,CAAC,CAChB,MACE,MACC,OAAO,SAAS,EAAE,QAAQ,gBAAgB,KAC1C,OAAO,SAAS,EAAE,QAAQ,gBAAgB,CAEhC,GACd,OAAO,KACL,UAAU,QAAQ,oBAAoB,MAAM,YAAY,EAAE,CAAE,WAAW,OAAO,MAAM,UAAU,CAAE,WAAW,OAAO,MAAM,UAAU,CAAE,GAAG,gIACzI;EAAA;EAIJ,KAAK,MAAM,UAAU,uBAAuB;GAC1C,MAAM,SAAS,MACZ,QAAQ,MAAM,OAAO,EAAE,QAAQ,YAAY,QAAQ,CAAC,CACpD,KAAK,MAAM,EAAE,QAAQ,OAAQ;GAChC,IAAI,OAAO,WAAW,GAAG;GAEzB,MAAM,cAAc,cAAc,QAAQ,KAAK,WAAW;GAC1D,IAAI,YAAY,UAAU,qBACxB,oBAAoB,KAAK,GAAG,QAAQ,GAAG,QAAQ;GAEjD,MAAM,UAAU,OAAO,OAAO,SAAS;GACvC,MAAM,WAAW,OAAO;GACxB,MAAM,QAAQ,UAAU;GACxB,MAAM,UAAoB,CAAC;GAC3B,IAAI,YAAY,UAAU,iBACxB,QAAQ,KACN,6CAA6C,YAAY,QAAQ,gBAAgB,YAAY,WAAW,QAAQ,CAAC,EAAE,EACrH;GAEF,IAAI,WAAW,SAAS,UAAU,QAChC,QAAQ,KAAK,OAAO,QAAQ,QAAQ,CAAC,EAAE,eAAe,QAAQ;GAEhE,IAAI,WAAW,sBAAsB,WAAW,UAAU,cACxD,QAAQ,KACN,6BAA6B,WAAW,QAAA,CAAS,QAAQ,CAAC,EAAE,iBAAiB,SAAS,QAAQ,CAAC,EAAE,UAAU,aAAa,EAC1H;GAEF,IAAI,WAAW,sBAAsB,WAAW,UAAU,iBACxD,QAAQ,KACN,6BAA6B,WAAW,QAAA,CAAS,QAAQ,CAAC,EAAE,iBAAiB,SAAS,QAAQ,CAAC,EAAE,UAAU,gBAAgB,EAC7H;GAGF,MAAM,UAAU,QAAQ,SAAS;GACjC,MAAM,QAAuB;IAC3B;IACA;IACA,OAAO,YAAY;IACnB;IACA;IACA;IACA;GACF;GACA,IAAI,SAAS;IACX,MAAM,SAAS,QAAQ,KAAK,IAAI;IAChC,OAAO,KAAK,UAAU,QAAQ,IAAI,OAAO,IAAI,MAAM,QAAQ;GAC7D;GACA,SAAS,KAAK,KAAK;EACrB;CACF;CAEA,OAAO;EAAE;EAAU;EAAQ,SAAS,OAAO,WAAW;EAAG;CAAoB;AAC/E;;;;;;;;;AAiBA,SAAgB,gBAAgB,QAAyC;CACvE,OAAO;EAAE,SAAS,OAAO;EAAS,QAAQ,CAAC,GAAG,OAAO,MAAM;CAAE;AAC/D"}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
2
|
import { a as scoreOrigin, i as rolloutRewardFields } from "./reward-nw2xZGZG.js";
|
|
3
|
-
import { s as runTaskScore } from "./run-record-
|
|
3
|
+
import { s as runTaskScore } from "./run-record-DualPTn2.js";
|
|
4
4
|
import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
|
|
5
5
|
import { a as assertMinted, r as ROLLOUT_SCHEMA } from "./schema-C1aaAxTf.js";
|
|
6
6
|
//#region src/rollout/mint.ts
|
|
@@ -315,4 +315,4 @@ async function mintRolloutRows(records, store, options = {}) {
|
|
|
315
315
|
//#endregion
|
|
316
316
|
export { unmintableReasons as n, mintRolloutRows as t };
|
|
317
317
|
|
|
318
|
-
//# sourceMappingURL=mint-
|
|
318
|
+
//# sourceMappingURL=mint-ySIIkKlV.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"mint-Cc1_zwRQ.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
|
|
1
|
+
{"version":3,"file":"mint-ySIIkKlV.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { E as MultishotResult, M as MultishotTransportRequest, T as MultishotPersona, c as RunMultishotMatrixResult, f as RunMultishotOptions, k as MultishotToolDefinition, s as RunMultishotMatrixOptions } from "../../matrix-
|
|
1
|
+
import { E as MultishotResult, M as MultishotTransportRequest, T as MultishotPersona, c as RunMultishotMatrixResult, f as RunMultishotOptions, k as MultishotToolDefinition, s as RunMultishotMatrixOptions } from "../../matrix-CyhW-vgJ.js";
|
|
2
2
|
//#region src/multishot/golden/compare.d.ts
|
|
3
3
|
interface CompareOptions {
|
|
4
4
|
/** Stop after this many mismatches. A structural divergence high in the
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { w as JudgeScore } from "../types-
|
|
2
|
-
import { A as MultishotToolExecutor, C as MultishotFatalToolError, D as MultishotShape, E as MultishotResult, F as assertMultishotShotResult, M as MultishotTransportRequest, N as MultishotTransportResponse, O as MultishotShotResultError, P as MultishotTransportToolCall, S as MultishotDriverEmptyError, T as MultishotPersona, _ as JudgeRunResult, a as MultishotCellOutput, b as runJudge, c as RunMultishotMatrixResult, d as MultishotShot, f as RunMultishotOptions, g as JudgeDimension, h as JudgeConfig, i as ConversationJudgeInput, j as MultishotTransport, k as MultishotToolDefinition, l as computeCellComposite, m as DEFAULT_JUDGE_MODEL, n as CellCompositeInput, o as MultishotJudges, p as runMultishot, r as CellCompositeScore, s as RunMultishotMatrixOptions, t as ArtifactJudgeInput, u as runMultishotMatrix, v as renderDimensions, w as MultishotMessage, x as MultishotArtifact, y as renderJsonFooter } from "../matrix-
|
|
1
|
+
import { w as JudgeScore } from "../types-CS0qk_Yp.js";
|
|
2
|
+
import { A as MultishotToolExecutor, C as MultishotFatalToolError, D as MultishotShape, E as MultishotResult, F as assertMultishotShotResult, M as MultishotTransportRequest, N as MultishotTransportResponse, O as MultishotShotResultError, P as MultishotTransportToolCall, S as MultishotDriverEmptyError, T as MultishotPersona, _ as JudgeRunResult, a as MultishotCellOutput, b as runJudge, c as RunMultishotMatrixResult, d as MultishotShot, f as RunMultishotOptions, g as JudgeDimension, h as JudgeConfig, i as ConversationJudgeInput, j as MultishotTransport, k as MultishotToolDefinition, l as computeCellComposite, m as DEFAULT_JUDGE_MODEL, n as CellCompositeInput, o as MultishotJudges, p as runMultishot, r as CellCompositeScore, s as RunMultishotMatrixOptions, t as ArtifactJudgeInput, u as runMultishotMatrix, v as renderDimensions, w as MultishotMessage, x as MultishotArtifact, y as renderJsonFooter } from "../matrix-CyhW-vgJ.js";
|
|
3
3
|
import { AgentProfile } from "@tangle-network/agent-interface";
|
|
4
4
|
//#region src/multishot/cost.d.ts
|
|
5
5
|
/**
|
package/dist/openapi.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"openapi": "3.1.0",
|
|
3
3
|
"info": {
|
|
4
4
|
"title": "@tangle-network/agent-eval — wire protocol",
|
|
5
|
-
"version": "0.
|
|
5
|
+
"version": "0.181.0",
|
|
6
6
|
"description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.1.0. Bumps on breaking changes to request/response schemas.",
|
|
7
7
|
"contact": {
|
|
8
8
|
"name": "Tangle Network",
|