@tangle-network/agent-eval 0.171.0 → 0.172.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +55 -0
- package/README.md +3 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/analyst/index.d.ts +5 -5
- package/dist/analyst/index.js +2 -2
- package/dist/{experiment-tracker-Dm8yQMqb.d.ts → attestation-CJBGmMVh.d.ts} +78 -2
- package/dist/attestation-CJBGmMVh.d.ts.map +1 -0
- package/dist/{experiment-tracker-BKEumQug.js → attestation-XSUpbc4o.js} +96 -2
- package/dist/attestation-XSUpbc4o.js.map +1 -0
- package/dist/{benchmark-command-D8k3Gf0J.js → benchmark-command-DoFcisuM.js} +2 -2
- package/dist/{benchmark-command-D8k3Gf0J.js.map → benchmark-command-DoFcisuM.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +2 -2
- package/dist/bounded-process-CVOC_D3H.js +181 -0
- package/dist/bounded-process-CVOC_D3H.js.map +1 -0
- package/dist/builder-eval/index.js +39 -98
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +8 -7
- package/dist/campaign/index.js +4 -4
- package/dist/{campaign-B72njjHj.js → campaign-Dp35pBbS.js} +4 -4
- package/dist/{campaign-B72njjHj.js.map → campaign-Dp35pBbS.js.map} +1 -1
- package/dist/canonical-CFpojCN5.d.ts +31 -0
- package/dist/canonical-CFpojCN5.d.ts.map +1 -0
- package/dist/{chat-client-DEtybj5i.js → chat-client-DI79OPye.js} +2 -2
- package/dist/{chat-client-DEtybj5i.js.map → chat-client-DI79OPye.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-CDtcZ3p9.d.ts → client-Df7wdslk.d.ts} +2 -2
- package/dist/{client-CDtcZ3p9.d.ts.map → client-Df7wdslk.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -9
- package/dist/contract/index.js +4 -4
- package/dist/{default-registry-B0s2zU-s.d.ts → default-registry-XxedTLwu.d.ts} +3 -3
- package/dist/{default-registry-B0s2zU-s.d.ts.map → default-registry-XxedTLwu.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Cjy2yhqP.d.ts → define-agent-eval-0wW7gFhr.d.ts} +4 -4
- package/dist/{define-agent-eval-Cjy2yhqP.d.ts.map → define-agent-eval-0wW7gFhr.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Dy8QgxAI.js → define-agent-eval-jS8xj_Q_.js} +2 -2
- package/dist/{define-agent-eval-Dy8QgxAI.js.map → define-agent-eval-jS8xj_Q_.js.map} +1 -1
- package/dist/descriptive-B2iPaT9J.d.ts +89 -0
- package/dist/descriptive-B2iPaT9J.d.ts.map +1 -0
- package/dist/{engine-CAmTUk52.d.ts → engine-BfRay1qD.d.ts} +2 -2
- package/dist/{engine-CAmTUk52.d.ts.map → engine-BfRay1qD.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +3 -54
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +1 -95
- package/dist/experiment/index.js.map +1 -1
- package/dist/{heldout-gate-Dh2b62w8.d.ts → heldout-gate-JgNRDZwZ.d.ts} +3 -3
- package/dist/{heldout-gate-Dh2b62w8.d.ts.map → heldout-gate-JgNRDZwZ.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-lfaSeKSD.d.ts → index-D-UdhAmg.d.ts} +3 -31
- package/dist/index-D-UdhAmg.d.ts.map +1 -0
- package/dist/{index-DT73JraI.d.ts → index-DDAPhUJJ.d.ts} +4 -4
- package/dist/{index-DT73JraI.d.ts.map → index-DDAPhUJJ.d.ts.map} +1 -1
- package/dist/{index-fNXZMCzX.d.ts → index-DnglhM0A.d.ts} +9 -9
- package/dist/{index-fNXZMCzX.d.ts.map → index-DnglhM0A.d.ts.map} +1 -1
- package/dist/{index-8VIogTyS.d.ts → index-_vPrVMRX.d.ts} +6 -6
- package/dist/{index-8VIogTyS.d.ts.map → index-_vPrVMRX.d.ts.map} +1 -1
- package/dist/index.d.ts +116 -15
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -6
- package/dist/index.js.map +1 -1
- package/dist/{integrity-CyWSSoQS.js → integrity-BWywb34E.js} +34 -11
- package/dist/{integrity-CyWSSoQS.js.map → integrity-BWywb34E.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +2 -1
- package/dist/{llm-judge-BtJ2Sfk_.js → llm-judge-aQHIk5_-.js} +16 -10
- package/dist/llm-judge-aQHIk5_-.js.map +1 -0
- package/dist/{matrix-DiHmUobV.d.ts → matrix-Ch8JO1pG.d.ts} +2 -2
- package/dist/{matrix-DiHmUobV.d.ts.map → matrix-Ch8JO1pG.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +162 -2
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +287 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-Cm6DU_Ao.js → produced-state-CxmbFxFd.js} +2 -2
- package/dist/{produced-state-Cm6DU_Ao.js.map → produced-state-CxmbFxFd.js.map} +1 -1
- package/dist/{promotion-policy-WSXtBgBb.d.ts → promotion-policy-BBBcz5_3.d.ts} +2 -2
- package/dist/{promotion-policy-WSXtBgBb.d.ts.map → promotion-policy-BBBcz5_3.d.ts.map} +1 -1
- package/dist/{provenance-CafMdZKM.d.ts → provenance-Dp-vvyrU.d.ts} +13 -5
- package/dist/provenance-Dp-vvyrU.d.ts.map +1 -0
- package/dist/rl.d.ts +1 -1
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -1
- package/dist/{skillopt-optimization-method-DzlF2RM7.js → skillopt-optimization-method-LHi02MzH.js} +2 -2
- package/dist/{skillopt-optimization-method-DzlF2RM7.js.map → skillopt-optimization-method-LHi02MzH.js.map} +1 -1
- package/dist/{statistical-heldout-UhiexnjU.d.ts → statistical-heldout-Yldkntvy.d.ts} +2 -2
- package/dist/{statistical-heldout-UhiexnjU.d.ts.map → statistical-heldout-Yldkntvy.d.ts.map} +1 -1
- package/dist/{store-tool-spans-D_qMl2__.d.ts → store-tool-spans-BvdUbeOB.d.ts} +3 -3
- package/dist/{store-tool-spans-D_qMl2__.d.ts.map → store-tool-spans-BvdUbeOB.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +35 -2
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +83 -38
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{tool-groups-RGYfVWpc.d.ts → tool-groups-DjwlMBvW.d.ts} +2 -2
- package/dist/tool-groups-DjwlMBvW.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/{types-CMyW4GnH.d.ts → types-CCZ34qmV.d.ts} +2 -2
- package/dist/{types-CMyW4GnH.d.ts.map → types-CCZ34qmV.d.ts.map} +1 -1
- package/dist/{types-DeIUdzNd.d.ts → types-CoPUTiXb.d.ts} +23 -92
- package/dist/types-CoPUTiXb.d.ts.map +1 -0
- package/dist/{types-JHMOqZI4.d.ts → types-nokrtr7M.d.ts} +11 -1
- package/dist/types-nokrtr7M.d.ts.map +1 -0
- package/docs/eval-surface-map.md +36 -0
- package/docs/insight-report.md +19 -0
- package/docs/plants.md +123 -0
- package/docs/public-api.md +45 -20
- package/package.json +1 -1
- package/dist/experiment-tracker-BKEumQug.js.map +0 -1
- package/dist/experiment-tracker-Dm8yQMqb.d.ts.map +0 -1
- package/dist/index-lfaSeKSD.d.ts.map +0 -1
- package/dist/llm-judge-BtJ2Sfk_.js.map +0 -1
- package/dist/provenance-CafMdZKM.d.ts.map +0 -1
- package/dist/tool-groups-RGYfVWpc.d.ts.map +0 -1
- package/dist/types-DeIUdzNd.d.ts.map +0 -1
- package/dist/types-JHMOqZI4.d.ts.map +0 -1
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","names":[],"sources":["../../src/meta-eval/calibration.ts","../../src/meta-eval/correlation-study.ts","../../src/meta-eval/sentinel.ts"],"sourcesContent":["/**\n * Calibration curve — binned \"if eval says X, what does reality show?\"\n *\n * Companion to correlationStudy. Raw correlation is a single number;\n * the calibration curve shows *where* the eval is well-calibrated vs\n * overconfident / underconfident. Buckets the eval metric, computes\n * mean outcome per bucket, reports expected-calibration-error (ECE).\n */\n\nimport { runMetricExtractor } from '../trace/query'\nimport type { TraceStore } from '../trace/store'\nimport type { EvalMetricSpec } from './correlation-study'\nimport type { DeploymentOutcome, OutcomeStore } from './outcome-store'\n\nexport interface CalibrationBin {\n lower: number\n upper: number\n n: number\n evalMean: number\n outcomeMean: number\n /** |outcomeMean − evalMean|; contributes to ECE weighted by n/total. */\n gap: number\n}\n\nexport interface CalibrationReport {\n evalMetric: string\n outcomeMetric: string\n n: number\n bins: CalibrationBin[]\n /** Expected Calibration Error — Σ (n_i/N) × |outcomeMean_i − evalMean_i|. */\n ece: number\n /** Max bin gap — upper bound on miscalibration. */\n maxGap: number\n}\n\nexport interface CalibrationOptions {\n bins?: number\n /** Equal-width (fixed bin edges) or equal-frequency (quantile bins). */\n binning?: 'equal-width' | 'equal-frequency'\n /** Clip eval values to [lo, hi] before binning. */\n range?: { lo: number; hi: number }\n}\n\nexport interface CalibrationPair {\n evalScore: number\n outcome: number\n}\n\nexport async function calibrationCurve(\n traceStore: TraceStore,\n outcomeStore: OutcomeStore,\n evalMetric: EvalMetricSpec,\n outcomeMetric: string,\n options: CalibrationOptions = {},\n): Promise<CalibrationReport | null> {\n const runs = await traceStore.listRuns()\n const outcomes = await outcomeStore.list()\n const byRun = new Map<string, DeploymentOutcome[]>()\n for (const o of outcomes) {\n const arr = byRun.get(o.runId) ?? []\n arr.push(o)\n byRun.set(o.runId, arr)\n }\n\n const extract = evalMetric.extract ?? runMetricExtractor(evalMetric.id)\n const pairs: Array<{ x: number; y: number }> = []\n for (const run of runs) {\n const os = byRun.get(run.runId)\n if (!os?.length) continue\n const x = await extract(run, traceStore)\n if (x === null || !Number.isFinite(x)) continue\n const latest = [...os].sort((a, b) => b.capturedAt - a.capturedAt)[0]!\n const y = latest.metrics[outcomeMetric]\n if (typeof y !== 'number' || !Number.isFinite(y)) continue\n pairs.push({ x, y })\n }\n if (pairs.length < 2) return null\n\n return calibrationFromPairs(\n pairs.map((p) => ({ evalScore: p.x, outcome: p.y })),\n evalMetric.id,\n outcomeMetric,\n options,\n )\n}\n\nfunction calibrationFromPairs(\n inputPairs: CalibrationPair[],\n evalMetric: string,\n outcomeMetric: string,\n options: CalibrationOptions = {},\n): CalibrationReport | null {\n const pairs = inputPairs.filter(\n (pair) => Number.isFinite(pair.evalScore) && Number.isFinite(pair.outcome),\n )\n if (pairs.length < 2) return null\n\n const numBins = options.bins ?? 10\n const binning = options.binning ?? 'equal-width'\n const xs = pairs.map((p) => p.evalScore)\n const lo = options.range?.lo ?? Math.min(...xs)\n const hi = options.range?.hi ?? Math.max(...xs)\n\n const bins: CalibrationBin[] = []\n if (binning === 'equal-frequency') {\n const sorted = [...pairs].sort((a, b) => a.evalScore - b.evalScore)\n const perBin = Math.max(1, Math.floor(sorted.length / numBins))\n for (let i = 0; i < sorted.length; i += perBin) {\n const chunk = sorted.slice(i, i + perBin)\n if (chunk.length === 0) continue\n bins.push(toBin(chunk))\n }\n } else {\n const width = (hi - lo) / numBins\n if (width === 0) return null\n for (let i = 0; i < numBins; i++) {\n const binLo = lo + i * width\n const binHi = i === numBins - 1 ? hi + 1e-9 : lo + (i + 1) * width\n const chunk = pairs.filter((p) => p.evalScore >= binLo && p.evalScore < binHi)\n if (chunk.length === 0) continue\n bins.push(toBin(chunk, binLo, binHi))\n }\n }\n\n const total = bins.reduce((a, b) => a + b.n, 0)\n const ece = bins.reduce((a, b) => a + (b.n / total) * b.gap, 0)\n const maxGap = bins.reduce((a, b) => Math.max(a, b.gap), 0)\n\n return { evalMetric, outcomeMetric, n: pairs.length, bins, ece, maxGap }\n}\n\nfunction toBin(chunk: CalibrationPair[], lower?: number, upper?: number): CalibrationBin {\n const xs = chunk.map((c) => c.evalScore)\n const ys = chunk.map((c) => c.outcome)\n const evalMean = mean(xs)\n const outcomeMean = mean(ys)\n return {\n lower: lower ?? Math.min(...xs),\n upper: upper ?? Math.max(...xs),\n n: chunk.length,\n evalMean,\n outcomeMean,\n gap: Math.abs(outcomeMean - evalMean),\n }\n}\n\nfunction mean(xs: number[]): number {\n return xs.reduce((a, b) => a + b, 0) / xs.length\n}\n","/**\n * Correlation study — \"does our eval score predict real-world outcomes?\"\n *\n * This is the load-bearing signal. Takes a TraceStore + OutcomeStore,\n * joins on runId, computes Pearson + Spearman + bootstrap CI for every\n * (evalMetric, outcomeMetric) pair the caller declares.\n *\n * Without this number the framework is ornamental. With it and r > 0.6\n * the framework is a moat — no other agent-eval tool publishes one.\n */\n\nimport { pearsonR, spearmanR } from '../statistics'\nimport { makeRng } from '../statistics/internal'\nimport { runMetricExtractor } from '../trace/query'\nimport type { Run } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport type { DeploymentOutcome, OutcomeFilter, OutcomeStore } from './outcome-store'\n\nexport interface EvalMetricSpec {\n id: string\n /** Extract a scalar from a run. Omit it and `id` must name one of\n * `RUN_METRICS`; any other `id` is refused. */\n extract?: (run: Run, store: TraceStore) => Promise<number | null>\n}\n\nexport interface OutcomePair {\n evalMetric: string\n outcomeMetric: string\n}\n\nexport interface CorrelationResult {\n evalMetric: string\n outcomeMetric: string\n n: number\n pearson: number\n spearman: number\n /** 95% bootstrap CI for Pearson. */\n pearsonCi95: { lower: number; upper: number }\n /** Rough verdict: 'strong' ≥ 0.7, 'moderate' ≥ 0.4, else 'weak'. */\n verdict: 'strong' | 'moderate' | 'weak'\n}\n\nexport interface CorrelationStudyResult {\n pairs: CorrelationResult[]\n joinedSamples: number\n skippedRuns: number\n}\n\nexport interface CorrelationStudyOptions {\n /** Only join outcomes captured within this window after run.startedAt. */\n maxCaptureLagMs?: number\n /** Restrict to a subset of outcomes (cohort, region, source). */\n outcomeFilter?: OutcomeFilter\n /** Which outcome per run to use when multiple exist. Default 'latest'. */\n reduction?: 'latest' | 'mean' | 'max'\n /** Bootstrap iterations for the CI. Default 500. */\n bootstrapIterations?: number\n /** Seed for the bootstrap resampler. Absent, the seed is derived from the\n * paired observations, so the same study reproduces the same interval. */\n seed?: number\n}\n\nexport async function correlationStudy(\n traceStore: TraceStore,\n outcomeStore: OutcomeStore,\n evalMetrics: EvalMetricSpec[],\n outcomeMetricNames: string[],\n options: CorrelationStudyOptions = {},\n): Promise<CorrelationStudyResult> {\n const runs = await traceStore.listRuns()\n const outcomes = await outcomeStore.list(options.outcomeFilter)\n const outcomesByRun = new Map<string, DeploymentOutcome[]>()\n for (const o of outcomes) {\n const arr = outcomesByRun.get(o.runId) ?? []\n arr.push(o)\n outcomesByRun.set(o.runId, arr)\n }\n\n const reduction = options.reduction ?? 'latest'\n const maxLag = options.maxCaptureLagMs ?? Infinity\n\n const pairs: Array<{ evalMetric: string; outcomeMetric: string; xs: number[]; ys: number[] }> = []\n for (const em of evalMetrics) {\n for (const om of outcomeMetricNames) {\n pairs.push({ evalMetric: em.id, outcomeMetric: om, xs: [], ys: [] })\n }\n }\n\n let joined = 0\n let skipped = 0\n for (const run of runs) {\n const os = outcomesByRun.get(run.runId)\n if (!os || os.length === 0) {\n skipped++\n continue\n }\n const eligible = os.filter((o) => o.capturedAt - run.startedAt <= maxLag)\n if (eligible.length === 0) {\n skipped++\n continue\n }\n\n for (const em of evalMetrics) {\n const extract = em.extract ?? runMetricExtractor(em.id)\n const x = await extract(run, traceStore)\n if (x === null || !Number.isFinite(x)) continue\n\n for (const om of outcomeMetricNames) {\n const values = eligible\n .map((o) => o.metrics[om])\n .filter((v): v is number => typeof v === 'number' && Number.isFinite(v))\n if (values.length === 0) continue\n const y = reduce(values, reduction, eligible)\n if (y === null) continue\n const pair = pairs.find((p) => p.evalMetric === em.id && p.outcomeMetric === om)!\n pair.xs.push(x)\n pair.ys.push(y)\n }\n }\n joined++\n }\n\n const results: CorrelationResult[] = pairs\n .filter((p) => p.xs.length >= 3)\n .map((p) => {\n const pearson = pearsonR(p.xs, p.ys)\n const spearman = spearmanR(p.xs, p.ys)\n const pearsonCi95 = bootstrapPearsonCi(\n p.xs,\n p.ys,\n options.bootstrapIterations ?? 500,\n options.seed,\n )\n const verdict: CorrelationResult['verdict'] =\n Math.abs(pearson) >= 0.7 ? 'strong' : Math.abs(pearson) >= 0.4 ? 'moderate' : 'weak'\n return {\n evalMetric: p.evalMetric,\n outcomeMetric: p.outcomeMetric,\n n: p.xs.length,\n pearson,\n spearman,\n pearsonCi95,\n verdict,\n }\n })\n\n return { pairs: results, joinedSamples: joined, skippedRuns: skipped }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────\n\nfunction reduce(\n values: number[],\n kind: 'latest' | 'mean' | 'max',\n outcomes: DeploymentOutcome[],\n): number | null {\n if (values.length === 0) return null\n if (kind === 'mean') return values.reduce((a, b) => a + b, 0) / values.length\n if (kind === 'max') return Math.max(...values)\n // 'latest': pick the outcome captured last, then lookup its metric\n const latest = [...outcomes].sort((a, b) => b.capturedAt - a.capturedAt)[0]\n if (!latest) return null\n const latestKey = Object.keys(latest.metrics)[0]\n const v = latestKey !== undefined ? latest.metrics[latestKey] : undefined\n // For 'latest' we already have `values` aligned; use the last-captured one\n const paired = outcomes\n .map((o) => {\n const k = Object.keys(o.metrics)[0]\n return {\n at: o.capturedAt,\n v: k !== undefined ? values.find((x) => o.metrics[k] === x) : undefined,\n }\n })\n .filter((p) => p.v !== undefined)\n if (paired.length === 0) return v ?? null\n return paired.sort((a, b) => b.at - a.at)[0]?.v ?? null\n}\n\nfunction bootstrapPearsonCi(\n xs: number[],\n ys: number[],\n iterations: number,\n seed: number | undefined,\n): { lower: number; upper: number } {\n const n = xs.length\n if (n < 3) return { lower: NaN, upper: NaN }\n const rng = makeRng(seed, xs, ys)\n const rs: number[] = []\n for (let b = 0; b < iterations; b++) {\n const rx: number[] = new Array(n)\n const ry: number[] = new Array(n)\n for (let i = 0; i < n; i++) {\n const idx = Math.floor(rng() * n)\n rx[i] = xs[idx]!\n ry[i] = ys[idx]!\n }\n const r = pearsonR(rx, ry)\n if (Number.isFinite(r)) rs.push(r)\n }\n rs.sort((a, b) => a - b)\n if (rs.length === 0) return { lower: NaN, upper: NaN }\n return {\n lower: rs[Math.floor(0.025 * rs.length)]!,\n upper: rs[Math.min(rs.length - 1, Math.floor(0.975 * rs.length))]!,\n }\n}\n","/**\n * Judge sentinel — eval trustworthiness as a continuously measured,\n * alarmed trend.\n *\n * Judges are models; models change underneath us; calibration decays.\n * This module composes three existing instruments into one loop:\n *\n * snapshot — adapters turn real instrument outputs (`calibrateJudge` /\n * `calibrateJudgeContinuous`, `corpusInterRaterAgreement` /\n * `continuousAgreement`, a rerun sentinel golden set) into\n * `SentinelSnapshot`s\n * series — `analyzeSeries` (src/series-convergence) classifies each\n * judge×metric history as stabilized / drifting / noisy\n * alarm — `judgeSentinelReport` turns trends + thresholds into named\n * alarms; `evalHealthStamp` is the tiny object a campaign\n * attaches to its verdicts\n *\n * Alarm conditions:\n * - convergence state `drifting-down` on any tracked metric\n * - irr below `minIrr`\n * - calibrationKappa / sentinelPassRate dropped vs the series baseline\n * beyond `maxKappaDrop` / `maxSentinelDrop`\n * - judge model changed with no post-change golden-grounded snapshot\n * (the silent-upgrade trap: judges agreeing with each other after a\n * model swap proves nothing — only re-measuring against gold does)\n * - newest snapshot older than `staleAfterDays` relative to `asOf`\n *\n * Wire-in: run the snapshot adapters from a post-campaign hook or a\n * nightly job, append to a `SentinelStore`, then compute\n * `judgeSentinelReport(await store.history(), { asOf })` with `asOf` =\n * the campaign timestamp. Attach `evalHealthStamp(report)` to campaign\n * verdicts: an alarmed sentinel means every downstream verdict carries\n * an integrity warning until the judge is recalibrated.\n *\n * The module is clock-free — every timestamp (`at`, `asOf`) is\n * caller-supplied ISO-8601, so reports are reproducible.\n */\n\nimport { ValidationError } from '../errors'\nimport type {\n CalibrationResult,\n CandidateScore,\n ContinuousAgreement,\n ContinuousCalibrationResult,\n GoldenItem,\n} from '../judge-calibration'\nimport {\n analyzeSeries,\n type SeriesConvergenceOptions,\n type SeriesConvergenceResult,\n} from '../series-convergence'\nimport type { CorpusAgreementReport } from '../statistics'\n\nconst SENTINEL_METRIC_NAMES = ['irr', 'calibrationKappa', 'sentinelPassRate'] as const\n\nexport type SentinelMetricName = (typeof SENTINEL_METRIC_NAMES)[number]\n\nexport interface SentinelMetrics {\n /** Inter-rater reliability — ICC(2,1) from agreement instruments. */\n irr?: number\n /** Weighted κ vs the human golden set (`calibrateJudge*`). */\n calibrationKappa?: number\n /** Fraction of a fixed sentinel golden set the judge still scores within tolerance. */\n sentinelPassRate?: number\n}\n\nexport interface SentinelSnapshot {\n /** Caller-supplied ISO-8601 timestamp — the module never reads a clock. */\n at: string\n judgeId: string\n /** Exact model identity behind the judge (pin the full version string). */\n judgeModel: string\n /**\n * Metrics measured at `at`. May be empty: an empty-metrics snapshot is a\n * model-change marker — it records the new `judgeModel` immediately and\n * the silent-upgrade alarm stays raised until a golden-grounded snapshot\n * (calibrationKappa or sentinelPassRate) follows.\n */\n metrics: SentinelMetrics\n}\n\n/** Identity + timestamp the caller supplies alongside an instrument output. */\nexport interface SnapshotMeta {\n at: string\n judgeId: string\n judgeModel: string\n}\n\nfunction parseIso(value: string, label: string): number {\n const ms = typeof value === 'string' ? Date.parse(value) : NaN\n if (!Number.isFinite(ms)) {\n throw new ValidationError(\n `judge-sentinel: ${label} is not a parseable ISO timestamp: ${JSON.stringify(value)}`,\n )\n }\n return ms\n}\n\n/** Shape guard for snapshots crossing a boundary (store append, JSONL load, report input). */\nexport function validateSentinelSnapshot(snapshot: SentinelSnapshot, source = 'snapshot'): void {\n parseIso(snapshot.at, `${source}.at`)\n if (typeof snapshot.judgeId !== 'string' || snapshot.judgeId.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeId must be a non-empty string`)\n }\n if (typeof snapshot.judgeModel !== 'string' || snapshot.judgeModel.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeModel must be a non-empty string`)\n }\n if (snapshot.metrics === null || typeof snapshot.metrics !== 'object') {\n throw new ValidationError(`judge-sentinel: ${source}.metrics must be an object (may be empty)`)\n }\n for (const [key, value] of Object.entries(snapshot.metrics)) {\n if (!(SENTINEL_METRIC_NAMES as readonly string[]).includes(key)) {\n // A typo'd key would silently never trend — reject instead.\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics carries unknown key \"${key}\" — known: ${SENTINEL_METRIC_NAMES.join(', ')}`,\n )\n }\n if (value !== undefined && !Number.isFinite(value)) {\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics.${key} is not a finite number: ${String(value)}`,\n )\n }\n }\n}\n\n// ── Snapshot adapters — real instrument output → SentinelSnapshot ───\n\n/**\n * From `calibrateJudge` / `calibrateJudgeContinuous` output. Prefers the\n * un-rounded continuous κ_w when the report carries one (integer κ\n * discards information for [0,1] judges).\n */\nexport function snapshotFromCalibration(\n report: CalibrationResult | ContinuousCalibrationResult,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const kappa = 'weightedKappaContinuous' in report ? report.weightedKappaContinuous : report.kappa\n if (!Number.isFinite(kappa)) {\n throw new ValidationError(\n `snapshotFromCalibration: κ is not finite (n=${report.n}) — calibrate against ≥2 joined golden items before snapshotting`,\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { calibrationKappa: kappa } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n/**\n * From `corpusInterRaterAgreement` (statistics.ts) or `continuousAgreement`\n * (judge-calibration.ts) output. IRR = ICC(2,1) — the reliability\n * coefficient both instruments compute. For ensemble-level agreement,\n * `meta.judgeId` names the ensemble and `meta.judgeModel` pins its\n * composition (e.g. a joined list of member model strings).\n */\nexport function snapshotFromAgreement(\n report: CorpusAgreementReport | ContinuousAgreement,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const irr = 'overallIcc' in report ? report.overallIcc : report.icc\n if (!Number.isFinite(irr)) {\n throw new ValidationError(\n 'snapshotFromAgreement: ICC is not finite — agreement needs ≥2 raters and ≥2 complete items',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { irr } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\nexport interface SentinelSetOptions {\n /** Max |judge − human| for an item to count as passed. Default 0.1 (tuned for [0,1] scores). */\n tolerance?: number\n}\n\n/**\n * From a rerun of a fixed sentinel golden set: the same `GoldenItem`s the\n * judge was originally calibrated on, re-scored periodically. Pass rate =\n * fraction of joined items where the judge stays within `tolerance` of the\n * human score. Items the judge didn't score are excluded from the\n * denominator; zero joined items is an error, not a 0% pass rate.\n */\nexport function snapshotFromSentinelSet(\n scores: CandidateScore[],\n golden: GoldenItem[],\n meta: SnapshotMeta,\n options: SentinelSetOptions = {},\n): SentinelSnapshot {\n const tolerance = options.tolerance ?? 0.1\n if (!Number.isFinite(tolerance) || tolerance < 0) {\n throw new ValidationError(\n `snapshotFromSentinelSet: tolerance must be a finite number ≥ 0, got ${tolerance}`,\n )\n }\n const goldenById = new Map<string, number>()\n for (const item of golden) {\n if (!Number.isFinite(item.humanScore)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: golden item \"${item.itemId}\" has a non-finite humanScore`,\n )\n }\n if (goldenById.has(item.itemId)) {\n throw new ValidationError(`snapshotFromSentinelSet: duplicate golden itemId \"${item.itemId}\"`)\n }\n goldenById.set(item.itemId, item.humanScore)\n }\n let joined = 0\n let passed = 0\n for (const s of scores) {\n const human = goldenById.get(s.itemId)\n if (human === undefined) continue\n if (!Number.isFinite(s.score)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: judge score for item \"${s.itemId}\" is not finite`,\n )\n }\n joined++\n if (Math.abs(s.score - human) <= tolerance) passed++\n }\n if (joined === 0) {\n throw new ValidationError(\n 'snapshotFromSentinelSet: no judge scores joined the golden set — itemIds mismatch or empty sentinel run',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { sentinelPassRate: passed / joined } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n// ── Persistence ──────────────────────────────────────────────────────\n\nexport interface SentinelStore {\n append(snapshot: SentinelSnapshot): Promise<void>\n /** All snapshots (optionally for one judge), in append order. */\n history(judgeId?: string): Promise<SentinelSnapshot[]>\n}\n\nexport function inMemorySentinelStore(initial: SentinelSnapshot[] = []): SentinelStore {\n for (const s of initial) validateSentinelSnapshot(s)\n const items = initial.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n items.push({ ...snapshot, metrics: { ...snapshot.metrics } })\n },\n async history(judgeId) {\n const all = items.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return judgeId === undefined ? all : all.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n/**\n * JSONL store — one snapshot per line, appended atomically per call.\n * A corrupt or shape-invalid line is a loud error naming the file and\n * line number: a sentinel history that silently drops records would\n * defeat the drift detection it exists to provide.\n */\nexport function fileSentinelStore(path: string): SentinelStore {\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n const fs = await import('node:fs/promises')\n const pathMod = await import('node:path')\n await fs.mkdir(pathMod.dirname(path), { recursive: true })\n await fs.appendFile(path, `${JSON.stringify(snapshot)}\\n`, 'utf8')\n },\n async history(judgeId) {\n const fs = await import('node:fs/promises')\n let raw: string\n try {\n raw = await fs.readFile(path, 'utf8')\n } catch (err) {\n if ((err as NodeJS.ErrnoException).code === 'ENOENT') return []\n throw err\n }\n const lines = raw.split('\\n')\n const snapshots: SentinelSnapshot[] = []\n for (let i = 0; i < lines.length; i++) {\n const line = lines[i]!\n if (line.trim().length === 0) continue\n let parsed: unknown\n try {\n parsed = JSON.parse(line)\n } catch (err) {\n throw new ValidationError(\n `judge-sentinel: store at ${path} has a corrupt JSONL line ${i + 1}: ${(err as Error).message}`,\n )\n }\n const snapshot = parsed as SentinelSnapshot\n validateSentinelSnapshot(snapshot, `${path}:${i + 1}`)\n snapshots.push(snapshot)\n }\n return judgeId === undefined ? snapshots : snapshots.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n// ── Trend report ─────────────────────────────────────────────────────\n\nexport interface SentinelThresholds {\n /** Floor for irr — below it the judge ensemble is unreliable regardless of trend. Default 0.6. */\n minIrr?: number\n /** Max tolerated calibrationKappa drop vs the series baseline. Default 0.15. */\n maxKappaDrop?: number\n /** Max tolerated sentinelPassRate drop vs the series baseline. Default 0.1. */\n maxSentinelDrop?: number\n /** Newest snapshot older than this (days, vs asOf) = the sentinel itself went blind. Default 30. */\n staleAfterDays?: number\n}\n\nexport interface JudgeSentinelOptions {\n /** Caller-supplied ISO timestamp staleness is measured against. */\n asOf: string\n thresholds?: SentinelThresholds\n /** Forwarded to `analyzeSeries` (window, stableCv, driftRun). */\n convergence?: SeriesConvergenceOptions\n}\n\nexport interface SentinelTrend {\n judgeId: string\n metric: SentinelMetricName\n /** Verbatim `analyzeSeries` state for this judge×metric history. */\n state: SeriesConvergenceResult['state']\n /** Newest value in the series. */\n current: number\n /** Oldest value in the series — the reference the drop thresholds compare against. */\n baseline: number\n /** current − baseline (signed; negative = decayed). */\n drift: number\n alarmed: boolean\n reason?: string\n}\n\nexport interface SentinelReport {\n perJudge: SentinelTrend[]\n alarms: string[]\n healthy: boolean\n /**\n * `judgeId:metric` pairs with too few snapshots for the convergence\n * machine. Named, not counted — a blind spot you can't see is a blind\n * spot you won't fix. Floor/drop checks still apply to thin series, so\n * insufficient history alone never masks an alarm.\n */\n insufficientHistory: string[]\n}\n\nconst MS_PER_DAY = 86_400_000\n\nexport function judgeSentinelReport(\n history: SentinelSnapshot[],\n opts: JudgeSentinelOptions,\n): SentinelReport {\n const asOfMs = parseIso(opts.asOf, 'asOf')\n const minIrr = opts.thresholds?.minIrr ?? 0.6\n const maxKappaDrop = opts.thresholds?.maxKappaDrop ?? 0.15\n const maxSentinelDrop = opts.thresholds?.maxSentinelDrop ?? 0.1\n const staleAfterDays = opts.thresholds?.staleAfterDays ?? 30\n\n const byJudge = new Map<string, Array<SentinelSnapshot & { atMs: number }>>()\n for (const snapshot of history) {\n validateSentinelSnapshot(snapshot)\n const atMs = parseIso(snapshot.at, `snapshot.at (judge \"${snapshot.judgeId}\")`)\n const arr = byJudge.get(snapshot.judgeId) ?? []\n arr.push({ ...snapshot, atMs })\n byJudge.set(snapshot.judgeId, arr)\n }\n\n const perJudge: SentinelTrend[] = []\n const alarms: string[] = []\n const insufficientHistory: string[] = []\n\n for (const judgeId of [...byJudge.keys()].sort()) {\n const snaps = byJudge.get(judgeId)!.sort((a, b) => a.atMs - b.atMs)\n const newest = snaps[snaps.length - 1]!\n\n const ageDays = (asOfMs - newest.atMs) / MS_PER_DAY\n if (ageDays > staleAfterDays) {\n alarms.push(\n `judge \"${judgeId}\": stale — newest snapshot ${newest.at} is ${ageDays.toFixed(1)}d old vs asOf (limit ${staleAfterDays}d)`,\n )\n }\n\n // Silent-upgrade trap: only the latest model change matters for current\n // trust, and only golden-grounded metrics clear it — irr alone proves\n // judges still agree with each other, not that they're still right.\n let changeIdx = -1\n for (let i = 1; i < snaps.length; i++) {\n if (snaps[i]!.judgeModel !== snaps[i - 1]!.judgeModel) changeIdx = i\n }\n if (changeIdx >= 0) {\n const recalibrated = snaps\n .slice(changeIdx)\n .some(\n (s) =>\n Number.isFinite(s.metrics.calibrationKappa) ||\n Number.isFinite(s.metrics.sentinelPassRate),\n )\n if (!recalibrated) {\n alarms.push(\n `judge \"${judgeId}\": model changed \"${snaps[changeIdx - 1]!.judgeModel}\" → \"${snaps[changeIdx]!.judgeModel}\" at ${snaps[changeIdx]!.at} with no post-change calibration snapshot — silent-upgrade trap; rerun calibrateJudge or the sentinel set against the new model`,\n )\n }\n }\n\n for (const metric of SENTINEL_METRIC_NAMES) {\n const values = snaps\n .filter((s) => typeof s.metrics[metric] === 'number')\n .map((s) => s.metrics[metric]!)\n if (values.length === 0) continue\n\n const convergence = analyzeSeries(values, opts.convergence)\n if (convergence.state === 'insufficient-data')\n insufficientHistory.push(`${judgeId}:${metric}`)\n\n const current = values[values.length - 1]!\n const baseline = values[0]!\n const drift = current - baseline\n const reasons: string[] = []\n if (convergence.state === 'drifting-down') {\n reasons.push(\n `convergence state drifting-down (tail run ${convergence.tailRun}, window mean ${convergence.windowMean.toFixed(3)})`,\n )\n }\n if (metric === 'irr' && current < minIrr) {\n reasons.push(`irr ${current.toFixed(3)} below floor ${minIrr}`)\n }\n if (metric === 'calibrationKappa' && baseline - current > maxKappaDrop) {\n reasons.push(\n `calibrationKappa dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxKappaDrop})`,\n )\n }\n if (metric === 'sentinelPassRate' && baseline - current > maxSentinelDrop) {\n reasons.push(\n `sentinelPassRate dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxSentinelDrop})`,\n )\n }\n\n const alarmed = reasons.length > 0\n const trend: SentinelTrend = {\n judgeId,\n metric,\n state: convergence.state,\n current,\n baseline,\n drift,\n alarmed,\n }\n if (alarmed) {\n trend.reason = reasons.join('; ')\n alarms.push(`judge \"${judgeId}\" ${metric}: ${trend.reason}`)\n }\n perJudge.push(trend)\n }\n }\n\n return { perJudge, alarms, healthy: alarms.length === 0, insufficientHistory }\n}\n\n// ── Verdict stamp ────────────────────────────────────────────────────\n\nexport interface EvalHealthStamp {\n healthy: boolean\n alarms: string[]\n}\n\n/**\n * The object a campaign attaches to its verdicts. Compute it from the\n * sentinel report in a post-campaign hook (or a nightly job feeding the\n * next day's campaigns) and store it alongside the verdict payload.\n * `healthy: false` means the judges that produced those verdicts have an\n * unresolved drift / decay / silent-upgrade / staleness alarm — treat the\n * verdicts as carrying an integrity warning until recalibration clears it.\n */\nexport function evalHealthStamp(report: SentinelReport): EvalHealthStamp {\n return { healthy: report.healthy, alarms: [...report.alarms] }\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAgDA,eAAsB,iBACpB,YACA,cACA,YACA,eACA,UAA8B,CAAC,GACI;CACnC,MAAM,OAAO,MAAM,WAAW,SAAS;CACvC,MAAM,WAAW,MAAM,aAAa,KAAK;CACzC,MAAM,wBAAQ,IAAI,IAAiC;CACnD,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,MAAM,MAAM,IAAI,EAAE,KAAK,KAAK,CAAC;EACnC,IAAI,KAAK,CAAC;EACV,MAAM,IAAI,EAAE,OAAO,GAAG;CACxB;CAEA,MAAM,UAAU,WAAW,WAAW,mBAAmB,WAAW,EAAE;CACtE,MAAM,QAAyC,CAAC;CAChD,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,KAAK,MAAM,IAAI,IAAI,KAAK;EAC9B,IAAI,CAAC,IAAI,QAAQ;EACjB,MAAM,IAAI,MAAM,QAAQ,KAAK,UAAU;EACvC,IAAI,MAAM,QAAQ,CAAC,OAAO,SAAS,CAAC,GAAG;EAEvC,MAAM,IADS,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,aAAa,EAAE,UAAU,CAAC,CAAC,EACnD,CAAC,QAAQ;EACzB,IAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAAG;EAClD,MAAM,KAAK;GAAE;GAAG;EAAE,CAAC;CACrB;CACA,IAAI,MAAM,SAAS,GAAG,OAAO;CAE7B,OAAO,qBACL,MAAM,KAAK,OAAO;EAAE,WAAW,EAAE;EAAG,SAAS,EAAE;CAAE,EAAE,GACnD,WAAW,IACX,eACA,OACF;AACF;AAEA,SAAS,qBACP,YACA,YACA,eACA,UAA8B,CAAC,GACL;CAC1B,MAAM,QAAQ,WAAW,QACtB,SAAS,OAAO,SAAS,KAAK,SAAS,KAAK,OAAO,SAAS,KAAK,OAAO,CAC3E;CACA,IAAI,MAAM,SAAS,GAAG,OAAO;CAE7B,MAAM,UAAU,QAAQ,QAAQ;CAChC,MAAM,UAAU,QAAQ,WAAW;CACnC,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,SAAS;CACvC,MAAM,KAAK,QAAQ,OAAO,MAAM,KAAK,IAAI,GAAG,EAAE;CAC9C,MAAM,KAAK,QAAQ,OAAO,MAAM,KAAK,IAAI,GAAG,EAAE;CAE9C,MAAM,OAAyB,CAAC;CAChC,IAAI,YAAY,mBAAmB;EACjC,MAAM,SAAS,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;EAClE,MAAM,SAAS,KAAK,IAAI,GAAG,KAAK,MAAM,OAAO,SAAS,OAAO,CAAC;EAC9D,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK,QAAQ;GAC9C,MAAM,QAAQ,OAAO,MAAM,GAAG,IAAI,MAAM;GACxC,IAAI,MAAM,WAAW,GAAG;GACxB,KAAK,KAAK,MAAM,KAAK,CAAC;EACxB;CACF,OAAO;EACL,MAAM,SAAS,KAAK,MAAM;EAC1B,IAAI,UAAU,GAAG,OAAO;EACxB,KAAK,IAAI,IAAI,GAAG,IAAI,SAAS,KAAK;GAChC,MAAM,QAAQ,KAAK,IAAI;GACvB,MAAM,QAAQ,MAAM,UAAU,IAAI,KAAK,OAAO,MAAM,IAAI,KAAK;GAC7D,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,aAAa,SAAS,EAAE,YAAY,KAAK;GAC7E,IAAI,MAAM,WAAW,GAAG;GACxB,KAAK,KAAK,MAAM,OAAO,OAAO,KAAK,CAAC;EACtC;CACF;CAEA,MAAM,QAAQ,KAAK,QAAQ,GAAG,MAAM,IAAI,EAAE,GAAG,CAAC;CAC9C,MAAM,MAAM,KAAK,QAAQ,GAAG,MAAM,IAAK,EAAE,IAAI,QAAS,EAAE,KAAK,CAAC;CAC9D,MAAM,SAAS,KAAK,QAAQ,GAAG,MAAM,KAAK,IAAI,GAAG,EAAE,GAAG,GAAG,CAAC;CAE1D,OAAO;EAAE;EAAY;EAAe,GAAG,MAAM;EAAQ;EAAM;EAAK;CAAO;AACzE;AAEA,SAAS,MAAM,OAA0B,OAAgB,OAAgC;CACvF,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,SAAS;CACvC,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,OAAO;CACrC,MAAM,WAAW,KAAK,EAAE;CACxB,MAAM,cAAc,KAAK,EAAE;CAC3B,OAAO;EACL,OAAO,SAAS,KAAK,IAAI,GAAG,EAAE;EAC9B,OAAO,SAAS,KAAK,IAAI,GAAG,EAAE;EAC9B,GAAG,MAAM;EACT;EACA;EACA,KAAK,KAAK,IAAI,cAAc,QAAQ;CACtC;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,OAAO,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,GAAG;AAC5C;;;;;;;;;;;;;ACtFA,eAAsB,iBACpB,YACA,cACA,aACA,oBACA,UAAmC,CAAC,GACH;CACjC,MAAM,OAAO,MAAM,WAAW,SAAS;CACvC,MAAM,WAAW,MAAM,aAAa,KAAK,QAAQ,aAAa;CAC9D,MAAM,gCAAgB,IAAI,IAAiC;CAC3D,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,MAAM,cAAc,IAAI,EAAE,KAAK,KAAK,CAAC;EAC3C,IAAI,KAAK,CAAC;EACV,cAAc,IAAI,EAAE,OAAO,GAAG;CAChC;CAEA,MAAM,YAAY,QAAQ,aAAa;CACvC,MAAM,SAAS,QAAQ,mBAAmB;CAE1C,MAAM,QAA0F,CAAC;CACjG,KAAK,MAAM,MAAM,aACf,KAAK,MAAM,MAAM,oBACf,MAAM,KAAK;EAAE,YAAY,GAAG;EAAI,eAAe;EAAI,IAAI,CAAC;EAAG,IAAI,CAAC;CAAE,CAAC;CAIvE,IAAI,SAAS;CACb,IAAI,UAAU;CACd,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,KAAK,cAAc,IAAI,IAAI,KAAK;EACtC,IAAI,CAAC,MAAM,GAAG,WAAW,GAAG;GAC1B;GACA;EACF;EACA,MAAM,WAAW,GAAG,QAAQ,MAAM,EAAE,aAAa,IAAI,aAAa,MAAM;EACxE,IAAI,SAAS,WAAW,GAAG;GACzB;GACA;EACF;EAEA,KAAK,MAAM,MAAM,aAAa;GAE5B,MAAM,IAAI,OADM,GAAG,WAAW,mBAAmB,GAAG,EAAE,EAAA,CAC9B,KAAK,UAAU;GACvC,IAAI,MAAM,QAAQ,CAAC,OAAO,SAAS,CAAC,GAAG;GAEvC,KAAK,MAAM,MAAM,oBAAoB;IACnC,MAAM,SAAS,SACZ,KAAK,MAAM,EAAE,QAAQ,GAAG,CAAC,CACzB,QAAQ,MAAmB,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,CAAC;IACzE,IAAI,OAAO,WAAW,GAAG;IACzB,MAAM,IAAI,OAAO,QAAQ,WAAW,QAAQ;IAC5C,IAAI,MAAM,MAAM;IAChB,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,eAAe,GAAG,MAAM,EAAE,kBAAkB,EAAE;IAC/E,KAAK,GAAG,KAAK,CAAC;IACd,KAAK,GAAG,KAAK,CAAC;GAChB;EACF;EACA;CACF;CA0BA,OAAO;EAAE,OAxB4B,MAClC,QAAQ,MAAM,EAAE,GAAG,UAAU,CAAC,CAAC,CAC/B,KAAK,MAAM;GACV,MAAM,UAAU,SAAS,EAAE,IAAI,EAAE,EAAE;GACnC,MAAM,WAAW,UAAU,EAAE,IAAI,EAAE,EAAE;GACrC,MAAM,cAAc,mBAClB,EAAE,IACF,EAAE,IACF,QAAQ,uBAAuB,KAC/B,QAAQ,IACV;GACA,MAAM,UACJ,KAAK,IAAI,OAAO,KAAK,KAAM,WAAW,KAAK,IAAI,OAAO,KAAK,KAAM,aAAa;GAChF,OAAO;IACL,YAAY,EAAE;IACd,eAAe,EAAE;IACjB,GAAG,EAAE,GAAG;IACR;IACA;IACA;IACA;GACF;EACF,CAEoB;EAAG,eAAe;EAAQ,aAAa;CAAQ;AACvE;AAIA,SAAS,OACP,QACA,MACA,UACe;CACf,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,IAAI,SAAS,QAAQ,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;CACvE,IAAI,SAAS,OAAO,OAAO,KAAK,IAAI,GAAG,MAAM;CAE7C,MAAM,SAAS,CAAC,GAAG,QAAQ,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,aAAa,EAAE,UAAU,CAAC,CAAC;CACzE,IAAI,CAAC,QAAQ,OAAO;CACpB,MAAM,YAAY,OAAO,KAAK,OAAO,OAAO,CAAC,CAAC;CAC9C,MAAM,IAAI,cAAc,KAAA,IAAY,OAAO,QAAQ,aAAa,KAAA;CAEhE,MAAM,SAAS,SACZ,KAAK,MAAM;EACV,MAAM,IAAI,OAAO,KAAK,EAAE,OAAO,CAAC,CAAC;EACjC,OAAO;GACL,IAAI,EAAE;GACN,GAAG,MAAM,KAAA,IAAY,OAAO,MAAM,MAAM,EAAE,QAAQ,OAAO,CAAC,IAAI,KAAA;EAChE;CACF,CAAC,CAAC,CACD,QAAQ,MAAM,EAAE,MAAM,KAAA,CAAS;CAClC,IAAI,OAAO,WAAW,GAAG,OAAO,KAAK;CACrC,OAAO,OAAO,MAAM,GAAG,MAAM,EAAE,KAAK,EAAE,EAAE,CAAC,CAAC,EAAE,EAAE,KAAK;AACrD;AAEA,SAAS,mBACP,IACA,IACA,YACA,MACkC;CAClC,MAAM,IAAI,GAAG;CACb,IAAI,IAAI,GAAG,OAAO;EAAE,OAAO;EAAK,OAAO;CAAI;CAC3C,MAAM,MAAM,QAAQ,MAAM,IAAI,EAAE;CAChC,MAAM,KAAe,CAAC;CACtB,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,KAAe,IAAI,MAAM,CAAC;EAChC,MAAM,KAAe,IAAI,MAAM,CAAC;EAChC,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;GAC1B,MAAM,MAAM,KAAK,MAAM,IAAI,IAAI,CAAC;GAChC,GAAG,KAAK,GAAG;GACX,GAAG,KAAK,GAAG;EACb;EACA,MAAM,IAAI,SAAS,IAAI,EAAE;EACzB,IAAI,OAAO,SAAS,CAAC,GAAG,GAAG,KAAK,CAAC;CACnC;CACA,GAAG,MAAM,GAAG,MAAM,IAAI,CAAC;CACvB,IAAI,GAAG,WAAW,GAAG,OAAO;EAAE,OAAO;EAAK,OAAO;CAAI;CACrD,OAAO;EACL,OAAO,GAAG,KAAK,MAAM,OAAQ,GAAG,MAAM;EACtC,OAAO,GAAG,KAAK,IAAI,GAAG,SAAS,GAAG,KAAK,MAAM,OAAQ,GAAG,MAAM,CAAC;CACjE;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACxJA,MAAM,wBAAwB;CAAC;CAAO;CAAoB;AAAkB;AAmC5E,SAAS,SAAS,OAAe,OAAuB;CACtD,MAAM,KAAK,OAAO,UAAU,WAAW,KAAK,MAAM,KAAK,IAAI;CAC3D,IAAI,CAAC,OAAO,SAAS,EAAE,GACrB,MAAM,IAAI,gBACR,mBAAmB,MAAM,qCAAqC,KAAK,UAAU,KAAK,GACpF;CAEF,OAAO;AACT;;AAGA,SAAgB,yBAAyB,UAA4B,SAAS,YAAkB;CAC9F,SAAS,SAAS,IAAI,GAAG,OAAO,IAAI;CACpC,IAAI,OAAO,SAAS,YAAY,YAAY,SAAS,QAAQ,WAAW,GACtE,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,oCAAoC;CAE1F,IAAI,OAAO,SAAS,eAAe,YAAY,SAAS,WAAW,WAAW,GAC5E,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,uCAAuC;CAE7F,IAAI,SAAS,YAAY,QAAQ,OAAO,SAAS,YAAY,UAC3D,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,0CAA0C;CAEhG,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,SAAS,OAAO,GAAG;EAC3D,IAAI,CAAE,sBAA4C,SAAS,GAAG,GAE5D,MAAM,IAAI,gBACR,mBAAmB,OAAO,gCAAgC,IAAI,aAAa,sBAAsB,KAAK,IAAI,GAC5G;EAEF,IAAI,UAAU,KAAA,KAAa,CAAC,OAAO,SAAS,KAAK,GAC/C,MAAM,IAAI,gBACR,mBAAmB,OAAO,WAAW,IAAI,2BAA2B,OAAO,KAAK,GAClF;CAEJ;AACF;;;;;;AASA,SAAgB,wBACd,QACA,MACkB;CAClB,MAAM,QAAQ,6BAA6B,SAAS,OAAO,0BAA0B,OAAO;CAC5F,IAAI,CAAC,OAAO,SAAS,KAAK,GACxB,MAAM,IAAI,gBACR,+CAA+C,OAAO,EAAE,iEAC1D;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,kBAAkB,MAAM;CAAE;CACnF,yBAAyB,QAAQ;CACjC,OAAO;AACT;;;;;;;;AASA,SAAgB,sBACd,QACA,MACkB;CAClB,MAAM,MAAM,gBAAgB,SAAS,OAAO,aAAa,OAAO;CAChE,IAAI,CAAC,OAAO,SAAS,GAAG,GACtB,MAAM,IAAI,gBACR,4FACF;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,IAAI;CAAE;CAC/D,yBAAyB,QAAQ;CACjC,OAAO;AACT;;;;;;;;AAcA,SAAgB,wBACd,QACA,QACA,MACA,UAA8B,CAAC,GACb;CAClB,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,SAAS,SAAS,KAAK,YAAY,GAC7C,MAAM,IAAI,gBACR,uEAAuE,WACzE;CAEF,MAAM,6BAAa,IAAI,IAAoB;CAC3C,KAAK,MAAM,QAAQ,QAAQ;EACzB,IAAI,CAAC,OAAO,SAAS,KAAK,UAAU,GAClC,MAAM,IAAI,gBACR,yCAAyC,KAAK,OAAO,8BACvD;EAEF,IAAI,WAAW,IAAI,KAAK,MAAM,GAC5B,MAAM,IAAI,gBAAgB,qDAAqD,KAAK,OAAO,EAAE;EAE/F,WAAW,IAAI,KAAK,QAAQ,KAAK,UAAU;CAC7C;CACA,IAAI,SAAS;CACb,IAAI,SAAS;CACb,KAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,QAAQ,WAAW,IAAI,EAAE,MAAM;EACrC,IAAI,UAAU,KAAA,GAAW;EACzB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAC1B,MAAM,IAAI,gBACR,kDAAkD,EAAE,OAAO,gBAC7D;EAEF;EACA,IAAI,KAAK,IAAI,EAAE,QAAQ,KAAK,KAAK,WAAW;CAC9C;CACA,IAAI,WAAW,GACb,MAAM,IAAI,gBACR,yGACF;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,kBAAkB,SAAS,OAAO;CAAE;CAC7F,yBAAyB,QAAQ;CACjC,OAAO;AACT;AAUA,SAAgB,sBAAsB,UAA8B,CAAC,GAAkB;CACrF,KAAK,MAAM,KAAK,SAAS,yBAAyB,CAAC;CACnD,MAAM,QAAQ,QAAQ,KAAK,OAAO;EAAE,GAAG;EAAG,SAAS,EAAE,GAAG,EAAE,QAAQ;CAAE,EAAE;CACtE,OAAO;EACL,MAAM,OAAO,UAAU;GACrB,yBAAyB,QAAQ;GACjC,MAAM,KAAK;IAAE,GAAG;IAAU,SAAS,EAAE,GAAG,SAAS,QAAQ;GAAE,CAAC;EAC9D;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,MAAM,MAAM,KAAK,OAAO;IAAE,GAAG;IAAG,SAAS,EAAE,GAAG,EAAE,QAAQ;GAAE,EAAE;GAClE,OAAO,YAAY,KAAA,IAAY,MAAM,IAAI,QAAQ,MAAM,EAAE,YAAY,OAAO;EAC9E;CACF;AACF;;;;;;;AAQA,SAAgB,kBAAkB,MAA6B;CAC7D,OAAO;EACL,MAAM,OAAO,UAAU;GACrB,yBAAyB,QAAQ;GACjC,MAAM,KAAK,MAAM,OAAO;GACxB,MAAM,UAAU,MAAM,OAAO;GAC7B,MAAM,GAAG,MAAM,QAAQ,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;GACzD,MAAM,GAAG,WAAW,MAAM,GAAG,KAAK,UAAU,QAAQ,EAAE,KAAK,MAAM;EACnE;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,KAAK,MAAM,OAAO;GACxB,IAAI;GACJ,IAAI;IACF,MAAM,MAAM,GAAG,SAAS,MAAM,MAAM;GACtC,SAAS,KAAK;IACZ,IAAK,IAA8B,SAAS,UAAU,OAAO,CAAC;IAC9D,MAAM;GACR;GACA,MAAM,QAAQ,IAAI,MAAM,IAAI;GAC5B,MAAM,YAAgC,CAAC;GACvC,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;IACrC,MAAM,OAAO,MAAM;IACnB,IAAI,KAAK,KAAK,CAAC,CAAC,WAAW,GAAG;IAC9B,IAAI;IACJ,IAAI;KACF,SAAS,KAAK,MAAM,IAAI;IAC1B,SAAS,KAAK;KACZ,MAAM,IAAI,gBACR,4BAA4B,KAAK,4BAA4B,IAAI,EAAE,IAAK,IAAc,SACxF;IACF;IACA,MAAM,WAAW;IACjB,yBAAyB,UAAU,GAAG,KAAK,GAAG,IAAI,GAAG;IACrD,UAAU,KAAK,QAAQ;GACzB;GACA,OAAO,YAAY,KAAA,IAAY,YAAY,UAAU,QAAQ,MAAM,EAAE,YAAY,OAAO;EAC1F;CACF;AACF;AAmDA,MAAM,aAAa;AAEnB,SAAgB,oBACd,SACA,MACgB;CAChB,MAAM,SAAS,SAAS,KAAK,MAAM,MAAM;CACzC,MAAM,SAAS,KAAK,YAAY,UAAU;CAC1C,MAAM,eAAe,KAAK,YAAY,gBAAgB;CACtD,MAAM,kBAAkB,KAAK,YAAY,mBAAmB;CAC5D,MAAM,iBAAiB,KAAK,YAAY,kBAAkB;CAE1D,MAAM,0BAAU,IAAI,IAAwD;CAC5E,KAAK,MAAM,YAAY,SAAS;EAC9B,yBAAyB,QAAQ;EACjC,MAAM,OAAO,SAAS,SAAS,IAAI,uBAAuB,SAAS,QAAQ,GAAG;EAC9E,MAAM,MAAM,QAAQ,IAAI,SAAS,OAAO,KAAK,CAAC;EAC9C,IAAI,KAAK;GAAE,GAAG;GAAU;EAAK,CAAC;EAC9B,QAAQ,IAAI,SAAS,SAAS,GAAG;CACnC;CAEA,MAAM,WAA4B,CAAC;CACnC,MAAM,SAAmB,CAAC;CAC1B,MAAM,sBAAgC,CAAC;CAEvC,KAAK,MAAM,WAAW,CAAC,GAAG,QAAQ,KAAK,CAAC,CAAC,CAAC,KAAK,GAAG;EAChD,MAAM,QAAQ,QAAQ,IAAI,OAAO,CAAC,CAAE,MAAM,GAAG,MAAM,EAAE,OAAO,EAAE,IAAI;EAClE,MAAM,SAAS,MAAM,MAAM,SAAS;EAEpC,MAAM,WAAW,SAAS,OAAO,QAAQ;EACzC,IAAI,UAAU,gBACZ,OAAO,KACL,UAAU,QAAQ,6BAA6B,OAAO,GAAG,MAAM,QAAQ,QAAQ,CAAC,EAAE,uBAAuB,eAAe,GAC1H;EAMF,IAAI,YAAY;EAChB,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAChC,IAAI,MAAM,EAAE,CAAE,eAAe,MAAM,IAAI,EAAE,CAAE,YAAY,YAAY;EAErE,IAAI,aAAa,GAQX;OAAA,CAPiB,MAClB,MAAM,SAAS,CAAC,CAChB,MACE,MACC,OAAO,SAAS,EAAE,QAAQ,gBAAgB,KAC1C,OAAO,SAAS,EAAE,QAAQ,gBAAgB,CAEhC,GACd,OAAO,KACL,UAAU,QAAQ,oBAAoB,MAAM,YAAY,EAAE,CAAE,WAAW,OAAO,MAAM,UAAU,CAAE,WAAW,OAAO,MAAM,UAAU,CAAE,GAAG,gIACzI;EAAA;EAIJ,KAAK,MAAM,UAAU,uBAAuB;GAC1C,MAAM,SAAS,MACZ,QAAQ,MAAM,OAAO,EAAE,QAAQ,YAAY,QAAQ,CAAC,CACpD,KAAK,MAAM,EAAE,QAAQ,OAAQ;GAChC,IAAI,OAAO,WAAW,GAAG;GAEzB,MAAM,cAAc,cAAc,QAAQ,KAAK,WAAW;GAC1D,IAAI,YAAY,UAAU,qBACxB,oBAAoB,KAAK,GAAG,QAAQ,GAAG,QAAQ;GAEjD,MAAM,UAAU,OAAO,OAAO,SAAS;GACvC,MAAM,WAAW,OAAO;GACxB,MAAM,QAAQ,UAAU;GACxB,MAAM,UAAoB,CAAC;GAC3B,IAAI,YAAY,UAAU,iBACxB,QAAQ,KACN,6CAA6C,YAAY,QAAQ,gBAAgB,YAAY,WAAW,QAAQ,CAAC,EAAE,EACrH;GAEF,IAAI,WAAW,SAAS,UAAU,QAChC,QAAQ,KAAK,OAAO,QAAQ,QAAQ,CAAC,EAAE,eAAe,QAAQ;GAEhE,IAAI,WAAW,sBAAsB,WAAW,UAAU,cACxD,QAAQ,KACN,6BAA6B,WAAW,QAAA,CAAS,QAAQ,CAAC,EAAE,iBAAiB,SAAS,QAAQ,CAAC,EAAE,UAAU,aAAa,EAC1H;GAEF,IAAI,WAAW,sBAAsB,WAAW,UAAU,iBACxD,QAAQ,KACN,6BAA6B,WAAW,QAAA,CAAS,QAAQ,CAAC,EAAE,iBAAiB,SAAS,QAAQ,CAAC,EAAE,UAAU,gBAAgB,EAC7H;GAGF,MAAM,UAAU,QAAQ,SAAS;GACjC,MAAM,QAAuB;IAC3B;IACA;IACA,OAAO,YAAY;IACnB;IACA;IACA;IACA;GACF;GACA,IAAI,SAAS;IACX,MAAM,SAAS,QAAQ,KAAK,IAAI;IAChC,OAAO,KAAK,UAAU,QAAQ,IAAI,OAAO,IAAI,MAAM,QAAQ;GAC7D;GACA,SAAS,KAAK,KAAK;EACrB;CACF;CAEA,OAAO;EAAE;EAAU;EAAQ,SAAS,OAAO,WAAW;EAAG;CAAoB;AAC/E;;;;;;;;;AAiBA,SAAgB,gBAAgB,QAAyC;CACvE,OAAO;EAAE,SAAS,OAAO;EAAS,QAAQ,CAAC,GAAG,OAAO,MAAM;CAAE;AAC/D"}
|
|
1
|
+
{"version":3,"file":"index.js","names":[],"sources":["../../src/meta-eval/calibration.ts","../../src/meta-eval/correlation-study.ts","../../src/meta-eval/plants.ts","../../src/meta-eval/sentinel.ts"],"sourcesContent":["/**\n * Calibration curve — binned \"if eval says X, what does reality show?\"\n *\n * Companion to correlationStudy. Raw correlation is a single number;\n * the calibration curve shows *where* the eval is well-calibrated vs\n * overconfident / underconfident. Buckets the eval metric, computes\n * mean outcome per bucket, reports expected-calibration-error (ECE).\n */\n\nimport { runMetricExtractor } from '../trace/query'\nimport type { TraceStore } from '../trace/store'\nimport type { EvalMetricSpec } from './correlation-study'\nimport type { DeploymentOutcome, OutcomeStore } from './outcome-store'\n\nexport interface CalibrationBin {\n lower: number\n upper: number\n n: number\n evalMean: number\n outcomeMean: number\n /** |outcomeMean − evalMean|; contributes to ECE weighted by n/total. */\n gap: number\n}\n\nexport interface CalibrationReport {\n evalMetric: string\n outcomeMetric: string\n n: number\n bins: CalibrationBin[]\n /** Expected Calibration Error — Σ (n_i/N) × |outcomeMean_i − evalMean_i|. */\n ece: number\n /** Max bin gap — upper bound on miscalibration. */\n maxGap: number\n}\n\nexport interface CalibrationOptions {\n bins?: number\n /** Equal-width (fixed bin edges) or equal-frequency (quantile bins). */\n binning?: 'equal-width' | 'equal-frequency'\n /** Clip eval values to [lo, hi] before binning. */\n range?: { lo: number; hi: number }\n}\n\nexport interface CalibrationPair {\n evalScore: number\n outcome: number\n}\n\nexport async function calibrationCurve(\n traceStore: TraceStore,\n outcomeStore: OutcomeStore,\n evalMetric: EvalMetricSpec,\n outcomeMetric: string,\n options: CalibrationOptions = {},\n): Promise<CalibrationReport | null> {\n const runs = await traceStore.listRuns()\n const outcomes = await outcomeStore.list()\n const byRun = new Map<string, DeploymentOutcome[]>()\n for (const o of outcomes) {\n const arr = byRun.get(o.runId) ?? []\n arr.push(o)\n byRun.set(o.runId, arr)\n }\n\n const extract = evalMetric.extract ?? runMetricExtractor(evalMetric.id)\n const pairs: Array<{ x: number; y: number }> = []\n for (const run of runs) {\n const os = byRun.get(run.runId)\n if (!os?.length) continue\n const x = await extract(run, traceStore)\n if (x === null || !Number.isFinite(x)) continue\n const latest = [...os].sort((a, b) => b.capturedAt - a.capturedAt)[0]!\n const y = latest.metrics[outcomeMetric]\n if (typeof y !== 'number' || !Number.isFinite(y)) continue\n pairs.push({ x, y })\n }\n if (pairs.length < 2) return null\n\n return calibrationFromPairs(\n pairs.map((p) => ({ evalScore: p.x, outcome: p.y })),\n evalMetric.id,\n outcomeMetric,\n options,\n )\n}\n\nfunction calibrationFromPairs(\n inputPairs: CalibrationPair[],\n evalMetric: string,\n outcomeMetric: string,\n options: CalibrationOptions = {},\n): CalibrationReport | null {\n const pairs = inputPairs.filter(\n (pair) => Number.isFinite(pair.evalScore) && Number.isFinite(pair.outcome),\n )\n if (pairs.length < 2) return null\n\n const numBins = options.bins ?? 10\n const binning = options.binning ?? 'equal-width'\n const xs = pairs.map((p) => p.evalScore)\n const lo = options.range?.lo ?? Math.min(...xs)\n const hi = options.range?.hi ?? Math.max(...xs)\n\n const bins: CalibrationBin[] = []\n if (binning === 'equal-frequency') {\n const sorted = [...pairs].sort((a, b) => a.evalScore - b.evalScore)\n const perBin = Math.max(1, Math.floor(sorted.length / numBins))\n for (let i = 0; i < sorted.length; i += perBin) {\n const chunk = sorted.slice(i, i + perBin)\n if (chunk.length === 0) continue\n bins.push(toBin(chunk))\n }\n } else {\n const width = (hi - lo) / numBins\n if (width === 0) return null\n for (let i = 0; i < numBins; i++) {\n const binLo = lo + i * width\n const binHi = i === numBins - 1 ? hi + 1e-9 : lo + (i + 1) * width\n const chunk = pairs.filter((p) => p.evalScore >= binLo && p.evalScore < binHi)\n if (chunk.length === 0) continue\n bins.push(toBin(chunk, binLo, binHi))\n }\n }\n\n const total = bins.reduce((a, b) => a + b.n, 0)\n const ece = bins.reduce((a, b) => a + (b.n / total) * b.gap, 0)\n const maxGap = bins.reduce((a, b) => Math.max(a, b.gap), 0)\n\n return { evalMetric, outcomeMetric, n: pairs.length, bins, ece, maxGap }\n}\n\nfunction toBin(chunk: CalibrationPair[], lower?: number, upper?: number): CalibrationBin {\n const xs = chunk.map((c) => c.evalScore)\n const ys = chunk.map((c) => c.outcome)\n const evalMean = mean(xs)\n const outcomeMean = mean(ys)\n return {\n lower: lower ?? Math.min(...xs),\n upper: upper ?? Math.max(...xs),\n n: chunk.length,\n evalMean,\n outcomeMean,\n gap: Math.abs(outcomeMean - evalMean),\n }\n}\n\nfunction mean(xs: number[]): number {\n return xs.reduce((a, b) => a + b, 0) / xs.length\n}\n","/**\n * Correlation study — \"does our eval score predict real-world outcomes?\"\n *\n * This is the load-bearing signal. Takes a TraceStore + OutcomeStore,\n * joins on runId, computes Pearson + Spearman + bootstrap CI for every\n * (evalMetric, outcomeMetric) pair the caller declares.\n *\n * Without this number the framework is ornamental. With it and r > 0.6\n * the framework is a moat — no other agent-eval tool publishes one.\n */\n\nimport { pearsonR, spearmanR } from '../statistics'\nimport { makeRng } from '../statistics/internal'\nimport { runMetricExtractor } from '../trace/query'\nimport type { Run } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport type { DeploymentOutcome, OutcomeFilter, OutcomeStore } from './outcome-store'\n\nexport interface EvalMetricSpec {\n id: string\n /** Extract a scalar from a run. Omit it and `id` must name one of\n * `RUN_METRICS`; any other `id` is refused. */\n extract?: (run: Run, store: TraceStore) => Promise<number | null>\n}\n\nexport interface OutcomePair {\n evalMetric: string\n outcomeMetric: string\n}\n\nexport interface CorrelationResult {\n evalMetric: string\n outcomeMetric: string\n n: number\n pearson: number\n spearman: number\n /** 95% bootstrap CI for Pearson. */\n pearsonCi95: { lower: number; upper: number }\n /** Rough verdict: 'strong' ≥ 0.7, 'moderate' ≥ 0.4, else 'weak'. */\n verdict: 'strong' | 'moderate' | 'weak'\n}\n\nexport interface CorrelationStudyResult {\n pairs: CorrelationResult[]\n joinedSamples: number\n skippedRuns: number\n}\n\nexport interface CorrelationStudyOptions {\n /** Only join outcomes captured within this window after run.startedAt. */\n maxCaptureLagMs?: number\n /** Restrict to a subset of outcomes (cohort, region, source). */\n outcomeFilter?: OutcomeFilter\n /** Which outcome per run to use when multiple exist. Default 'latest'. */\n reduction?: 'latest' | 'mean' | 'max'\n /** Bootstrap iterations for the CI. Default 500. */\n bootstrapIterations?: number\n /** Seed for the bootstrap resampler. Absent, the seed is derived from the\n * paired observations, so the same study reproduces the same interval. */\n seed?: number\n}\n\nexport async function correlationStudy(\n traceStore: TraceStore,\n outcomeStore: OutcomeStore,\n evalMetrics: EvalMetricSpec[],\n outcomeMetricNames: string[],\n options: CorrelationStudyOptions = {},\n): Promise<CorrelationStudyResult> {\n const runs = await traceStore.listRuns()\n const outcomes = await outcomeStore.list(options.outcomeFilter)\n const outcomesByRun = new Map<string, DeploymentOutcome[]>()\n for (const o of outcomes) {\n const arr = outcomesByRun.get(o.runId) ?? []\n arr.push(o)\n outcomesByRun.set(o.runId, arr)\n }\n\n const reduction = options.reduction ?? 'latest'\n const maxLag = options.maxCaptureLagMs ?? Infinity\n\n const pairs: Array<{ evalMetric: string; outcomeMetric: string; xs: number[]; ys: number[] }> = []\n for (const em of evalMetrics) {\n for (const om of outcomeMetricNames) {\n pairs.push({ evalMetric: em.id, outcomeMetric: om, xs: [], ys: [] })\n }\n }\n\n let joined = 0\n let skipped = 0\n for (const run of runs) {\n const os = outcomesByRun.get(run.runId)\n if (!os || os.length === 0) {\n skipped++\n continue\n }\n const eligible = os.filter((o) => o.capturedAt - run.startedAt <= maxLag)\n if (eligible.length === 0) {\n skipped++\n continue\n }\n\n for (const em of evalMetrics) {\n const extract = em.extract ?? runMetricExtractor(em.id)\n const x = await extract(run, traceStore)\n if (x === null || !Number.isFinite(x)) continue\n\n for (const om of outcomeMetricNames) {\n const values = eligible\n .map((o) => o.metrics[om])\n .filter((v): v is number => typeof v === 'number' && Number.isFinite(v))\n if (values.length === 0) continue\n const y = reduce(values, reduction, eligible)\n if (y === null) continue\n const pair = pairs.find((p) => p.evalMetric === em.id && p.outcomeMetric === om)!\n pair.xs.push(x)\n pair.ys.push(y)\n }\n }\n joined++\n }\n\n const results: CorrelationResult[] = pairs\n .filter((p) => p.xs.length >= 3)\n .map((p) => {\n const pearson = pearsonR(p.xs, p.ys)\n const spearman = spearmanR(p.xs, p.ys)\n const pearsonCi95 = bootstrapPearsonCi(\n p.xs,\n p.ys,\n options.bootstrapIterations ?? 500,\n options.seed,\n )\n const verdict: CorrelationResult['verdict'] =\n Math.abs(pearson) >= 0.7 ? 'strong' : Math.abs(pearson) >= 0.4 ? 'moderate' : 'weak'\n return {\n evalMetric: p.evalMetric,\n outcomeMetric: p.outcomeMetric,\n n: p.xs.length,\n pearson,\n spearman,\n pearsonCi95,\n verdict,\n }\n })\n\n return { pairs: results, joinedSamples: joined, skippedRuns: skipped }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────\n\nfunction reduce(\n values: number[],\n kind: 'latest' | 'mean' | 'max',\n outcomes: DeploymentOutcome[],\n): number | null {\n if (values.length === 0) return null\n if (kind === 'mean') return values.reduce((a, b) => a + b, 0) / values.length\n if (kind === 'max') return Math.max(...values)\n // 'latest': pick the outcome captured last, then lookup its metric\n const latest = [...outcomes].sort((a, b) => b.capturedAt - a.capturedAt)[0]\n if (!latest) return null\n const latestKey = Object.keys(latest.metrics)[0]\n const v = latestKey !== undefined ? latest.metrics[latestKey] : undefined\n // For 'latest' we already have `values` aligned; use the last-captured one\n const paired = outcomes\n .map((o) => {\n const k = Object.keys(o.metrics)[0]\n return {\n at: o.capturedAt,\n v: k !== undefined ? values.find((x) => o.metrics[k] === x) : undefined,\n }\n })\n .filter((p) => p.v !== undefined)\n if (paired.length === 0) return v ?? null\n return paired.sort((a, b) => b.at - a.at)[0]?.v ?? null\n}\n\nfunction bootstrapPearsonCi(\n xs: number[],\n ys: number[],\n iterations: number,\n seed: number | undefined,\n): { lower: number; upper: number } {\n const n = xs.length\n if (n < 3) return { lower: NaN, upper: NaN }\n const rng = makeRng(seed, xs, ys)\n const rs: number[] = []\n for (let b = 0; b < iterations; b++) {\n const rx: number[] = new Array(n)\n const ry: number[] = new Array(n)\n for (let i = 0; i < n; i++) {\n const idx = Math.floor(rng() * n)\n rx[i] = xs[idx]!\n ry[i] = ys[idx]!\n }\n const r = pearsonR(rx, ry)\n if (Number.isFinite(r)) rs.push(r)\n }\n rs.sort((a, b) => a - b)\n if (rs.length === 0) return { lower: NaN, upper: NaN }\n return {\n lower: rs[Math.floor(0.025 * rs.length)]!,\n upper: rs[Math.min(rs.length - 1, Math.floor(0.975 * rs.length))]!,\n }\n}\n","/**\n * Plants — seeded known-wrong items that measure the grader, not the work.\n *\n * A grading run reports how the work scored. It cannot report whether the\n * grader would have noticed a wrong answer, because every item it saw was\n * authored in good faith. A plant closes that hole: an item authored wrong by\n * construction is mixed into the live set, graded by the same path as\n * everything else, and the share of plants the grader refused is the catch\n * rate.\n *\n * Measured motive: a sibling lab ran a deliverable gate that accepted any\n * non-empty submission. It produced six false certifications in seventeen\n * deliveries, and no agent lied — the gate never asked a question the format\n * could fail. A catch rate is the number that would have shown it on day one.\n *\n * This module composes existing primitives rather than adding parallel ones:\n *\n * - A plant IS a {@link GoldenItem} from `../judge-calibration`. Its\n * `humanScore` is the grade a working grader owes the item, so the same\n * array feeds `calibrateJudge` unchanged.\n * - The grader's output is `CandidateScore[]`, the array `calibrateJudge` and\n * `snapshotFromSentinelSet` already consume.\n * - \"Caught\" is `snapshotFromSentinelSet`'s join with the labels inverted:\n * the grade lands on the side of `acceptThreshold` the seed demands.\n * - The manifest is sealed with `hashCanonical` from `../ledger-core/canonical`,\n * the digest the sealed-experiment path uses, so the answer key cannot be\n * revised once the results are in.\n *\n * Blindness has two halves, and this module owns one. It never puts a plant\n * flag on a graded item: `seedPlants` returns the mixed set and a manifest,\n * and only the manifest knows which ids are seeded. Keeping the manifest out\n * of the graded workspace and publishing its `seal` before grading is the\n * caller's half; {@link catchRate} refuses a manifest whose contents no longer\n * match its seal.\n *\n * Refusals, because a catch rate that cannot refuse is not a measurement:\n *\n * - a seeded id with no result makes the report `incomplete`, never a rate\n * over the results that did come back;\n * - zero seeded plants makes it `not_evaluated`, never 1.0;\n * - a result for an id the manifest never handed out is refused outright.\n */\n\nimport { CaptureIntegrityError, ValidationError } from '../errors'\nimport type { GoldenItem } from '../judge-calibration'\nimport { hashCanonical, type LedgerHash } from '../ledger-core/canonical'\nimport { mulberry32 } from '../statistics/random'\n\n/**\n * How a plant item was authored wrong. The class is reported separately in\n * {@link CatchRateReport.byKind} because a grader is routinely sharp on one\n * and blind to another.\n */\nexport type PlantKind =\n /** A load-bearing value is altered: a number off by one, a comparison flipped. */\n | 'wrong-value'\n /** The item carries its own check, and that check passes without testing the claim. */\n | 'self-certifying'\n /** The check names an input that does not exist, so it cannot run at all. */\n | 'unreachable-input'\n /** A copy of an item already in the set, which is owed a duplicate flag rather than a second grade. */\n | 'duplicate'\n\nconst PLANT_KINDS: readonly PlantKind[] = [\n 'wrong-value',\n 'self-certifying',\n 'unreachable-input',\n 'duplicate',\n]\n\n/** What a working grader owes a seeded item. */\nexport type PlantExpectation = 'reject' | 'accept'\n\nconst PLANT_EXPECTATIONS: readonly PlantExpectation[] = ['reject', 'accept']\n\n/** The label boundary a plant record is checked against at definition time. */\nconst RECORD_LABEL_BOUNDARY = 0.5\n\nexport interface Plant {\n /** Name of the plant record. Reported in `missedIds` and `missingIds`. */\n id: string\n kind: PlantKind\n /** The seeded item, indistinguishable from a real one once mixed. */\n item: GoldenItem\n /** The verdict a working grader owes this item. */\n expectedVerdict: PlantExpectation\n}\n\n/**\n * Build one plant record and refuse an incoherent one.\n *\n * The refusal that matters is the last: a record whose `expectedVerdict`\n * disagrees with `item.humanScore` inverts the measurement silently, because\n * the same item then reads as wrong here and as correct to every calibration\n * instrument that joins on the id.\n *\n * `item` is copied field by field so a later mutation of the caller's object\n * cannot change what the manifest sealed, and `group` is dropped when it is\n * absent so the record always has a canonical JSON form.\n */\nexport function definePlant(input: {\n id: string\n kind: PlantKind\n item: GoldenItem\n expectedVerdict: PlantExpectation\n}): Plant {\n const { id, kind, item, expectedVerdict } = input\n if (typeof id !== 'string' || id.trim() === '') {\n throw new ValidationError('definePlant: id must be a non-empty string')\n }\n if (!PLANT_KINDS.includes(kind)) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has kind ${JSON.stringify(kind)}; expected one of ${PLANT_KINDS.join(', ')}`,\n )\n }\n if (!PLANT_EXPECTATIONS.includes(expectedVerdict)) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has expectedVerdict ${JSON.stringify(expectedVerdict)}; expected one of ${PLANT_EXPECTATIONS.join(', ')}`,\n )\n }\n if (typeof item.itemId !== 'string' || item.itemId.trim() === '') {\n throw new ValidationError(`definePlant: plant \"${id}\" has an empty item.itemId`)\n }\n if (!Number.isFinite(item.humanScore) || item.humanScore < 0 || item.humanScore > 1) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has humanScore ${item.humanScore}; expected a finite number in [0, 1]`,\n )\n }\n if (item.group !== undefined && typeof item.group !== 'string') {\n throw new ValidationError(`definePlant: plant \"${id}\" has a non-string item.group`)\n }\n if (item.humanScore === RECORD_LABEL_BOUNDARY) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" has humanScore ${RECORD_LABEL_BOUNDARY}, which states neither a rejection nor an acceptance`,\n )\n }\n const labelSays: PlantExpectation = item.humanScore < RECORD_LABEL_BOUNDARY ? 'reject' : 'accept'\n if (labelSays !== expectedVerdict) {\n throw new ValidationError(\n `definePlant: plant \"${id}\" expects the grader to ${expectedVerdict} it, but humanScore ${item.humanScore} says ${labelSays}`,\n )\n }\n return {\n id,\n kind,\n item: {\n itemId: item.itemId,\n humanScore: item.humanScore,\n ...(item.group === undefined ? {} : { group: item.group }),\n },\n expectedVerdict,\n }\n}\n\nexport interface PlantManifest {\n /**\n * Digest over the seeded order, the plants, and the threshold. Publish it\n * before the grading run: a manifest edited afterwards to match the results\n * no longer matches its seal, and {@link catchRate} refuses it.\n */\n seal: LedgerHash\n /** The seed that fixed the mix order. */\n seed: number\n /** The grade at or above which the graded policy's own gate accepts an item. */\n acceptThreshold: number\n /** Every item id in the seeded set, in the order handed out. */\n itemIds: string[]\n /** The seeded plants. This is the answer key; keep it out of the graded workspace. */\n plants: Plant[]\n}\n\nexport interface SeededGradingSet {\n /**\n * The mixed set in seeded order. Operator-side: it still carries every\n * item's `humanScore`, so hand the graded policy the payload each `itemId`\n * names, never this array.\n */\n items: GoldenItem[]\n manifest: PlantManifest\n}\n\nexport interface SeedPlantsOptions {\n /**\n * Fixes the mix order. The same dataset, plants, threshold, and seed always\n * produce the same seeded order and the same seal.\n */\n seed?: number\n /**\n * The grade at or above which the graded policy's own gate accepts an item.\n * Supply the threshold your gate uses; the default suits a judge scoring in\n * [0, 1] with a pass at the midpoint.\n */\n acceptThreshold?: number\n}\n\n/**\n * Mix plants into a grading set and seal which items they are.\n *\n * Refusals: a duplicate id anywhere in the mixed set (the join is by id, so a\n * repeat makes one of the two unscoreable), a plant whose item id collides\n * with a dataset item (it would shadow real work and grade it as a plant), a\n * dataset item with no usable label, and a plant whose expectation the run's\n * `acceptThreshold` contradicts.\n */\nexport function seedPlants(\n dataset: readonly GoldenItem[],\n plants: readonly Plant[],\n options: SeedPlantsOptions = {},\n): SeededGradingSet {\n const seed = options.seed ?? 7\n const acceptThreshold = options.acceptThreshold ?? 0.5\n if (!Number.isFinite(seed)) {\n throw new ValidationError(`seedPlants: seed must be a finite number, got ${seed}`)\n }\n if (!Number.isFinite(acceptThreshold) || acceptThreshold <= 0 || acceptThreshold > 1) {\n throw new ValidationError(\n `seedPlants: acceptThreshold must be a finite number in (0, 1], got ${acceptThreshold}`,\n )\n }\n\n const datasetItems: GoldenItem[] = []\n const seen = new Set<string>()\n for (const item of dataset) {\n if (typeof item.itemId !== 'string' || item.itemId.trim() === '') {\n throw new ValidationError('seedPlants: a dataset item has an empty itemId')\n }\n if (!Number.isFinite(item.humanScore)) {\n throw new ValidationError(\n `seedPlants: dataset item \"${item.itemId}\" has a non-finite humanScore`,\n )\n }\n if (seen.has(item.itemId)) {\n throw new ValidationError(`seedPlants: duplicate dataset itemId \"${item.itemId}\"`)\n }\n seen.add(item.itemId)\n datasetItems.push({\n itemId: item.itemId,\n humanScore: item.humanScore,\n ...(item.group === undefined ? {} : { group: item.group }),\n })\n }\n\n const sealedPlants: Plant[] = []\n const plantIds = new Set<string>()\n for (const plant of plants) {\n const record = definePlant(plant)\n if (plantIds.has(record.id)) {\n throw new ValidationError(`seedPlants: duplicate plant id \"${record.id}\"`)\n }\n if (seen.has(record.item.itemId)) {\n throw new ValidationError(\n `seedPlants: plant \"${record.id}\" reuses itemId \"${record.item.itemId}\", which is already in the set`,\n )\n }\n const thresholdSays: PlantExpectation =\n record.item.humanScore >= acceptThreshold ? 'accept' : 'reject'\n if (thresholdSays !== record.expectedVerdict) {\n throw new ValidationError(\n `seedPlants: plant \"${record.id}\" expects the grader to ${record.expectedVerdict} it, but humanScore ${record.item.humanScore} is on the ${thresholdSays} side of acceptThreshold ${acceptThreshold}`,\n )\n }\n plantIds.add(record.id)\n seen.add(record.item.itemId)\n sealedPlants.push(record)\n }\n\n const items = shuffled(\n [...datasetItems, ...sealedPlants.map((plant) => plant.item)],\n mulberry32(seed),\n )\n const itemIds = items.map((item) => item.itemId)\n const manifest: PlantManifest = {\n seal: sealManifest({ seed, acceptThreshold, itemIds, plants: sealedPlants }),\n seed,\n acceptThreshold,\n itemIds,\n plants: sealedPlants,\n }\n return { items, manifest }\n}\n\n/**\n * Order by one independent uniform key per item, which is a uniform\n * permutation and a pure function of the seed. A comparator that returns a\n * fresh random sign instead is neither: it is not a consistent ordering, so\n * the permutation it produces is biased and depends on the sort algorithm.\n */\nfunction shuffled<T>(items: readonly T[], random: () => number): T[] {\n return items\n .map((item) => ({ item, key: random() }))\n .sort((left, right) => left.key - right.key)\n .map((entry) => entry.item)\n}\n\nfunction sealManifest(contents: Omit<PlantManifest, 'seal'>): LedgerHash {\n return hashCanonical({\n scheme: 'agent-eval.plant-manifest.v1',\n seed: contents.seed,\n acceptThreshold: contents.acceptThreshold,\n itemIds: contents.itemIds,\n plants: contents.plants,\n })\n}\n\n/**\n * One grader outcome for one item. `score: null` says the grader ran and\n * declined to decide — the check never tested the item's defect. Any\n * `CandidateScore` from a judge run is already a valid outcome.\n */\nexport interface PlantOutcome {\n itemId: string\n score: number | null\n}\n\n/**\n * `evaluated` — a rate stands. `incomplete` — a seeded id had no result.\n * `not_evaluated` — nothing was seeded, or nothing seeded was decided.\n */\nexport type CatchRateStatus = 'evaluated' | 'incomplete' | 'not_evaluated'\n\nexport interface PlantKindCounts {\n seeded: number\n caught: number\n missed: number\n indecisive: number\n /** caught / (caught + missed), or null when the report is not `evaluated`. */\n rate: number | null\n}\n\n/**\n * How the same grader treated the unseeded items of the same set. No labels\n * are needed to read it: a grader that refuses everything scores `rate` 1.0\n * on reject-plants, and a `rejectionRate` of 1.0 here is what separates that\n * reflex from discrimination.\n */\nexport interface UnseededRejection {\n n: number\n decided: number\n rejected: number\n /** rejected / decided, or null when nothing unseeded was decided. */\n rejectionRate: number | null\n}\n\nexport interface CatchRateReport {\n status: CatchRateStatus\n /** Why the status is not `evaluated`. Absent when it is. */\n reason?: string\n seeded: number\n caught: number\n missed: number\n /**\n * The grader returned a result and declined to decide. Counted apart: it\n * enters neither side of `rate`.\n */\n indecisive: number\n /** caught / (caught + missed), or null unless the status is `evaluated`. */\n rate: number | null\n /** One entry per kind actually seeded. A kind nobody seeded is absent, never zero. */\n byKind: Partial<Record<PlantKind, PlantKindCounts>>\n /** Plant ids the grader graded as the seed says it must not. */\n missedIds: string[]\n /** Plant ids with no result at all — the reason a status is `incomplete`. */\n missingIds: string[]\n unseeded: UnseededRejection\n}\n\n/**\n * Score a grading run against its sealed manifest.\n *\n * Refusals: a manifest whose contents no longer match its seal, a result for\n * an id the manifest never handed out, and a repeated result id. Each says\n * the results and the manifest describe different runs, and a rate computed\n * across two runs is a fabrication.\n */\nexport function catchRate(\n results: readonly PlantOutcome[],\n manifest: PlantManifest,\n): CatchRateReport {\n const expectedSeal = sealManifest({\n seed: manifest.seed,\n acceptThreshold: manifest.acceptThreshold,\n itemIds: manifest.itemIds,\n plants: manifest.plants,\n })\n if (expectedSeal !== manifest.seal) {\n throw new CaptureIntegrityError(\n `catchRate: manifest contents hash to ${expectedSeal} but the manifest carries seal ${manifest.seal} — the plant set changed after it was sealed`,\n )\n }\n\n const handedOut = new Set(manifest.itemIds)\n const scoreByItemId = new Map<string, number | null>()\n for (const result of results) {\n if (!handedOut.has(result.itemId)) {\n throw new ValidationError(\n `catchRate: result for \"${result.itemId}\", which this manifest never handed out`,\n )\n }\n if (scoreByItemId.has(result.itemId)) {\n throw new ValidationError(`catchRate: duplicate result for \"${result.itemId}\"`)\n }\n if (result.score !== null && !Number.isFinite(result.score)) {\n throw new ValidationError(\n `catchRate: result for \"${result.itemId}\" has score ${result.score}; expected a finite number or null`,\n )\n }\n scoreByItemId.set(result.itemId, result.score)\n }\n\n const byKind: Partial<Record<PlantKind, PlantKindCounts>> = {}\n const missedIds: string[] = []\n const missingIds: string[] = []\n let caught = 0\n let missed = 0\n let indecisive = 0\n\n for (const plant of manifest.plants) {\n let counts = byKind[plant.kind]\n if (counts === undefined) {\n counts = { seeded: 0, caught: 0, missed: 0, indecisive: 0, rate: null }\n byKind[plant.kind] = counts\n }\n counts.seeded += 1\n if (!scoreByItemId.has(plant.item.itemId)) {\n missingIds.push(plant.id)\n continue\n }\n const score = scoreByItemId.get(plant.item.itemId) ?? null\n if (score === null) {\n indecisive += 1\n counts.indecisive += 1\n continue\n }\n const graderSaid: PlantExpectation = score >= manifest.acceptThreshold ? 'accept' : 'reject'\n if (graderSaid === plant.expectedVerdict) {\n caught += 1\n counts.caught += 1\n } else {\n missed += 1\n counts.missed += 1\n missedIds.push(plant.id)\n }\n }\n\n const seeded = manifest.plants.length\n const decided = caught + missed\n const report: CatchRateReport = {\n status: 'evaluated',\n seeded,\n caught,\n missed,\n indecisive,\n rate: null,\n byKind,\n missedIds,\n missingIds,\n unseeded: unseededRejection(manifest, scoreByItemId),\n }\n\n if (seeded === 0) {\n return {\n ...report,\n status: 'not_evaluated',\n reason: 'no plant was seeded, so the grader was never asked a question it could fail',\n }\n }\n if (missingIds.length > 0) {\n return {\n ...report,\n status: 'incomplete',\n reason: `${missingIds.length} of ${seeded} seeded plants have no result: ${missingIds.join(', ')}`,\n }\n }\n if (decided === 0) {\n return {\n ...report,\n status: 'not_evaluated',\n reason: `all ${seeded} seeded plants are indecisive: no check tested the seeded defect`,\n }\n }\n\n for (const counts of Object.values(byKind)) {\n const kindDecided = counts.caught + counts.missed\n counts.rate = kindDecided === 0 ? null : counts.caught / kindDecided\n }\n report.rate = caught / decided\n return report\n}\n\nfunction unseededRejection(\n manifest: PlantManifest,\n scoreByItemId: ReadonlyMap<string, number | null>,\n): UnseededRejection {\n const plantItemIds = new Set(manifest.plants.map((plant) => plant.item.itemId))\n let n = 0\n let decided = 0\n let rejected = 0\n for (const itemId of manifest.itemIds) {\n if (plantItemIds.has(itemId)) continue\n n += 1\n const score = scoreByItemId.get(itemId)\n if (score === undefined || score === null) continue\n decided += 1\n if (score < manifest.acceptThreshold) rejected += 1\n }\n return { n, decided, rejected, rejectionRate: decided === 0 ? null : rejected / decided }\n}\n","/**\n * Judge sentinel — eval trustworthiness as a continuously measured,\n * alarmed trend.\n *\n * Judges are models; models change underneath us; calibration decays.\n * This module composes three existing instruments into one loop:\n *\n * snapshot — adapters turn real instrument outputs (`calibrateJudge` /\n * `calibrateJudgeContinuous`, `corpusInterRaterAgreement` /\n * `continuousAgreement`, a rerun sentinel golden set) into\n * `SentinelSnapshot`s\n * series — `analyzeSeries` (src/series-convergence) classifies each\n * judge×metric history as stabilized / drifting / noisy\n * alarm — `judgeSentinelReport` turns trends + thresholds into named\n * alarms; `evalHealthStamp` is the tiny object a campaign\n * attaches to its verdicts\n *\n * Alarm conditions:\n * - convergence state `drifting-down` on any tracked metric\n * - irr below `minIrr`\n * - calibrationKappa / sentinelPassRate dropped vs the series baseline\n * beyond `maxKappaDrop` / `maxSentinelDrop`\n * - judge model changed with no post-change golden-grounded snapshot\n * (the silent-upgrade trap: judges agreeing with each other after a\n * model swap proves nothing — only re-measuring against gold does)\n * - newest snapshot older than `staleAfterDays` relative to `asOf`\n *\n * Wire-in: run the snapshot adapters from a post-campaign hook or a\n * nightly job, append to a `SentinelStore`, then compute\n * `judgeSentinelReport(await store.history(), { asOf })` with `asOf` =\n * the campaign timestamp. Attach `evalHealthStamp(report)` to campaign\n * verdicts: an alarmed sentinel means every downstream verdict carries\n * an integrity warning until the judge is recalibrated.\n *\n * The module is clock-free — every timestamp (`at`, `asOf`) is\n * caller-supplied ISO-8601, so reports are reproducible.\n */\n\nimport { ValidationError } from '../errors'\nimport type {\n CalibrationResult,\n CandidateScore,\n ContinuousAgreement,\n ContinuousCalibrationResult,\n GoldenItem,\n} from '../judge-calibration'\nimport {\n analyzeSeries,\n type SeriesConvergenceOptions,\n type SeriesConvergenceResult,\n} from '../series-convergence'\nimport type { CorpusAgreementReport } from '../statistics'\n\nconst SENTINEL_METRIC_NAMES = ['irr', 'calibrationKappa', 'sentinelPassRate'] as const\n\nexport type SentinelMetricName = (typeof SENTINEL_METRIC_NAMES)[number]\n\nexport interface SentinelMetrics {\n /** Inter-rater reliability — ICC(2,1) from agreement instruments. */\n irr?: number\n /** Weighted κ vs the human golden set (`calibrateJudge*`). */\n calibrationKappa?: number\n /** Fraction of a fixed sentinel golden set the judge still scores within tolerance. */\n sentinelPassRate?: number\n}\n\nexport interface SentinelSnapshot {\n /** Caller-supplied ISO-8601 timestamp — the module never reads a clock. */\n at: string\n judgeId: string\n /** Exact model identity behind the judge (pin the full version string). */\n judgeModel: string\n /**\n * Metrics measured at `at`. May be empty: an empty-metrics snapshot is a\n * model-change marker — it records the new `judgeModel` immediately and\n * the silent-upgrade alarm stays raised until a golden-grounded snapshot\n * (calibrationKappa or sentinelPassRate) follows.\n */\n metrics: SentinelMetrics\n}\n\n/** Identity + timestamp the caller supplies alongside an instrument output. */\nexport interface SnapshotMeta {\n at: string\n judgeId: string\n judgeModel: string\n}\n\nfunction parseIso(value: string, label: string): number {\n const ms = typeof value === 'string' ? Date.parse(value) : NaN\n if (!Number.isFinite(ms)) {\n throw new ValidationError(\n `judge-sentinel: ${label} is not a parseable ISO timestamp: ${JSON.stringify(value)}`,\n )\n }\n return ms\n}\n\n/** Shape guard for snapshots crossing a boundary (store append, JSONL load, report input). */\nexport function validateSentinelSnapshot(snapshot: SentinelSnapshot, source = 'snapshot'): void {\n parseIso(snapshot.at, `${source}.at`)\n if (typeof snapshot.judgeId !== 'string' || snapshot.judgeId.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeId must be a non-empty string`)\n }\n if (typeof snapshot.judgeModel !== 'string' || snapshot.judgeModel.length === 0) {\n throw new ValidationError(`judge-sentinel: ${source}.judgeModel must be a non-empty string`)\n }\n if (snapshot.metrics === null || typeof snapshot.metrics !== 'object') {\n throw new ValidationError(`judge-sentinel: ${source}.metrics must be an object (may be empty)`)\n }\n for (const [key, value] of Object.entries(snapshot.metrics)) {\n if (!(SENTINEL_METRIC_NAMES as readonly string[]).includes(key)) {\n // A typo'd key would silently never trend — reject instead.\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics carries unknown key \"${key}\" — known: ${SENTINEL_METRIC_NAMES.join(', ')}`,\n )\n }\n if (value !== undefined && !Number.isFinite(value)) {\n throw new ValidationError(\n `judge-sentinel: ${source}.metrics.${key} is not a finite number: ${String(value)}`,\n )\n }\n }\n}\n\n// ── Snapshot adapters — real instrument output → SentinelSnapshot ───\n\n/**\n * From `calibrateJudge` / `calibrateJudgeContinuous` output. Prefers the\n * un-rounded continuous κ_w when the report carries one (integer κ\n * discards information for [0,1] judges).\n */\nexport function snapshotFromCalibration(\n report: CalibrationResult | ContinuousCalibrationResult,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const kappa = 'weightedKappaContinuous' in report ? report.weightedKappaContinuous : report.kappa\n if (!Number.isFinite(kappa)) {\n throw new ValidationError(\n `snapshotFromCalibration: κ is not finite (n=${report.n}) — calibrate against ≥2 joined golden items before snapshotting`,\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { calibrationKappa: kappa } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n/**\n * From `corpusInterRaterAgreement` (statistics.ts) or `continuousAgreement`\n * (judge-calibration.ts) output. IRR = ICC(2,1) — the reliability\n * coefficient both instruments compute. For ensemble-level agreement,\n * `meta.judgeId` names the ensemble and `meta.judgeModel` pins its\n * composition (e.g. a joined list of member model strings).\n */\nexport function snapshotFromAgreement(\n report: CorpusAgreementReport | ContinuousAgreement,\n meta: SnapshotMeta,\n): SentinelSnapshot {\n const irr = 'overallIcc' in report ? report.overallIcc : report.icc\n if (!Number.isFinite(irr)) {\n throw new ValidationError(\n 'snapshotFromAgreement: ICC is not finite — agreement needs ≥2 raters and ≥2 complete items',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { irr } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\nexport interface SentinelSetOptions {\n /** Max |judge − human| for an item to count as passed. Default 0.1 (tuned for [0,1] scores). */\n tolerance?: number\n}\n\n/**\n * From a rerun of a fixed sentinel golden set: the same `GoldenItem`s the\n * judge was originally calibrated on, re-scored periodically. Pass rate =\n * fraction of joined items where the judge stays within `tolerance` of the\n * human score. Items the judge didn't score are excluded from the\n * denominator; zero joined items is an error, not a 0% pass rate.\n */\nexport function snapshotFromSentinelSet(\n scores: CandidateScore[],\n golden: GoldenItem[],\n meta: SnapshotMeta,\n options: SentinelSetOptions = {},\n): SentinelSnapshot {\n const tolerance = options.tolerance ?? 0.1\n if (!Number.isFinite(tolerance) || tolerance < 0) {\n throw new ValidationError(\n `snapshotFromSentinelSet: tolerance must be a finite number ≥ 0, got ${tolerance}`,\n )\n }\n const goldenById = new Map<string, number>()\n for (const item of golden) {\n if (!Number.isFinite(item.humanScore)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: golden item \"${item.itemId}\" has a non-finite humanScore`,\n )\n }\n if (goldenById.has(item.itemId)) {\n throw new ValidationError(`snapshotFromSentinelSet: duplicate golden itemId \"${item.itemId}\"`)\n }\n goldenById.set(item.itemId, item.humanScore)\n }\n let joined = 0\n let passed = 0\n for (const s of scores) {\n const human = goldenById.get(s.itemId)\n if (human === undefined) continue\n if (!Number.isFinite(s.score)) {\n throw new ValidationError(\n `snapshotFromSentinelSet: judge score for item \"${s.itemId}\" is not finite`,\n )\n }\n joined++\n if (Math.abs(s.score - human) <= tolerance) passed++\n }\n if (joined === 0) {\n throw new ValidationError(\n 'snapshotFromSentinelSet: no judge scores joined the golden set — itemIds mismatch or empty sentinel run',\n )\n }\n const snapshot: SentinelSnapshot = { ...meta, metrics: { sentinelPassRate: passed / joined } }\n validateSentinelSnapshot(snapshot)\n return snapshot\n}\n\n// ── Persistence ──────────────────────────────────────────────────────\n\nexport interface SentinelStore {\n append(snapshot: SentinelSnapshot): Promise<void>\n /** All snapshots (optionally for one judge), in append order. */\n history(judgeId?: string): Promise<SentinelSnapshot[]>\n}\n\nexport function inMemorySentinelStore(initial: SentinelSnapshot[] = []): SentinelStore {\n for (const s of initial) validateSentinelSnapshot(s)\n const items = initial.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n items.push({ ...snapshot, metrics: { ...snapshot.metrics } })\n },\n async history(judgeId) {\n const all = items.map((s) => ({ ...s, metrics: { ...s.metrics } }))\n return judgeId === undefined ? all : all.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n/**\n * JSONL store — one snapshot per line, appended atomically per call.\n * A corrupt or shape-invalid line is a loud error naming the file and\n * line number: a sentinel history that silently drops records would\n * defeat the drift detection it exists to provide.\n */\nexport function fileSentinelStore(path: string): SentinelStore {\n return {\n async append(snapshot) {\n validateSentinelSnapshot(snapshot)\n const fs = await import('node:fs/promises')\n const pathMod = await import('node:path')\n await fs.mkdir(pathMod.dirname(path), { recursive: true })\n await fs.appendFile(path, `${JSON.stringify(snapshot)}\\n`, 'utf8')\n },\n async history(judgeId) {\n const fs = await import('node:fs/promises')\n let raw: string\n try {\n raw = await fs.readFile(path, 'utf8')\n } catch (err) {\n if ((err as NodeJS.ErrnoException).code === 'ENOENT') return []\n throw err\n }\n const lines = raw.split('\\n')\n const snapshots: SentinelSnapshot[] = []\n for (let i = 0; i < lines.length; i++) {\n const line = lines[i]!\n if (line.trim().length === 0) continue\n let parsed: unknown\n try {\n parsed = JSON.parse(line)\n } catch (err) {\n throw new ValidationError(\n `judge-sentinel: store at ${path} has a corrupt JSONL line ${i + 1}: ${(err as Error).message}`,\n )\n }\n const snapshot = parsed as SentinelSnapshot\n validateSentinelSnapshot(snapshot, `${path}:${i + 1}`)\n snapshots.push(snapshot)\n }\n return judgeId === undefined ? snapshots : snapshots.filter((s) => s.judgeId === judgeId)\n },\n }\n}\n\n// ── Trend report ─────────────────────────────────────────────────────\n\nexport interface SentinelThresholds {\n /** Floor for irr — below it the judge ensemble is unreliable regardless of trend. Default 0.6. */\n minIrr?: number\n /** Max tolerated calibrationKappa drop vs the series baseline. Default 0.15. */\n maxKappaDrop?: number\n /** Max tolerated sentinelPassRate drop vs the series baseline. Default 0.1. */\n maxSentinelDrop?: number\n /** Newest snapshot older than this (days, vs asOf) = the sentinel itself went blind. Default 30. */\n staleAfterDays?: number\n}\n\nexport interface JudgeSentinelOptions {\n /** Caller-supplied ISO timestamp staleness is measured against. */\n asOf: string\n thresholds?: SentinelThresholds\n /** Forwarded to `analyzeSeries` (window, stableCv, driftRun). */\n convergence?: SeriesConvergenceOptions\n}\n\nexport interface SentinelTrend {\n judgeId: string\n metric: SentinelMetricName\n /** Verbatim `analyzeSeries` state for this judge×metric history. */\n state: SeriesConvergenceResult['state']\n /** Newest value in the series. */\n current: number\n /** Oldest value in the series — the reference the drop thresholds compare against. */\n baseline: number\n /** current − baseline (signed; negative = decayed). */\n drift: number\n alarmed: boolean\n reason?: string\n}\n\nexport interface SentinelReport {\n perJudge: SentinelTrend[]\n alarms: string[]\n healthy: boolean\n /**\n * `judgeId:metric` pairs with too few snapshots for the convergence\n * machine. Named, not counted — a blind spot you can't see is a blind\n * spot you won't fix. Floor/drop checks still apply to thin series, so\n * insufficient history alone never masks an alarm.\n */\n insufficientHistory: string[]\n}\n\nconst MS_PER_DAY = 86_400_000\n\nexport function judgeSentinelReport(\n history: SentinelSnapshot[],\n opts: JudgeSentinelOptions,\n): SentinelReport {\n const asOfMs = parseIso(opts.asOf, 'asOf')\n const minIrr = opts.thresholds?.minIrr ?? 0.6\n const maxKappaDrop = opts.thresholds?.maxKappaDrop ?? 0.15\n const maxSentinelDrop = opts.thresholds?.maxSentinelDrop ?? 0.1\n const staleAfterDays = opts.thresholds?.staleAfterDays ?? 30\n\n const byJudge = new Map<string, Array<SentinelSnapshot & { atMs: number }>>()\n for (const snapshot of history) {\n validateSentinelSnapshot(snapshot)\n const atMs = parseIso(snapshot.at, `snapshot.at (judge \"${snapshot.judgeId}\")`)\n const arr = byJudge.get(snapshot.judgeId) ?? []\n arr.push({ ...snapshot, atMs })\n byJudge.set(snapshot.judgeId, arr)\n }\n\n const perJudge: SentinelTrend[] = []\n const alarms: string[] = []\n const insufficientHistory: string[] = []\n\n for (const judgeId of [...byJudge.keys()].sort()) {\n const snaps = byJudge.get(judgeId)!.sort((a, b) => a.atMs - b.atMs)\n const newest = snaps[snaps.length - 1]!\n\n const ageDays = (asOfMs - newest.atMs) / MS_PER_DAY\n if (ageDays > staleAfterDays) {\n alarms.push(\n `judge \"${judgeId}\": stale — newest snapshot ${newest.at} is ${ageDays.toFixed(1)}d old vs asOf (limit ${staleAfterDays}d)`,\n )\n }\n\n // Silent-upgrade trap: only the latest model change matters for current\n // trust, and only golden-grounded metrics clear it — irr alone proves\n // judges still agree with each other, not that they're still right.\n let changeIdx = -1\n for (let i = 1; i < snaps.length; i++) {\n if (snaps[i]!.judgeModel !== snaps[i - 1]!.judgeModel) changeIdx = i\n }\n if (changeIdx >= 0) {\n const recalibrated = snaps\n .slice(changeIdx)\n .some(\n (s) =>\n Number.isFinite(s.metrics.calibrationKappa) ||\n Number.isFinite(s.metrics.sentinelPassRate),\n )\n if (!recalibrated) {\n alarms.push(\n `judge \"${judgeId}\": model changed \"${snaps[changeIdx - 1]!.judgeModel}\" → \"${snaps[changeIdx]!.judgeModel}\" at ${snaps[changeIdx]!.at} with no post-change calibration snapshot — silent-upgrade trap; rerun calibrateJudge or the sentinel set against the new model`,\n )\n }\n }\n\n for (const metric of SENTINEL_METRIC_NAMES) {\n const values = snaps\n .filter((s) => typeof s.metrics[metric] === 'number')\n .map((s) => s.metrics[metric]!)\n if (values.length === 0) continue\n\n const convergence = analyzeSeries(values, opts.convergence)\n if (convergence.state === 'insufficient-data')\n insufficientHistory.push(`${judgeId}:${metric}`)\n\n const current = values[values.length - 1]!\n const baseline = values[0]!\n const drift = current - baseline\n const reasons: string[] = []\n if (convergence.state === 'drifting-down') {\n reasons.push(\n `convergence state drifting-down (tail run ${convergence.tailRun}, window mean ${convergence.windowMean.toFixed(3)})`,\n )\n }\n if (metric === 'irr' && current < minIrr) {\n reasons.push(`irr ${current.toFixed(3)} below floor ${minIrr}`)\n }\n if (metric === 'calibrationKappa' && baseline - current > maxKappaDrop) {\n reasons.push(\n `calibrationKappa dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxKappaDrop})`,\n )\n }\n if (metric === 'sentinelPassRate' && baseline - current > maxSentinelDrop) {\n reasons.push(\n `sentinelPassRate dropped ${(baseline - current).toFixed(3)} from baseline ${baseline.toFixed(3)} (limit ${maxSentinelDrop})`,\n )\n }\n\n const alarmed = reasons.length > 0\n const trend: SentinelTrend = {\n judgeId,\n metric,\n state: convergence.state,\n current,\n baseline,\n drift,\n alarmed,\n }\n if (alarmed) {\n trend.reason = reasons.join('; ')\n alarms.push(`judge \"${judgeId}\" ${metric}: ${trend.reason}`)\n }\n perJudge.push(trend)\n }\n }\n\n return { perJudge, alarms, healthy: alarms.length === 0, insufficientHistory }\n}\n\n// ── Verdict stamp ────────────────────────────────────────────────────\n\nexport interface EvalHealthStamp {\n healthy: boolean\n alarms: string[]\n}\n\n/**\n * The object a campaign attaches to its verdicts. Compute it from the\n * sentinel report in a post-campaign hook (or a nightly job feeding the\n * next day's campaigns) and store it alongside the verdict payload.\n * `healthy: false` means the judges that produced those verdicts have an\n * unresolved drift / decay / silent-upgrade / staleness alarm — treat the\n * verdicts as carrying an integrity warning until recalibration clears it.\n */\nexport function evalHealthStamp(report: SentinelReport): EvalHealthStamp {\n return { healthy: report.healthy, alarms: [...report.alarms] }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;AAgDA,eAAsB,iBACpB,YACA,cACA,YACA,eACA,UAA8B,CAAC,GACI;CACnC,MAAM,OAAO,MAAM,WAAW,SAAS;CACvC,MAAM,WAAW,MAAM,aAAa,KAAK;CACzC,MAAM,wBAAQ,IAAI,IAAiC;CACnD,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,MAAM,MAAM,IAAI,EAAE,KAAK,KAAK,CAAC;EACnC,IAAI,KAAK,CAAC;EACV,MAAM,IAAI,EAAE,OAAO,GAAG;CACxB;CAEA,MAAM,UAAU,WAAW,WAAW,mBAAmB,WAAW,EAAE;CACtE,MAAM,QAAyC,CAAC;CAChD,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,KAAK,MAAM,IAAI,IAAI,KAAK;EAC9B,IAAI,CAAC,IAAI,QAAQ;EACjB,MAAM,IAAI,MAAM,QAAQ,KAAK,UAAU;EACvC,IAAI,MAAM,QAAQ,CAAC,OAAO,SAAS,CAAC,GAAG;EAEvC,MAAM,IADS,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,aAAa,EAAE,UAAU,CAAC,CAAC,EACnD,CAAC,QAAQ;EACzB,IAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAAG;EAClD,MAAM,KAAK;GAAE;GAAG;EAAE,CAAC;CACrB;CACA,IAAI,MAAM,SAAS,GAAG,OAAO;CAE7B,OAAO,qBACL,MAAM,KAAK,OAAO;EAAE,WAAW,EAAE;EAAG,SAAS,EAAE;CAAE,EAAE,GACnD,WAAW,IACX,eACA,OACF;AACF;AAEA,SAAS,qBACP,YACA,YACA,eACA,UAA8B,CAAC,GACL;CAC1B,MAAM,QAAQ,WAAW,QACtB,SAAS,OAAO,SAAS,KAAK,SAAS,KAAK,OAAO,SAAS,KAAK,OAAO,CAC3E;CACA,IAAI,MAAM,SAAS,GAAG,OAAO;CAE7B,MAAM,UAAU,QAAQ,QAAQ;CAChC,MAAM,UAAU,QAAQ,WAAW;CACnC,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,SAAS;CACvC,MAAM,KAAK,QAAQ,OAAO,MAAM,KAAK,IAAI,GAAG,EAAE;CAC9C,MAAM,KAAK,QAAQ,OAAO,MAAM,KAAK,IAAI,GAAG,EAAE;CAE9C,MAAM,OAAyB,CAAC;CAChC,IAAI,YAAY,mBAAmB;EACjC,MAAM,SAAS,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;EAClE,MAAM,SAAS,KAAK,IAAI,GAAG,KAAK,MAAM,OAAO,SAAS,OAAO,CAAC;EAC9D,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK,QAAQ;GAC9C,MAAM,QAAQ,OAAO,MAAM,GAAG,IAAI,MAAM;GACxC,IAAI,MAAM,WAAW,GAAG;GACxB,KAAK,KAAK,MAAM,KAAK,CAAC;EACxB;CACF,OAAO;EACL,MAAM,SAAS,KAAK,MAAM;EAC1B,IAAI,UAAU,GAAG,OAAO;EACxB,KAAK,IAAI,IAAI,GAAG,IAAI,SAAS,KAAK;GAChC,MAAM,QAAQ,KAAK,IAAI;GACvB,MAAM,QAAQ,MAAM,UAAU,IAAI,KAAK,OAAO,MAAM,IAAI,KAAK;GAC7D,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,aAAa,SAAS,EAAE,YAAY,KAAK;GAC7E,IAAI,MAAM,WAAW,GAAG;GACxB,KAAK,KAAK,MAAM,OAAO,OAAO,KAAK,CAAC;EACtC;CACF;CAEA,MAAM,QAAQ,KAAK,QAAQ,GAAG,MAAM,IAAI,EAAE,GAAG,CAAC;CAC9C,MAAM,MAAM,KAAK,QAAQ,GAAG,MAAM,IAAK,EAAE,IAAI,QAAS,EAAE,KAAK,CAAC;CAC9D,MAAM,SAAS,KAAK,QAAQ,GAAG,MAAM,KAAK,IAAI,GAAG,EAAE,GAAG,GAAG,CAAC;CAE1D,OAAO;EAAE;EAAY;EAAe,GAAG,MAAM;EAAQ;EAAM;EAAK;CAAO;AACzE;AAEA,SAAS,MAAM,OAA0B,OAAgB,OAAgC;CACvF,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,SAAS;CACvC,MAAM,KAAK,MAAM,KAAK,MAAM,EAAE,OAAO;CACrC,MAAM,WAAW,KAAK,EAAE;CACxB,MAAM,cAAc,KAAK,EAAE;CAC3B,OAAO;EACL,OAAO,SAAS,KAAK,IAAI,GAAG,EAAE;EAC9B,OAAO,SAAS,KAAK,IAAI,GAAG,EAAE;EAC9B,GAAG,MAAM;EACT;EACA;EACA,KAAK,KAAK,IAAI,cAAc,QAAQ;CACtC;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,OAAO,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,GAAG;AAC5C;;;;;;;;;;;;;ACtFA,eAAsB,iBACpB,YACA,cACA,aACA,oBACA,UAAmC,CAAC,GACH;CACjC,MAAM,OAAO,MAAM,WAAW,SAAS;CACvC,MAAM,WAAW,MAAM,aAAa,KAAK,QAAQ,aAAa;CAC9D,MAAM,gCAAgB,IAAI,IAAiC;CAC3D,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,MAAM,cAAc,IAAI,EAAE,KAAK,KAAK,CAAC;EAC3C,IAAI,KAAK,CAAC;EACV,cAAc,IAAI,EAAE,OAAO,GAAG;CAChC;CAEA,MAAM,YAAY,QAAQ,aAAa;CACvC,MAAM,SAAS,QAAQ,mBAAmB;CAE1C,MAAM,QAA0F,CAAC;CACjG,KAAK,MAAM,MAAM,aACf,KAAK,MAAM,MAAM,oBACf,MAAM,KAAK;EAAE,YAAY,GAAG;EAAI,eAAe;EAAI,IAAI,CAAC;EAAG,IAAI,CAAC;CAAE,CAAC;CAIvE,IAAI,SAAS;CACb,IAAI,UAAU;CACd,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,KAAK,cAAc,IAAI,IAAI,KAAK;EACtC,IAAI,CAAC,MAAM,GAAG,WAAW,GAAG;GAC1B;GACA;EACF;EACA,MAAM,WAAW,GAAG,QAAQ,MAAM,EAAE,aAAa,IAAI,aAAa,MAAM;EACxE,IAAI,SAAS,WAAW,GAAG;GACzB;GACA;EACF;EAEA,KAAK,MAAM,MAAM,aAAa;GAE5B,MAAM,IAAI,OADM,GAAG,WAAW,mBAAmB,GAAG,EAAE,EAAA,CAC9B,KAAK,UAAU;GACvC,IAAI,MAAM,QAAQ,CAAC,OAAO,SAAS,CAAC,GAAG;GAEvC,KAAK,MAAM,MAAM,oBAAoB;IACnC,MAAM,SAAS,SACZ,KAAK,MAAM,EAAE,QAAQ,GAAG,CAAC,CACzB,QAAQ,MAAmB,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,CAAC;IACzE,IAAI,OAAO,WAAW,GAAG;IACzB,MAAM,IAAI,OAAO,QAAQ,WAAW,QAAQ;IAC5C,IAAI,MAAM,MAAM;IAChB,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,eAAe,GAAG,MAAM,EAAE,kBAAkB,EAAE;IAC/E,KAAK,GAAG,KAAK,CAAC;IACd,KAAK,GAAG,KAAK,CAAC;GAChB;EACF;EACA;CACF;CA0BA,OAAO;EAAE,OAxB4B,MAClC,QAAQ,MAAM,EAAE,GAAG,UAAU,CAAC,CAAC,CAC/B,KAAK,MAAM;GACV,MAAM,UAAU,SAAS,EAAE,IAAI,EAAE,EAAE;GACnC,MAAM,WAAW,UAAU,EAAE,IAAI,EAAE,EAAE;GACrC,MAAM,cAAc,mBAClB,EAAE,IACF,EAAE,IACF,QAAQ,uBAAuB,KAC/B,QAAQ,IACV;GACA,MAAM,UACJ,KAAK,IAAI,OAAO,KAAK,KAAM,WAAW,KAAK,IAAI,OAAO,KAAK,KAAM,aAAa;GAChF,OAAO;IACL,YAAY,EAAE;IACd,eAAe,EAAE;IACjB,GAAG,EAAE,GAAG;IACR;IACA;IACA;IACA;GACF;EACF,CAEoB;EAAG,eAAe;EAAQ,aAAa;CAAQ;AACvE;AAIA,SAAS,OACP,QACA,MACA,UACe;CACf,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,IAAI,SAAS,QAAQ,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;CACvE,IAAI,SAAS,OAAO,OAAO,KAAK,IAAI,GAAG,MAAM;CAE7C,MAAM,SAAS,CAAC,GAAG,QAAQ,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,aAAa,EAAE,UAAU,CAAC,CAAC;CACzE,IAAI,CAAC,QAAQ,OAAO;CACpB,MAAM,YAAY,OAAO,KAAK,OAAO,OAAO,CAAC,CAAC;CAC9C,MAAM,IAAI,cAAc,KAAA,IAAY,OAAO,QAAQ,aAAa,KAAA;CAEhE,MAAM,SAAS,SACZ,KAAK,MAAM;EACV,MAAM,IAAI,OAAO,KAAK,EAAE,OAAO,CAAC,CAAC;EACjC,OAAO;GACL,IAAI,EAAE;GACN,GAAG,MAAM,KAAA,IAAY,OAAO,MAAM,MAAM,EAAE,QAAQ,OAAO,CAAC,IAAI,KAAA;EAChE;CACF,CAAC,CAAC,CACD,QAAQ,MAAM,EAAE,MAAM,KAAA,CAAS;CAClC,IAAI,OAAO,WAAW,GAAG,OAAO,KAAK;CACrC,OAAO,OAAO,MAAM,GAAG,MAAM,EAAE,KAAK,EAAE,EAAE,CAAC,CAAC,EAAE,EAAE,KAAK;AACrD;AAEA,SAAS,mBACP,IACA,IACA,YACA,MACkC;CAClC,MAAM,IAAI,GAAG;CACb,IAAI,IAAI,GAAG,OAAO;EAAE,OAAO;EAAK,OAAO;CAAI;CAC3C,MAAM,MAAM,QAAQ,MAAM,IAAI,EAAE;CAChC,MAAM,KAAe,CAAC;CACtB,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,KAAe,IAAI,MAAM,CAAC;EAChC,MAAM,KAAe,IAAI,MAAM,CAAC;EAChC,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;GAC1B,MAAM,MAAM,KAAK,MAAM,IAAI,IAAI,CAAC;GAChC,GAAG,KAAK,GAAG;GACX,GAAG,KAAK,GAAG;EACb;EACA,MAAM,IAAI,SAAS,IAAI,EAAE;EACzB,IAAI,OAAO,SAAS,CAAC,GAAG,GAAG,KAAK,CAAC;CACnC;CACA,GAAG,MAAM,GAAG,MAAM,IAAI,CAAC;CACvB,IAAI,GAAG,WAAW,GAAG,OAAO;EAAE,OAAO;EAAK,OAAO;CAAI;CACrD,OAAO;EACL,OAAO,GAAG,KAAK,MAAM,OAAQ,GAAG,MAAM;EACtC,OAAO,GAAG,KAAK,IAAI,GAAG,SAAS,GAAG,KAAK,MAAM,OAAQ,GAAG,MAAM,CAAC;CACjE;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AC9IA,MAAM,cAAoC;CACxC;CACA;CACA;CACA;AACF;AAKA,MAAM,qBAAkD,CAAC,UAAU,QAAQ;;AAG3E,MAAM,wBAAwB;;;;;;;;;;;;;AAwB9B,SAAgB,YAAY,OAKlB;CACR,MAAM,EAAE,IAAI,MAAM,MAAM,oBAAoB;CAC5C,IAAI,OAAO,OAAO,YAAY,GAAG,KAAK,MAAM,IAC1C,MAAM,IAAI,gBAAgB,4CAA4C;CAExE,IAAI,CAAC,YAAY,SAAS,IAAI,GAC5B,MAAM,IAAI,gBACR,uBAAuB,GAAG,aAAa,KAAK,UAAU,IAAI,EAAE,oBAAoB,YAAY,KAAK,IAAI,GACvG;CAEF,IAAI,CAAC,mBAAmB,SAAS,eAAe,GAC9C,MAAM,IAAI,gBACR,uBAAuB,GAAG,wBAAwB,KAAK,UAAU,eAAe,EAAE,oBAAoB,mBAAmB,KAAK,IAAI,GACpI;CAEF,IAAI,OAAO,KAAK,WAAW,YAAY,KAAK,OAAO,KAAK,MAAM,IAC5D,MAAM,IAAI,gBAAgB,uBAAuB,GAAG,2BAA2B;CAEjF,IAAI,CAAC,OAAO,SAAS,KAAK,UAAU,KAAK,KAAK,aAAa,KAAK,KAAK,aAAa,GAChF,MAAM,IAAI,gBACR,uBAAuB,GAAG,mBAAmB,KAAK,WAAW,qCAC/D;CAEF,IAAI,KAAK,UAAU,KAAA,KAAa,OAAO,KAAK,UAAU,UACpD,MAAM,IAAI,gBAAgB,uBAAuB,GAAG,8BAA8B;CAEpF,IAAI,KAAK,eAAe,uBACtB,MAAM,IAAI,gBACR,uBAAuB,GAAG,mBAAmB,sBAAsB,qDACrE;CAEF,MAAM,YAA8B,KAAK,aAAa,wBAAwB,WAAW;CACzF,IAAI,cAAc,iBAChB,MAAM,IAAI,gBACR,uBAAuB,GAAG,0BAA0B,gBAAgB,sBAAsB,KAAK,WAAW,QAAQ,WACpH;CAEF,OAAO;EACL;EACA;EACA,MAAM;GACJ,QAAQ,KAAK;GACb,YAAY,KAAK;GACjB,GAAI,KAAK,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,KAAK,MAAM;EAC1D;EACA;CACF;AACF;;;;;;;;;;AAoDA,SAAgB,WACd,SACA,QACA,UAA6B,CAAC,GACZ;CAClB,MAAM,OAAO,QAAQ,QAAQ;CAC7B,MAAM,kBAAkB,QAAQ,mBAAmB;CACnD,IAAI,CAAC,OAAO,SAAS,IAAI,GACvB,MAAM,IAAI,gBAAgB,iDAAiD,MAAM;CAEnF,IAAI,CAAC,OAAO,SAAS,eAAe,KAAK,mBAAmB,KAAK,kBAAkB,GACjF,MAAM,IAAI,gBACR,sEAAsE,iBACxE;CAGF,MAAM,eAA6B,CAAC;CACpC,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,QAAQ,SAAS;EAC1B,IAAI,OAAO,KAAK,WAAW,YAAY,KAAK,OAAO,KAAK,MAAM,IAC5D,MAAM,IAAI,gBAAgB,gDAAgD;EAE5E,IAAI,CAAC,OAAO,SAAS,KAAK,UAAU,GAClC,MAAM,IAAI,gBACR,6BAA6B,KAAK,OAAO,8BAC3C;EAEF,IAAI,KAAK,IAAI,KAAK,MAAM,GACtB,MAAM,IAAI,gBAAgB,yCAAyC,KAAK,OAAO,EAAE;EAEnF,KAAK,IAAI,KAAK,MAAM;EACpB,aAAa,KAAK;GAChB,QAAQ,KAAK;GACb,YAAY,KAAK;GACjB,GAAI,KAAK,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,KAAK,MAAM;EAC1D,CAAC;CACH;CAEA,MAAM,eAAwB,CAAC;CAC/B,MAAM,2BAAW,IAAI,IAAY;CACjC,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,SAAS,YAAY,KAAK;EAChC,IAAI,SAAS,IAAI,OAAO,EAAE,GACxB,MAAM,IAAI,gBAAgB,mCAAmC,OAAO,GAAG,EAAE;EAE3E,IAAI,KAAK,IAAI,OAAO,KAAK,MAAM,GAC7B,MAAM,IAAI,gBACR,sBAAsB,OAAO,GAAG,mBAAmB,OAAO,KAAK,OAAO,+BACxE;EAEF,MAAM,gBACJ,OAAO,KAAK,cAAc,kBAAkB,WAAW;EACzD,IAAI,kBAAkB,OAAO,iBAC3B,MAAM,IAAI,gBACR,sBAAsB,OAAO,GAAG,0BAA0B,OAAO,gBAAgB,sBAAsB,OAAO,KAAK,WAAW,aAAa,cAAc,2BAA2B,iBACtL;EAEF,SAAS,IAAI,OAAO,EAAE;EACtB,KAAK,IAAI,OAAO,KAAK,MAAM;EAC3B,aAAa,KAAK,MAAM;CAC1B;CAEA,MAAM,QAAQ,SACZ,CAAC,GAAG,cAAc,GAAG,aAAa,KAAK,UAAU,MAAM,IAAI,CAAC,GAC5D,WAAW,IAAI,CACjB;CACA,MAAM,UAAU,MAAM,KAAK,SAAS,KAAK,MAAM;CAQ/C,OAAO;EAAE;EAAO,UAAA;GANd,MAAM,aAAa;IAAE;IAAM;IAAiB;IAAS,QAAQ;GAAa,CAAC;GAC3E;GACA;GACA;GACA,QAAQ;EAEa;CAAE;AAC3B;;;;;;;AAQA,SAAS,SAAY,OAAqB,QAA2B;CACnE,OAAO,MACJ,KAAK,UAAU;EAAE;EAAM,KAAK,OAAO;CAAE,EAAE,CAAC,CACxC,MAAM,MAAM,UAAU,KAAK,MAAM,MAAM,GAAG,CAAC,CAC3C,KAAK,UAAU,MAAM,IAAI;AAC9B;AAEA,SAAS,aAAa,UAAmD;CACvE,OAAO,cAAc;EACnB,QAAQ;EACR,MAAM,SAAS;EACf,iBAAiB,SAAS;EAC1B,SAAS,SAAS;EAClB,QAAQ,SAAS;CACnB,CAAC;AACH;;;;;;;;;AAwEA,SAAgB,UACd,SACA,UACiB;CACjB,MAAM,eAAe,aAAa;EAChC,MAAM,SAAS;EACf,iBAAiB,SAAS;EAC1B,SAAS,SAAS;EAClB,QAAQ,SAAS;CACnB,CAAC;CACD,IAAI,iBAAiB,SAAS,MAC5B,MAAM,IAAI,sBACR,wCAAwC,aAAa,iCAAiC,SAAS,KAAK,6CACtG;CAGF,MAAM,YAAY,IAAI,IAAI,SAAS,OAAO;CAC1C,MAAM,gCAAgB,IAAI,IAA2B;CACrD,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,UAAU,IAAI,OAAO,MAAM,GAC9B,MAAM,IAAI,gBACR,0BAA0B,OAAO,OAAO,wCAC1C;EAEF,IAAI,cAAc,IAAI,OAAO,MAAM,GACjC,MAAM,IAAI,gBAAgB,oCAAoC,OAAO,OAAO,EAAE;EAEhF,IAAI,OAAO,UAAU,QAAQ,CAAC,OAAO,SAAS,OAAO,KAAK,GACxD,MAAM,IAAI,gBACR,0BAA0B,OAAO,OAAO,cAAc,OAAO,MAAM,mCACrE;EAEF,cAAc,IAAI,OAAO,QAAQ,OAAO,KAAK;CAC/C;CAEA,MAAM,SAAsD,CAAC;CAC7D,MAAM,YAAsB,CAAC;CAC7B,MAAM,aAAuB,CAAC;CAC9B,IAAI,SAAS;CACb,IAAI,SAAS;CACb,IAAI,aAAa;CAEjB,KAAK,MAAM,SAAS,SAAS,QAAQ;EACnC,IAAI,SAAS,OAAO,MAAM;EAC1B,IAAI,WAAW,KAAA,GAAW;GACxB,SAAS;IAAE,QAAQ;IAAG,QAAQ;IAAG,QAAQ;IAAG,YAAY;IAAG,MAAM;GAAK;GACtE,OAAO,MAAM,QAAQ;EACvB;EACA,OAAO,UAAU;EACjB,IAAI,CAAC,cAAc,IAAI,MAAM,KAAK,MAAM,GAAG;GACzC,WAAW,KAAK,MAAM,EAAE;GACxB;EACF;EACA,MAAM,QAAQ,cAAc,IAAI,MAAM,KAAK,MAAM,KAAK;EACtD,IAAI,UAAU,MAAM;GAClB,cAAc;GACd,OAAO,cAAc;GACrB;EACF;EAEA,KADqC,SAAS,SAAS,kBAAkB,WAAW,cACjE,MAAM,iBAAiB;GACxC,UAAU;GACV,OAAO,UAAU;EACnB,OAAO;GACL,UAAU;GACV,OAAO,UAAU;GACjB,UAAU,KAAK,MAAM,EAAE;EACzB;CACF;CAEA,MAAM,SAAS,SAAS,OAAO;CAC/B,MAAM,UAAU,SAAS;CACzB,MAAM,SAA0B;EAC9B,QAAQ;EACR;EACA;EACA;EACA;EACA,MAAM;EACN;EACA;EACA;EACA,UAAU,kBAAkB,UAAU,aAAa;CACrD;CAEA,IAAI,WAAW,GACb,OAAO;EACL,GAAG;EACH,QAAQ;EACR,QAAQ;CACV;CAEF,IAAI,WAAW,SAAS,GACtB,OAAO;EACL,GAAG;EACH,QAAQ;EACR,QAAQ,GAAG,WAAW,OAAO,MAAM,OAAO,iCAAiC,WAAW,KAAK,IAAI;CACjG;CAEF,IAAI,YAAY,GACd,OAAO;EACL,GAAG;EACH,QAAQ;EACR,QAAQ,OAAO,OAAO;CACxB;CAGF,KAAK,MAAM,UAAU,OAAO,OAAO,MAAM,GAAG;EAC1C,MAAM,cAAc,OAAO,SAAS,OAAO;EAC3C,OAAO,OAAO,gBAAgB,IAAI,OAAO,OAAO,SAAS;CAC3D;CACA,OAAO,OAAO,SAAS;CACvB,OAAO;AACT;AAEA,SAAS,kBACP,UACA,eACmB;CACnB,MAAM,eAAe,IAAI,IAAI,SAAS,OAAO,KAAK,UAAU,MAAM,KAAK,MAAM,CAAC;CAC9E,IAAI,IAAI;CACR,IAAI,UAAU;CACd,IAAI,WAAW;CACf,KAAK,MAAM,UAAU,SAAS,SAAS;EACrC,IAAI,aAAa,IAAI,MAAM,GAAG;EAC9B,KAAK;EACL,MAAM,QAAQ,cAAc,IAAI,MAAM;EACtC,IAAI,UAAU,KAAA,KAAa,UAAU,MAAM;EAC3C,WAAW;EACX,IAAI,QAAQ,SAAS,iBAAiB,YAAY;CACpD;CACA,OAAO;EAAE;EAAG;EAAS;EAAU,eAAe,YAAY,IAAI,OAAO,WAAW;CAAQ;AAC1F;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACrcA,MAAM,wBAAwB;CAAC;CAAO;CAAoB;AAAkB;AAmC5E,SAAS,SAAS,OAAe,OAAuB;CACtD,MAAM,KAAK,OAAO,UAAU,WAAW,KAAK,MAAM,KAAK,IAAI;CAC3D,IAAI,CAAC,OAAO,SAAS,EAAE,GACrB,MAAM,IAAI,gBACR,mBAAmB,MAAM,qCAAqC,KAAK,UAAU,KAAK,GACpF;CAEF,OAAO;AACT;;AAGA,SAAgB,yBAAyB,UAA4B,SAAS,YAAkB;CAC9F,SAAS,SAAS,IAAI,GAAG,OAAO,IAAI;CACpC,IAAI,OAAO,SAAS,YAAY,YAAY,SAAS,QAAQ,WAAW,GACtE,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,oCAAoC;CAE1F,IAAI,OAAO,SAAS,eAAe,YAAY,SAAS,WAAW,WAAW,GAC5E,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,uCAAuC;CAE7F,IAAI,SAAS,YAAY,QAAQ,OAAO,SAAS,YAAY,UAC3D,MAAM,IAAI,gBAAgB,mBAAmB,OAAO,0CAA0C;CAEhG,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,SAAS,OAAO,GAAG;EAC3D,IAAI,CAAE,sBAA4C,SAAS,GAAG,GAE5D,MAAM,IAAI,gBACR,mBAAmB,OAAO,gCAAgC,IAAI,aAAa,sBAAsB,KAAK,IAAI,GAC5G;EAEF,IAAI,UAAU,KAAA,KAAa,CAAC,OAAO,SAAS,KAAK,GAC/C,MAAM,IAAI,gBACR,mBAAmB,OAAO,WAAW,IAAI,2BAA2B,OAAO,KAAK,GAClF;CAEJ;AACF;;;;;;AASA,SAAgB,wBACd,QACA,MACkB;CAClB,MAAM,QAAQ,6BAA6B,SAAS,OAAO,0BAA0B,OAAO;CAC5F,IAAI,CAAC,OAAO,SAAS,KAAK,GACxB,MAAM,IAAI,gBACR,+CAA+C,OAAO,EAAE,iEAC1D;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,kBAAkB,MAAM;CAAE;CACnF,yBAAyB,QAAQ;CACjC,OAAO;AACT;;;;;;;;AASA,SAAgB,sBACd,QACA,MACkB;CAClB,MAAM,MAAM,gBAAgB,SAAS,OAAO,aAAa,OAAO;CAChE,IAAI,CAAC,OAAO,SAAS,GAAG,GACtB,MAAM,IAAI,gBACR,4FACF;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,IAAI;CAAE;CAC/D,yBAAyB,QAAQ;CACjC,OAAO;AACT;;;;;;;;AAcA,SAAgB,wBACd,QACA,QACA,MACA,UAA8B,CAAC,GACb;CAClB,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,SAAS,SAAS,KAAK,YAAY,GAC7C,MAAM,IAAI,gBACR,uEAAuE,WACzE;CAEF,MAAM,6BAAa,IAAI,IAAoB;CAC3C,KAAK,MAAM,QAAQ,QAAQ;EACzB,IAAI,CAAC,OAAO,SAAS,KAAK,UAAU,GAClC,MAAM,IAAI,gBACR,yCAAyC,KAAK,OAAO,8BACvD;EAEF,IAAI,WAAW,IAAI,KAAK,MAAM,GAC5B,MAAM,IAAI,gBAAgB,qDAAqD,KAAK,OAAO,EAAE;EAE/F,WAAW,IAAI,KAAK,QAAQ,KAAK,UAAU;CAC7C;CACA,IAAI,SAAS;CACb,IAAI,SAAS;CACb,KAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,QAAQ,WAAW,IAAI,EAAE,MAAM;EACrC,IAAI,UAAU,KAAA,GAAW;EACzB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAC1B,MAAM,IAAI,gBACR,kDAAkD,EAAE,OAAO,gBAC7D;EAEF;EACA,IAAI,KAAK,IAAI,EAAE,QAAQ,KAAK,KAAK,WAAW;CAC9C;CACA,IAAI,WAAW,GACb,MAAM,IAAI,gBACR,yGACF;CAEF,MAAM,WAA6B;EAAE,GAAG;EAAM,SAAS,EAAE,kBAAkB,SAAS,OAAO;CAAE;CAC7F,yBAAyB,QAAQ;CACjC,OAAO;AACT;AAUA,SAAgB,sBAAsB,UAA8B,CAAC,GAAkB;CACrF,KAAK,MAAM,KAAK,SAAS,yBAAyB,CAAC;CACnD,MAAM,QAAQ,QAAQ,KAAK,OAAO;EAAE,GAAG;EAAG,SAAS,EAAE,GAAG,EAAE,QAAQ;CAAE,EAAE;CACtE,OAAO;EACL,MAAM,OAAO,UAAU;GACrB,yBAAyB,QAAQ;GACjC,MAAM,KAAK;IAAE,GAAG;IAAU,SAAS,EAAE,GAAG,SAAS,QAAQ;GAAE,CAAC;EAC9D;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,MAAM,MAAM,KAAK,OAAO;IAAE,GAAG;IAAG,SAAS,EAAE,GAAG,EAAE,QAAQ;GAAE,EAAE;GAClE,OAAO,YAAY,KAAA,IAAY,MAAM,IAAI,QAAQ,MAAM,EAAE,YAAY,OAAO;EAC9E;CACF;AACF;;;;;;;AAQA,SAAgB,kBAAkB,MAA6B;CAC7D,OAAO;EACL,MAAM,OAAO,UAAU;GACrB,yBAAyB,QAAQ;GACjC,MAAM,KAAK,MAAM,OAAO;GACxB,MAAM,UAAU,MAAM,OAAO;GAC7B,MAAM,GAAG,MAAM,QAAQ,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;GACzD,MAAM,GAAG,WAAW,MAAM,GAAG,KAAK,UAAU,QAAQ,EAAE,KAAK,MAAM;EACnE;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,KAAK,MAAM,OAAO;GACxB,IAAI;GACJ,IAAI;IACF,MAAM,MAAM,GAAG,SAAS,MAAM,MAAM;GACtC,SAAS,KAAK;IACZ,IAAK,IAA8B,SAAS,UAAU,OAAO,CAAC;IAC9D,MAAM;GACR;GACA,MAAM,QAAQ,IAAI,MAAM,IAAI;GAC5B,MAAM,YAAgC,CAAC;GACvC,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;IACrC,MAAM,OAAO,MAAM;IACnB,IAAI,KAAK,KAAK,CAAC,CAAC,WAAW,GAAG;IAC9B,IAAI;IACJ,IAAI;KACF,SAAS,KAAK,MAAM,IAAI;IAC1B,SAAS,KAAK;KACZ,MAAM,IAAI,gBACR,4BAA4B,KAAK,4BAA4B,IAAI,EAAE,IAAK,IAAc,SACxF;IACF;IACA,MAAM,WAAW;IACjB,yBAAyB,UAAU,GAAG,KAAK,GAAG,IAAI,GAAG;IACrD,UAAU,KAAK,QAAQ;GACzB;GACA,OAAO,YAAY,KAAA,IAAY,YAAY,UAAU,QAAQ,MAAM,EAAE,YAAY,OAAO;EAC1F;CACF;AACF;AAmDA,MAAM,aAAa;AAEnB,SAAgB,oBACd,SACA,MACgB;CAChB,MAAM,SAAS,SAAS,KAAK,MAAM,MAAM;CACzC,MAAM,SAAS,KAAK,YAAY,UAAU;CAC1C,MAAM,eAAe,KAAK,YAAY,gBAAgB;CACtD,MAAM,kBAAkB,KAAK,YAAY,mBAAmB;CAC5D,MAAM,iBAAiB,KAAK,YAAY,kBAAkB;CAE1D,MAAM,0BAAU,IAAI,IAAwD;CAC5E,KAAK,MAAM,YAAY,SAAS;EAC9B,yBAAyB,QAAQ;EACjC,MAAM,OAAO,SAAS,SAAS,IAAI,uBAAuB,SAAS,QAAQ,GAAG;EAC9E,MAAM,MAAM,QAAQ,IAAI,SAAS,OAAO,KAAK,CAAC;EAC9C,IAAI,KAAK;GAAE,GAAG;GAAU;EAAK,CAAC;EAC9B,QAAQ,IAAI,SAAS,SAAS,GAAG;CACnC;CAEA,MAAM,WAA4B,CAAC;CACnC,MAAM,SAAmB,CAAC;CAC1B,MAAM,sBAAgC,CAAC;CAEvC,KAAK,MAAM,WAAW,CAAC,GAAG,QAAQ,KAAK,CAAC,CAAC,CAAC,KAAK,GAAG;EAChD,MAAM,QAAQ,QAAQ,IAAI,OAAO,CAAC,CAAE,MAAM,GAAG,MAAM,EAAE,OAAO,EAAE,IAAI;EAClE,MAAM,SAAS,MAAM,MAAM,SAAS;EAEpC,MAAM,WAAW,SAAS,OAAO,QAAQ;EACzC,IAAI,UAAU,gBACZ,OAAO,KACL,UAAU,QAAQ,6BAA6B,OAAO,GAAG,MAAM,QAAQ,QAAQ,CAAC,EAAE,uBAAuB,eAAe,GAC1H;EAMF,IAAI,YAAY;EAChB,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAChC,IAAI,MAAM,EAAE,CAAE,eAAe,MAAM,IAAI,EAAE,CAAE,YAAY,YAAY;EAErE,IAAI,aAAa,GAQX;OAAA,CAPiB,MAClB,MAAM,SAAS,CAAC,CAChB,MACE,MACC,OAAO,SAAS,EAAE,QAAQ,gBAAgB,KAC1C,OAAO,SAAS,EAAE,QAAQ,gBAAgB,CAEhC,GACd,OAAO,KACL,UAAU,QAAQ,oBAAoB,MAAM,YAAY,EAAE,CAAE,WAAW,OAAO,MAAM,UAAU,CAAE,WAAW,OAAO,MAAM,UAAU,CAAE,GAAG,gIACzI;EAAA;EAIJ,KAAK,MAAM,UAAU,uBAAuB;GAC1C,MAAM,SAAS,MACZ,QAAQ,MAAM,OAAO,EAAE,QAAQ,YAAY,QAAQ,CAAC,CACpD,KAAK,MAAM,EAAE,QAAQ,OAAQ;GAChC,IAAI,OAAO,WAAW,GAAG;GAEzB,MAAM,cAAc,cAAc,QAAQ,KAAK,WAAW;GAC1D,IAAI,YAAY,UAAU,qBACxB,oBAAoB,KAAK,GAAG,QAAQ,GAAG,QAAQ;GAEjD,MAAM,UAAU,OAAO,OAAO,SAAS;GACvC,MAAM,WAAW,OAAO;GACxB,MAAM,QAAQ,UAAU;GACxB,MAAM,UAAoB,CAAC;GAC3B,IAAI,YAAY,UAAU,iBACxB,QAAQ,KACN,6CAA6C,YAAY,QAAQ,gBAAgB,YAAY,WAAW,QAAQ,CAAC,EAAE,EACrH;GAEF,IAAI,WAAW,SAAS,UAAU,QAChC,QAAQ,KAAK,OAAO,QAAQ,QAAQ,CAAC,EAAE,eAAe,QAAQ;GAEhE,IAAI,WAAW,sBAAsB,WAAW,UAAU,cACxD,QAAQ,KACN,6BAA6B,WAAW,QAAA,CAAS,QAAQ,CAAC,EAAE,iBAAiB,SAAS,QAAQ,CAAC,EAAE,UAAU,aAAa,EAC1H;GAEF,IAAI,WAAW,sBAAsB,WAAW,UAAU,iBACxD,QAAQ,KACN,6BAA6B,WAAW,QAAA,CAAS,QAAQ,CAAC,EAAE,iBAAiB,SAAS,QAAQ,CAAC,EAAE,UAAU,gBAAgB,EAC7H;GAGF,MAAM,UAAU,QAAQ,SAAS;GACjC,MAAM,QAAuB;IAC3B;IACA;IACA,OAAO,YAAY;IACnB;IACA;IACA;IACA;GACF;GACA,IAAI,SAAS;IACX,MAAM,SAAS,QAAQ,KAAK,IAAI;IAChC,OAAO,KAAK,UAAU,QAAQ,IAAI,OAAO,IAAI,MAAM,QAAQ;GAC7D;GACA,SAAS,KAAK,KAAK;EACrB;CACF;CAEA,OAAO;EAAE;EAAU;EAAQ,SAAS,OAAO,WAAW;EAAG;CAAoB;AAC/E;;;;;;;;;AAiBA,SAAgB,gBAAgB,QAAyC;CACvE,OAAO;EAAE,SAAS,OAAO;EAAS,QAAQ,CAAC,GAAG,OAAO,MAAM;CAAE;AAC/D"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { E as MultishotResult, M as MultishotTransportRequest, T as MultishotPersona, c as RunMultishotMatrixResult, f as RunMultishotOptions, k as MultishotToolDefinition, s as RunMultishotMatrixOptions } from "../../matrix-
|
|
1
|
+
import { E as MultishotResult, M as MultishotTransportRequest, T as MultishotPersona, c as RunMultishotMatrixResult, f as RunMultishotOptions, k as MultishotToolDefinition, s as RunMultishotMatrixOptions } from "../../matrix-Ch8JO1pG.js";
|
|
2
2
|
//#region src/multishot/golden/compare.d.ts
|
|
3
3
|
interface CompareOptions {
|
|
4
4
|
/** Stop after this many mismatches. A structural divergence high in the
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { w as JudgeScore } from "../types-
|
|
2
|
-
import { A as MultishotToolExecutor, C as MultishotFatalToolError, D as MultishotShape, E as MultishotResult, F as assertMultishotShotResult, M as MultishotTransportRequest, N as MultishotTransportResponse, O as MultishotShotResultError, P as MultishotTransportToolCall, S as MultishotDriverEmptyError, T as MultishotPersona, _ as JudgeRunResult, a as MultishotCellOutput, b as runJudge, c as RunMultishotMatrixResult, d as MultishotShot, f as RunMultishotOptions, g as JudgeDimension, h as JudgeConfig, i as ConversationJudgeInput, j as MultishotTransport, k as MultishotToolDefinition, l as computeCellComposite, m as DEFAULT_JUDGE_MODEL, n as CellCompositeInput, o as MultishotJudges, p as runMultishot, r as CellCompositeScore, s as RunMultishotMatrixOptions, t as ArtifactJudgeInput, u as runMultishotMatrix, v as renderDimensions, w as MultishotMessage, x as MultishotArtifact, y as renderJsonFooter } from "../matrix-
|
|
1
|
+
import { w as JudgeScore } from "../types-nokrtr7M.js";
|
|
2
|
+
import { A as MultishotToolExecutor, C as MultishotFatalToolError, D as MultishotShape, E as MultishotResult, F as assertMultishotShotResult, M as MultishotTransportRequest, N as MultishotTransportResponse, O as MultishotShotResultError, P as MultishotTransportToolCall, S as MultishotDriverEmptyError, T as MultishotPersona, _ as JudgeRunResult, a as MultishotCellOutput, b as runJudge, c as RunMultishotMatrixResult, d as MultishotShot, f as RunMultishotOptions, g as JudgeDimension, h as JudgeConfig, i as ConversationJudgeInput, j as MultishotTransport, k as MultishotToolDefinition, l as computeCellComposite, m as DEFAULT_JUDGE_MODEL, n as CellCompositeInput, o as MultishotJudges, p as runMultishot, r as CellCompositeScore, s as RunMultishotMatrixOptions, t as ArtifactJudgeInput, u as runMultishotMatrix, v as renderDimensions, w as MultishotMessage, x as MultishotArtifact, y as renderJsonFooter } from "../matrix-Ch8JO1pG.js";
|
|
3
3
|
import { AgentProfile } from "@tangle-network/agent-interface";
|
|
4
4
|
//#region src/multishot/cost.d.ts
|
|
5
5
|
/**
|
package/dist/openapi.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"openapi": "3.1.0",
|
|
3
3
|
"info": {
|
|
4
4
|
"title": "@tangle-network/agent-eval — wire protocol",
|
|
5
|
-
"version": "0.
|
|
5
|
+
"version": "0.172.0",
|
|
6
6
|
"description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.1.0. Bumps on breaking changes to request/response schemas.",
|
|
7
7
|
"contact": {
|
|
8
8
|
"name": "Tangle Network",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
2
|
import { r as canonicalString } from "./canonical-DPyQ_rpt.js";
|
|
3
|
-
import { J as recoverTruncatedJson, n as JudgeParseError } from "./llm-judge-
|
|
3
|
+
import { J as recoverTruncatedJson, n as JudgeParseError } from "./llm-judge-aQHIk5_-.js";
|
|
4
4
|
import { i as CostLedger } from "./cost-ledger-B1qx30B4.js";
|
|
5
5
|
import { t as certificationEvidenceDigest } from "./verdict-BQ3pCFf8.js";
|
|
6
6
|
import { f as ModelSubstitutionError, h as assertServedModel, o as costReceiptFromLlm, s as costReceiptFromLlmError, u as maximumChargeForLlmRequest } from "./llm-client-BFMRpmqb.js";
|
|
@@ -583,4 +583,4 @@ function extractProducedState(events) {
|
|
|
583
583
|
//#endregion
|
|
584
584
|
export { CODING_HARNESSES as a, agentProfileId as c, harnessAxisOf as d, verifyCompletion as i, agentProfileModelId as l, completionVerdict as n, HARNESS_NATIVE_MODEL as o, createLlmCorrectnessChecker as r, agentProfileHash as s, extractProducedState as t, expandProfileAxes as u };
|
|
585
585
|
|
|
586
|
-
//# sourceMappingURL=produced-state-
|
|
586
|
+
//# sourceMappingURL=produced-state-CxmbFxFd.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"produced-state-Cm6DU_Ao.js","names":[],"sources":["../src/agent-profile.ts","../src/completion-verifier.ts","../src/produced-state.ts"],"sourcesContent":["import { createHash } from 'node:crypto'\nimport type { AgentProfile } from '@tangle-network/agent-interface'\nimport { type HarnessType, harnessSupportsModel } from '@tangle-network/agent-interface'\nimport { ValidationError } from './errors'\nimport { canonicalString } from './ledger-core/canonical'\n\nexport type { AgentProfile, HarnessType } from '@tangle-network/agent-interface'\n\n/**\n * The agentic coding harnesses an eval sweeps by default — the ones we care about\n * ranking. This is the SINGLE source of that list; consumers import it instead of\n * re-declaring their own (a re-declared list is how the fleet drifts). Pass an\n * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known\n * harness) to widen beyond these.\n */\nexport const CODING_HARNESSES: readonly HarnessType[] = [\n 'opencode',\n 'claude-code',\n 'codex',\n 'kimi-code',\n]\n\nexport interface ProfileAxisSpec {\n /** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the\n * harness and model vary. `model.default` is the fallback model. */\n base: AgentProfile\n /** Harnesses to cross. Default: {@link CODING_HARNESSES}. */\n harnesses?: readonly HarnessType[]\n /** Models to cross. Default: `[base.model.default]` — one model, i.e. today's\n * single-model behaviour, so omitting this never changes an existing run. */\n models?: readonly string[]\n /** Force every (harness, model) pair verbatim, even ones the harness can't run —\n * for deliberately testing failure modes. Default (false): SNAP instead — a\n * vendor-locked harness runs only the swept models in its family, or its native\n * default when it supports none, so no harness is dropped and none gets a\n * guaranteed-failing foreign-model cell. */\n keepIncompatible?: boolean\n}\n\n/** Model sentinel for a vendor-locked harness that supports none of the swept models:\n * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness\n * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi\n * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship\n * table that would rot as router catalogs change. */\nexport const HARNESS_NATIVE_MODEL = 'default'\n\n/**\n * Expand a base profile across the harness × model matrix into the `AgentProfile[]`\n * that `runProfileMatrix` / `selfImprove` score — the ONE place \"which harnesses ×\n * which models do we evaluate\" lives, so no product hand-rolls its own harness list\n * or column→profile mapping (the pattern that let those copies drift and silently\n * break the harness pivot).\n *\n * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,\n * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both\n * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row\n * and results join back by harness/model via {@link harnessAxisOf} with no\n * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or\n * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every\n * requested harness runs; `keepIncompatible` forces every pair verbatim.\n *\n * Omit `harnesses`/`models` to sweep the full default set — the \"turn it on for\n * everything we care about\" switch, identical in shape whether one harness or all.\n */\nexport function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[] {\n const harnesses = spec.harnesses ?? CODING_HARNESSES\n if (harnesses.length === 0) throw new ValidationError('expandProfileAxes: no harnesses to sweep')\n const baseModel = spec.base.model?.default\n const models = spec.models ?? (baseModel ? [baseModel] : [])\n if (models.length === 0) {\n throw new ValidationError(\n 'expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied',\n )\n }\n const out: AgentProfile[] = []\n const seen = new Set<string>()\n for (const harness of harnesses) {\n // A universal (router-backed) harness — opencode/pi/claudish — runs every swept\n // model. A vendor-locked harness — codex/claude-code/kimi-code — runs only the\n // swept models in its own family; when it supports NONE of them it snaps to its\n // native default (the `HARNESS_NATIVE_MODEL` sentinel it resolves at runtime)\n // rather than being dropped, so every requested harness still appears in the\n // sweep on a model it can actually run — e.g. sweeping `deepseek/x` puts opencode\n // on deepseek and kimi-code on its own Kimi model, a real head-to-head.\n // `keepIncompatible` forces every (harness, model) pair verbatim (failure-mode runs).\n const supported = spec.keepIncompatible\n ? models\n : models.filter((model) => harnessSupportsModel(harness, model))\n const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL]\n for (const model of effective) {\n const profile: AgentProfile = {\n ...spec.base,\n harness,\n name: `${spec.base.name ?? 'agent'}/${harness}/${model}`,\n model: { ...spec.base.model, default: model },\n metadata: { ...(spec.base.metadata ?? {}), harness, harnessModel: model },\n }\n const id = agentProfileId(profile)\n if (seen.has(id)) continue\n seen.add(id)\n out.push(profile)\n }\n }\n if (out.length === 0) {\n // Unreachable in normal use — snapping guarantees ≥1 cell per harness — but keep a\n // fail-closed guard so a future refactor can't silently produce an empty sweep.\n throw new ValidationError(\n `expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(', ')}], models=[${models.join(', ')}]).`,\n )\n }\n return out\n}\n\n/**\n * Read the (harness, model) a matrix cell ran under, off a profile or a result row's\n * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile\n * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by\n * this instead of recomputing an id (recomputing the wrong key is what broke the pivot\n * in the hand-rolled copies).\n */\nexport function harnessAxisOf(\n profile: Pick<AgentProfile, 'metadata'>,\n): { harness: HarnessType; model: string } | undefined {\n const m = profile.metadata as Record<string, unknown> | undefined\n const harness = m?.harness\n const model = m?.harnessModel\n if (typeof harness === 'string' && typeof model === 'string') {\n return { harness: harness as HarnessType, model }\n }\n return undefined\n}\n\n/**\n * Collision-resistant, path-safe, human-readable profile id for eval artifacts.\n * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix\n * keys, and directory names where two profiles must not collapse onto one row.\n * The suffix is the first 64 bits of the behaviour hash, enough for ordinary\n * eval matrices while keeping filenames readable.\n */\nexport function agentProfileId(profile: AgentProfile): string {\n const label = pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? 'profile'\n return `${label}-${agentProfileHash(profile).slice(0, 16)}`\n}\n\n/**\n * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete\n * model id because run records reject bare/missing model aliases.\n */\nexport function agentProfileModelId(profile: AgentProfile): string {\n const model = profile.model?.default?.trim()\n if (!model) {\n const label = agentProfileDisplayLabel(profile) ?? 'unnamed profile'\n throw new ValidationError(\n `AgentProfile \"${label}\" has no model.default — cannot record eval run`,\n )\n }\n return model\n}\n\nfunction agentProfileDisplayLabel(profile: AgentProfile): string | undefined {\n return profile.name?.trim() || profile.version?.trim() || undefined\n}\n\nfunction pathSafeProfileLabel(label: string | undefined): string | undefined {\n const safe = label\n ?.trim()\n .replace(/[^A-Za-z0-9._-]+/g, '-')\n .replace(/-+/g, '-')\n .replace(/^-|-$/g, '')\n return safe || undefined\n}\n\nfunction compact<T extends Record<string, unknown>>(input: T): Partial<T> {\n const out: Record<string, unknown> = {}\n for (const [key, value] of Object.entries(input)) {\n if (value !== undefined) out[key] = value\n }\n return out as Partial<T>\n}\n\n/**\n * Deterministic behaviour identity for the canonical\n * `@tangle-network/agent-interface` AgentProfile.\n *\n * `name` and `description` are labels and do not affect the hash. Profile\n * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,\n * and extensions do affect the hash. Resource array order is hash-bearing\n * because mount order can change agent behaviour. Undefined fields are treated\n * as absent; explicit `null` fields remain hash-bearing.\n */\nexport function agentProfileHash(profile: AgentProfile): string {\n const model = agentProfileModelId(profile)\n const behaviour = compact({\n ...profile,\n name: undefined,\n description: undefined,\n tags: profile.tags ? [...profile.tags].sort() : undefined,\n model: compact({ ...profile.model, default: model }),\n })\n return createHash('sha256').update(canonicalString(behaviour)).digest('hex')\n}\n","/**\n * Completion verifier — the task-completion oracle.\n *\n * Answers the only eval question that is not a proxy: did the agent actually\n * COMPLETE the task — produce every required deliverable, persisted and\n * correct — rather than describe what should be done. A fluent transcript\n * that never produces the artifact scores zero here.\n *\n * Per requirement, a two-stage check:\n * 1. Structural — a produced item (vault artifact / approved proposal /\n * tool call) of the right kind is matched against the requirement and\n * carries non-empty content. Deterministic; no LLM.\n * 2. Correctness — only if structurally present AND the matched item\n * carries content, one targeted check decides whether that item\n * actually fulfils the requirement. A hallucinated artifact fails here;\n * an absent one already failed stage 1.\n *\n * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —\n * checker failures — are excluded from the denominator, never scored as\n * zeros). Quality dimensions are meaningless on an incomplete task — callers\n * gate on `fullyComplete` / `completionRate` before scoring quality.\n */\n\nimport { randomUUID } from 'node:crypto'\nimport type { ChatClient, ChatRequest } from './analyst/chat-client'\nimport type { Artifact } from './artifact-validator'\nimport { CostLedger, type CostLedgerHandle } from './cost-ledger'\nimport { assertServedModel, ModelSubstitutionError } from './integrity/served-model'\nimport { recoverTruncatedJson } from './json-recovery'\nimport { JudgeParseError } from './judges'\nimport {\n costReceiptFromLlm,\n costReceiptFromLlmError,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport { packageVersion } from './package-version'\nimport type { RawProviderEvent, RawProviderSink } from './trace/raw-provider-sink'\nimport {\n certificationEvidenceDigest,\n type DefaultVerdict,\n type VerdictCertification,\n} from './verdict'\nimport type { CheckerIdentity, VerificationStrategySource } from './verification-strategy'\n\n/** What kind of produced state can satisfy a requirement structurally. */\nexport type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any'\n\nexport interface CompletionRequirement {\n /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */\n reqId: string\n /** Human-readable description of the required deliverable. */\n title: string\n /** Optional kind/category hint, matched against a produced item's kind. */\n category?: string\n /** What produced state satisfies this requirement. Defaults to 'any'. */\n satisfiedBy?: SatisfiedBy\n}\n\nexport interface TaskGold {\n taskId: string\n requirements: CompletionRequirement[]\n}\n\nexport interface ProducedProposal {\n id: string\n title: string\n status: 'pending' | 'approved' | 'rejected'\n /** Optional persisted body — when present, enables a correctness check. */\n content?: string\n}\n\n/** Everything observable about what a run actually produced. */\nexport interface ProducedState {\n /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */\n artifacts: Artifact[]\n /** Proposals / filings the agent created. */\n proposals: ProducedProposal[]\n /** Names of tools the agent invoked. */\n toolCalls: string[]\n}\n\nexport interface RequirementCheck {\n reqId: string\n title: string\n /** A produced item of the right kind matched the requirement, non-empty. */\n structurallyPresent: boolean\n /**\n * Whether the matched item actually fulfils the requirement. `null` when\n * not structurally present, when the matched item carries no content\n * to assess, or when the correctness check itself failed (`unmeasured`).\n */\n correct: boolean | null\n /** structurallyPresent && !unmeasured && correct !== false. */\n satisfied: boolean\n /**\n * Set when the correctness check itself errored (LLM call failure or an\n * unparseable response after retry). The requirement's fulfilment is\n * UNKNOWN — `correct` stays null, `satisfied` is false, and\n * `completionVerdict` excludes the row from `completionRate`'s\n * denominator. Never folded into a zero: a synthetic zero is\n * indistinguishable from a real failure (see `JudgeParseError`).\n */\n unmeasured?: true\n /** Why the correctness check could not be measured (present iff `unmeasured`). */\n unmeasuredReason?: string\n /** Human-readable evidence for the verdict. */\n evidence: string[]\n}\n\n/** Extends the substrate verdict spine: `valid` = `fullyComplete` and\n * `score` = `completionRate` — derived in `completionVerdict()`, the one\n * place those equalities hold by construction. */\nexport interface CompletionVerdict extends DefaultVerdict {\n taskId: string\n requirements: RequirementCheck[]\n /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */\n completionRate: number\n /** Every measurable requirement satisfied (false when anything is unmeasured). */\n fullyComplete: boolean\n /** Requirements whose correctness check errored — reported, never scored as zero. */\n unmeasuredCount: number\n}\n\n/**\n * Construct a `CompletionVerdict` from the per-requirement checks, deriving\n * `completionRate` / `fullyComplete` and the spine fields (`valid` =\n * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero\n * requirements — a verdict over nothing is a misconfiguration, mirroring\n * `verifyCompletion`'s gold-spec guard.\n */\nexport function completionVerdict(input: {\n taskId: string\n requirements: RequirementCheck[]\n /** What certified the correctness stage, when anything did. Omitted =\n * an uncertified verdict — the honest default for a bare checker. */\n certification?: VerdictCertification\n}): CompletionVerdict {\n if (input.requirements.length === 0) {\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`,\n )\n }\n const measurable = input.requirements.filter((r) => !r.unmeasured)\n const unmeasuredCount = input.requirements.length - measurable.length\n if (measurable.length === 0) {\n // Every check errored: this is an infrastructure failure, not a scored\n // run. A 0-rate verdict here would be a fabricated measurement.\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? 'unknown reason'})`,\n )\n }\n const satisfiedCount = measurable.filter((r) => r.satisfied).length\n const completionRate = satisfiedCount / measurable.length\n // A run with unmeasured rows can still report a rate over what WAS\n // measured, but must not claim full completion.\n const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length\n return {\n taskId: input.taskId,\n requirements: input.requirements,\n completionRate,\n fullyComplete,\n unmeasuredCount,\n valid: fullyComplete,\n score: completionRate,\n ...(input.certification === undefined ? {} : { certification: input.certification }),\n }\n}\n\n/**\n * What a correctness checker declares about itself so the completion\n * verdict can carry a certification: which strategy member it discharges,\n * its exact identity, and the steps its answers rest on unverified.\n */\nexport interface CorrectnessCheckerAttestation {\n strategy: VerificationStrategySource\n checker: CheckerIdentity\n assumptions: string[]\n}\n\n/**\n * Decides whether a produced item's content actually fulfils a requirement.\n * Injected so the structural verifier stays pure and unit-testable; the\n * production implementation is `createLlmCorrectnessChecker`.\n *\n * `attestation` is optional metadata on the function value: a checker that\n * carries one yields certified completion verdicts; a bare function yields\n * the same verdict uncertified. A plain arrow function remains a valid\n * checker.\n */\nexport interface CorrectnessChecker {\n (\n requirement: CompletionRequirement,\n content: string,\n ): Promise<{ correct: boolean; reason: string }>\n attestation?: CorrectnessCheckerAttestation\n}\n\nconst STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'of',\n 'for',\n 'and',\n 'or',\n 'to',\n 'in',\n 'on',\n 'with',\n 'by',\n])\n\n// Deliverable-FORM vocabulary — words that name the SHAPE of an output, not its\n// domain content. A correct \"swap-comparison view persisted as a ui/** artifact\"\n// is an OpenUI JSON whose body says nothing about \"artifact\" / \"persisted\" /\n// \"view\"; the discriminative tokens are the domain nouns (swap, comparison).\n// Stripped from the REQUIREMENT side of structural recall so a deliverable is\n// matched on what it IS about, not on the boilerplate describing its form. The\n// correctness checker strips the same class via TITLE_STOPWORDS. Anti-game holds:\n// the distinctive domain tokens remain, so an off-topic item still fails.\nconst REQUIREMENT_FORM_STOPWORDS = new Set([\n 'generated',\n 'generate',\n 'view',\n 'render',\n 'rendered',\n 'persisted',\n 'persist',\n 'artifact',\n 'file',\n 'document',\n 'note',\n 'proposal',\n 'deliverable',\n 'output',\n 'created',\n 'create',\n 'produce',\n 'produced',\n 'flag',\n])\n\nconst MATCH_THRESHOLD = 0.5\nconst MIN_CONTENT_CHARS = 50\n\nfunction tokens(s: string, extraStop?: Set<string>): Set<string> {\n return new Set(\n s\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)),\n )\n}\n\n/**\n * Recall of the requirement's tokens within a candidate's identifying text.\n * Recall, not Jaccard — a candidate's path/id legitimately carries extra\n * tokens the requirement does not name. The requirement side drops\n * deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.\n */\nfunction tokenRecall(requirementText: string, candidateText: string): number {\n const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS)\n if (req.size === 0) return 0\n const cand = tokens(candidateText)\n let hit = 0\n for (const t of req) if (cand.has(t)) hit++\n return hit / req.size\n}\n\ninterface Candidate {\n reqIndex: number\n /** Unique key for a produced item — each item satisfies at most one requirement. */\n itemKey: string\n score: number\n evidence: string\n /** Content to correctness-check, or null when the matched item has none. */\n content: string | null\n}\n\nfunction artifactCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n artifacts: Artifact[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n artifacts.forEach((a, i) => {\n if ((a.content ?? '').trim().length < MIN_CONTENT_CHARS) return\n // Match against the artifact CONTENT too, not just its path + kind — a\n // generated view / note whose path is generic still satisfies a requirement\n // when its body covers it (e.g. an OpenUI comparison grounded in the on-file\n // figures). Bounded slice keeps the recall text cheap; MATCH_THRESHOLD holds.\n let score = tokenRecall(\n reqText,\n `${a.path ?? ''} ${a.kind} ${(a.content ?? '').slice(0, 4000)}`,\n )\n if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) {\n score = Math.max(score, 1)\n }\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `artifact:${i}`,\n score,\n evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,\n content: a.content ?? null,\n })\n })\n return out\n}\n\nfunction proposalCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n proposals: ProducedProposal[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n for (const p of proposals) {\n // Pending or rejected work is not a completed deliverable.\n if (p.status !== 'approved') continue\n // A proposal needs an assessable BODY to be a deliverable. A bare title is\n // not completion: correctness cannot be judged on it, so a title-only match\n // would auto-pass the oracle (structurallyPresent && correct===null →\n // satisfied) with no verifiable content. Tool calls are the only\n // legitimately content-less deliverable (`toolCallCandidates`).\n const body = (p.content ?? '').trim()\n if (body.length < MIN_CONTENT_CHARS) continue\n // Match against the body as well as the (often short) title — a refusal /\n // flag / analysis proposal whose title is a label still satisfies a\n // descriptively-worded requirement when its content covers it. MATCH_THRESHOLD\n // + the requirement's distinctive tokens keep an off-topic proposal out;\n // correctness (a SEMANTIC checker, NOT this lexical pass) then judges\n // polarity/fulfilment, so a negation that merely contains the tokens fails.\n // Structural and correctness must use different evidence or the two-stage\n // check collapses to one lexical gate.\n const score = tokenRecall(reqText, `${p.title} ${body}`)\n if (score < MATCH_THRESHOLD) continue\n out.push({\n reqIndex,\n itemKey: `proposal:${p.id}`,\n score,\n evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,\n content: body,\n })\n }\n return out\n}\n\nfunction toolCallCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n toolCalls: string[],\n): Candidate[] {\n const out: Candidate[] = []\n toolCalls.forEach((name, i) => {\n const score = tokenRecall(req.title, name)\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `tool:${i}`,\n score,\n evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,\n content: null,\n })\n })\n return out\n}\n\n/**\n * Verify whether a run completed the task. `checkCorrectness` is injected —\n * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.\n *\n * Throws on a gold spec with no requirements: an eval task that requires\n * nothing is a misconfiguration, not a vacuously-complete task.\n */\nexport async function verifyCompletion(\n gold: TaskGold,\n state: ProducedState,\n checkCorrectness: CorrectnessChecker,\n): Promise<CompletionVerdict> {\n if (gold.requirements.length === 0) {\n throw new Error(\n `verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`,\n )\n }\n\n // Collect every above-threshold (requirement, produced-item) candidate, then\n // assign greedily by descending score: each requirement and each produced\n // item is used at most once. One deliverable fulfils one requirement.\n const candidates: Candidate[] = []\n gold.requirements.forEach((req, i) => {\n const by = req.satisfiedBy ?? 'any'\n if (by === 'artifact' || by === 'any') {\n candidates.push(...artifactCandidates(req, i, state.artifacts))\n }\n if (by === 'proposal' || by === 'any') {\n candidates.push(...proposalCandidates(req, i, state.proposals))\n }\n if (by === 'tool-call' || by === 'any') {\n candidates.push(...toolCallCandidates(req, i, state.toolCalls))\n }\n })\n candidates.sort((a, b) => b.score - a.score)\n\n const assigned = new Map<number, Candidate>()\n const itemTaken = new Set<string>()\n for (const c of candidates) {\n if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue\n assigned.set(c.reqIndex, c)\n itemTaken.add(c.itemKey)\n }\n\n const requirements: RequirementCheck[] = []\n for (let i = 0; i < gold.requirements.length; i++) {\n const req = gold.requirements[i]!\n const match = assigned.get(i)\n const evidence: string[] = []\n let correct: boolean | null = null\n let unmeasuredReason: string | undefined\n\n if (match) {\n evidence.push(match.evidence)\n if (match.content !== null) {\n try {\n const r = await checkCorrectness(req, match.content)\n correct = r.correct\n evidence.push(`correctness: ${r.correct ? 'pass' : 'fail'} — ${r.reason}`)\n } catch (err) {\n // The CHECKER failed, not the requirement. Recording this as a\n // zero would fabricate a model failure out of an infrastructure\n // one; the requirement is unmeasured and leaves the denominator.\n unmeasuredReason =\n err instanceof JudgeParseError\n ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}`\n : `checker call failed: ${err instanceof Error ? err.message : String(err)}`\n evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`)\n }\n } else {\n evidence.push('correctness: not assessed — matched item carries no content')\n }\n } else {\n const by = req.satisfiedBy ?? 'any'\n const kind = by === 'any' ? 'artifact/proposal/tool-call' : by\n evidence.push(`no produced ${kind} matched this requirement`)\n }\n\n const structurallyPresent = match !== undefined\n const unmeasured = unmeasuredReason !== undefined\n const satisfied = structurallyPresent && !unmeasured && correct !== false\n requirements.push({\n reqId: req.reqId,\n title: req.title,\n structurallyPresent,\n correct,\n satisfied,\n ...(unmeasured ? { unmeasured: true as const, unmeasuredReason } : {}),\n evidence,\n })\n }\n\n // The verdict blends the deterministic structural stage with the injected\n // correctness stage, so the certification names the checker's own member\n // and carries the structural stage's lexical nature as an assumption. A\n // checker without an attestation yields an uncertified verdict.\n const attestation = checkCorrectness.attestation\n const certification: VerdictCertification | undefined = attestation\n ? {\n strategy: attestation.strategy,\n checker: attestation.checker,\n assumptions: [\n 'structural matching is lexical token recall over produced items — the correctness stage only sees items it matched',\n ...attestation.assumptions,\n ],\n evidenceDigest: certificationEvidenceDigest({ taskId: gold.taskId, requirements }),\n }\n : undefined\n\n return completionVerdict({\n taskId: gold.taskId,\n requirements,\n ...(certification === undefined ? {} : { certification }),\n })\n}\n\nexport interface LlmCorrectnessCheckerOpts {\n model?: string\n /** Optional ledger for direct use. */\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /** Max chars of artifact content sent to the checker. */\n maxContentChars?: number\n /**\n * Checker LLM calls per requirement before giving up (parse failures and\n * call errors both consume attempts). The failure then surfaces as an\n * `unmeasured` requirement, never a zero.\n */\n maxAttempts?: number\n /**\n * Forensic capture of every checker request/response/error — without it a\n * checker failure is unauditable (the agent-turn raws never contain the\n * checker's own calls). Same sink contract as `LlmClient`.\n */\n rawSink?: RawProviderSink\n}\n\n/**\n * Parse the correctness checker's model response. Tolerates a response\n * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the\n * verdict boolean usually lands in the first few tokens, so a recovered\n * prefix with a boolean `correct` is a real measurement, not a guess.\n * Fails loud (JudgeParseError) when no boolean verdict is recoverable.\n */\nexport function parseCorrectnessResponse(raw: string): { correct: boolean; reason: string } {\n const readVerdict = (candidate: unknown): { correct: boolean; reason: string } | null => {\n if (candidate === null || typeof candidate !== 'object') return null\n const { correct, reason } = candidate as { correct?: unknown; reason?: unknown }\n if (typeof correct !== 'boolean') return null\n return { correct, reason: typeof reason === 'string' ? reason : '' }\n }\n\n const match = raw.match(/\\{[\\s\\S]*\\}/)\n if (match) {\n try {\n const strict = readVerdict(JSON.parse(match[0]))\n if (strict) return strict\n } catch {\n // fall through to truncation recovery\n }\n }\n // The strict path needs a closing `}`; a cap-hit response has none. Take\n // everything from the first `{` and auto-close it.\n const start = raw.indexOf('{')\n if (start !== -1) {\n const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)))\n if (recovered) return recovered\n }\n throw new JudgeParseError('correctness-checker', raw)\n}\n\n/**\n * Production `CorrectnessChecker` — one LLM call per matched artifact,\n * deterministic (temperature 0), structured JSON out. Judges fulfilment\n * only: a plan, a gesture, or a description of what should be done does not\n * fulfil a requirement — the artifact must BE the deliverable.\n */\nexport function createLlmCorrectnessChecker(\n chat: ChatClient,\n opts: LlmCorrectnessCheckerOpts = {},\n): CorrectnessChecker {\n const model = opts.model ?? 'claude-sonnet-4-6'\n const maxContentChars = opts.maxContentChars ?? 8000\n const maxAttempts = opts.maxAttempts ?? 2\n const costLedger = opts.costLedger ?? new CostLedger()\n const sink = opts.rawSink\n const record = async (event: RawProviderEvent): Promise<void> => {\n // Forensic capture is best-effort; the verdict is the system of record.\n try {\n await sink?.record(event)\n } catch {\n // Intentionally swallowed.\n }\n }\n const checker: CorrectnessChecker = async (requirement, content) => {\n const request = {\n model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}.',\n },\n {\n role: 'user' as const,\n content: `Requirement: ${requirement.title}\\n${\n requirement.category ? `Category: ${requirement.category}\\n` : ''\n }\\nProduced artifact:\\n${content.slice(0, maxContentChars)}`,\n },\n ],\n temperature: 0,\n maxTokens: 200,\n } satisfies ChatRequest\n let lastErr: unknown\n for (let attempt = 0; attempt < maxAttempts; attempt++) {\n const started = Date.now()\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'request',\n timestamp: started,\n requestBody: request,\n redactedFields: [],\n })\n try {\n const paid = await costLedger.runPaidCall({\n channel: 'verifier',\n phase: opts.costPhase ?? 'completion.correctness',\n actor: 'correctness-checker',\n model,\n maximumCharge:\n chat.maximumAttempts === undefined\n ? undefined\n : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),\n tags: {\n ...opts.costTags,\n requirementId: requirement.reqId,\n attempt: String(attempt),\n },\n signal: opts.signal,\n execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),\n receipt: costReceiptFromLlm,\n receiptFromError: costReceiptFromLlmError,\n })\n if (!paid.succeeded) throw paid.error\n const resp = paid.value\n // Hold the transport to its own word: a verdict produced by a model\n // other than the one requested is not this model's verdict. A\n // transport that echoes no id cannot be made to prove identity here —\n // callers needing that proof enable it at the client\n // (`LlmClientOptions.assertServedModel`) or gate on `assertModelsServed`.\n assertServedModel(model, resp.servedModel, {\n allowUnreported: true,\n context: `correctness checker for requirement ${requirement.reqId}`,\n })\n const raw = resp.content\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'response',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n responseBody: resp,\n redactedFields: [],\n })\n return parseCorrectnessResponse(raw)\n } catch (err) {\n lastErr = err\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n // Substitution is a verdict about identity, not a transient fault:\n // another attempt reaches the same wrong model and spends more tokens.\n if (err instanceof ModelSubstitutionError) throw err\n }\n }\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n }\n checker.attestation = {\n strategy: 'judge',\n checker: { name: 'llm-correctness-checker', version: model },\n assumptions: [\n 'served-model identity is accepted when the transport reports none (assertServedModel allowUnreported)',\n ],\n }\n return checker\n}\n\n/** Stopwords for requirement-title tokenization — drops the imperative verbs\n * ('review', 'update', …) common to deliverable titles so recall keys on the\n * substantive nouns, not the boilerplate ask. */\nconst TITLE_STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'and',\n 'or',\n 'for',\n 'to',\n 'of',\n 'in',\n 'on',\n 'with',\n 'review',\n 'update',\n 'new',\n 'proposed',\n])\n\n/**\n * Deterministic `CorrectnessChecker` — the no-LLM counterpart to\n * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its\n * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`\n * of the requirement title's significant tokens. No network.\n *\n * Polarity-blind: token recall credits a negation that contains the\n * requirement's tokens (\"I will NOT produce the comparison\" recalls every token\n * of \"produce the comparison\"). The structural match stage is ALSO lexical, so\n * pairing the two collapses to a single gameable gate. Use this only as an\n * opt-in structural pre-filter or for tasks whose requirements have no polarity\n * to invert; for produced-state grading the correctness checker MUST be semantic\n * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.\n */\nexport function createTokenRecallChecker(\n opts: { minRecall?: number; minContentLength?: number } = {},\n): CorrectnessChecker {\n const minRecall = opts.minRecall ?? 0.5\n const minLen = opts.minContentLength ?? 120\n const checker: CorrectnessChecker = async (requirement, content) => {\n const body = content.trim()\n if (body.length < minLen)\n return {\n correct: false,\n reason: `content too thin (${body.length} chars) to be the deliverable`,\n }\n const titleTokens = requirement.title\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t))\n if (titleTokens.length === 0)\n return {\n correct: true,\n reason: 'requirement title has no significant tokens — structural match accepted',\n }\n const lower = body.toLowerCase()\n const hits = titleTokens.filter((t) => lower.includes(t)).length\n const recall = hits / titleTokens.length\n return recall >= minRecall\n ? {\n correct: true,\n reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`,\n }\n : {\n correct: false,\n reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`,\n }\n }\n // 'schema': token recall checks shape, not meaning — the member whose\n // documented failure mode (\"a well-formed wrong answer passes\") is\n // exactly this checker's polarity blindness.\n checker.attestation = {\n strategy: 'schema',\n checker: { name: 'token-recall-checker', version: packageVersion() },\n assumptions: ['polarity-blind: a negation that recalls the requirement tokens passes'],\n }\n return checker\n}\n","/**\n * Produced-state extraction — normalize a run's runtime event stream into the\n * typed `ProducedState` the completion oracle consumes.\n *\n * `ProducedState` answers \"what did the agent actually produce\" — vault\n * artifacts, proposals, tool calls. The runtime emits these as a stream of\n * events; this module is the single normalization point from that stream to\n * the shape `verifyCompletion` expects.\n *\n * Input is structurally typed (`RuntimeEventLike`) so this module does not\n * depend on agent-runtime — agent-runtime's `RuntimeStreamEvent` satisfies it\n * structurally. The `content` on `ArtifactEventLike` and the whole\n * `proposal_created` variant are the runtime-side enrichments this contract\n * requires; the runtime emits them, this module consumes them.\n */\n\nimport type { Artifact } from './artifact-validator'\nimport type { ProducedProposal, ProducedState } from './completion-verifier'\n\n/** A tool the agent invoked. */\nexport interface ToolCallEventLike {\n type: 'tool_call'\n toolName: string\n}\n\n/**\n * An artifact the agent produced. `content` is the enriched field — the\n * runtime's base `artifact` event carries only metadata; the completion\n * oracle needs the body to verify the deliverable, so the runtime emits it.\n */\nexport interface ArtifactEventLike {\n type: 'artifact'\n artifactId: string\n name?: string\n mimeType?: string\n uri?: string\n content?: string\n}\n\n/** A proposal / filing the agent created. */\nexport interface ProposalEventLike {\n type: 'proposal_created'\n proposalId: string\n title: string\n status?: 'pending' | 'approved' | 'rejected'\n // body of the proposal (e.g. a submit_proposal `description`). When present,\n // the completion oracle correctness-checks it like artifact content; absent,\n // the proposal is graded presence-only.\n content?: string\n}\n\n/**\n * The subset of runtime stream events `extractProducedState` consumes.\n * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;\n * the `{ type: string }` catch-all keeps the input permissive so callers can\n * pass the whole unfiltered telemetry stream — unrecognized events are skipped.\n */\nexport type RuntimeEventLike =\n | ToolCallEventLike\n | ArtifactEventLike\n | ProposalEventLike\n | { type: string }\n\nfunction artifactKind(mimeType: string | undefined): string {\n if (!mimeType) return 'file'\n if (mimeType.includes('json')) return 'json'\n if (mimeType.startsWith('text/')) return 'text'\n return 'file'\n}\n\n/**\n * Normalize a run's runtime event stream into `ProducedState`.\n *\n * Pure and total — unrecognized event types are skipped. `toolCalls` is\n * deduplicated by name in first-seen order (completion cares about a tool's\n * presence, not its call count). An artifact with neither a name nor a uri\n * still yields an entry keyed by its `artifactId` so it is never silently\n * dropped; an artifact with no `content` yields empty content, which the\n * completion oracle's structural check then rejects on its own.\n */\nexport function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState {\n const artifacts: Artifact[] = []\n const proposals: ProducedProposal[] = []\n const toolCalls: string[] = []\n const seenTools = new Set<string>()\n\n for (const ev of events) {\n if (ev.type === 'tool_call') {\n const name = (ev as ToolCallEventLike).toolName\n if (name && !seenTools.has(name)) {\n seenTools.add(name)\n toolCalls.push(name)\n }\n } else if (ev.type === 'artifact') {\n const a = ev as ArtifactEventLike\n artifacts.push({\n kind: artifactKind(a.mimeType),\n path: a.name ?? a.uri ?? a.artifactId,\n content: a.content ?? '',\n })\n } else if (ev.type === 'proposal_created') {\n const p = ev as ProposalEventLike\n proposals.push({\n id: p.proposalId,\n title: p.title,\n status: p.status ?? 'pending',\n ...(p.content !== undefined ? { content: p.content } : {}),\n })\n }\n }\n\n return { artifacts, proposals, toolCalls }\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAa,mBAA2C;CACtD;CACA;CACA;CACA;AACF;;;;;;AAwBA,MAAa,uBAAuB;;;;;;;;;;;;;;;;;;;AAoBpC,SAAgB,kBAAkB,MAAuC;CACvE,MAAM,YAAY,KAAK,aAAa;CACpC,IAAI,UAAU,WAAW,GAAG,MAAM,IAAI,gBAAgB,0CAA0C;CAChG,MAAM,YAAY,KAAK,KAAK,OAAO;CACnC,MAAM,SAAS,KAAK,WAAW,YAAY,CAAC,SAAS,IAAI,CAAC;CAC1D,IAAI,OAAO,WAAW,GACpB,MAAM,IAAI,gBACR,kGACF;CAEF,MAAM,MAAsB,CAAC;CAC7B,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,WAAW,WAAW;EAS/B,MAAM,YAAY,KAAK,mBACnB,SACA,OAAO,QAAQ,UAAU,qBAAqB,SAAS,KAAK,CAAC;EACjE,MAAM,YAAY,UAAU,SAAS,IAAI,YAAY,CAAC,oBAAoB;EAC1E,KAAK,MAAM,SAAS,WAAW;GAC7B,MAAM,UAAwB;IAC5B,GAAG,KAAK;IACR;IACA,MAAM,GAAG,KAAK,KAAK,QAAQ,QAAQ,GAAG,QAAQ,GAAG;IACjD,OAAO;KAAE,GAAG,KAAK,KAAK;KAAO,SAAS;IAAM;IAC5C,UAAU;KAAE,GAAI,KAAK,KAAK,YAAY,CAAC;KAAI;KAAS,cAAc;IAAM;GAC1E;GACA,MAAM,KAAK,eAAe,OAAO;GACjC,IAAI,KAAK,IAAI,EAAE,GAAG;GAClB,KAAK,IAAI,EAAE;GACX,IAAI,KAAK,OAAO;EAClB;CACF;CACA,IAAI,IAAI,WAAW,GAGjB,MAAM,IAAI,gBACR,uDAAuD,UAAU,KAAK,IAAI,EAAE,aAAa,OAAO,KAAK,IAAI,EAAE,IAC7G;CAEF,OAAO;AACT;;;;;;;;AASA,SAAgB,cACd,SACqD;CACrD,MAAM,IAAI,QAAQ;CAClB,MAAM,UAAU,GAAG;CACnB,MAAM,QAAQ,GAAG;CACjB,IAAI,OAAO,YAAY,YAAY,OAAO,UAAU,UAClD,OAAO;EAAW;EAAwB;CAAM;AAGpD;;;;;;;;AASA,SAAgB,eAAe,SAA+B;CAE5D,OAAO,GADO,qBAAqB,yBAAyB,OAAO,CAAC,KAAK,UACzD,GAAG,iBAAiB,OAAO,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1D;;;;;AAMA,SAAgB,oBAAoB,SAA+B;CACjE,MAAM,QAAQ,QAAQ,OAAO,SAAS,KAAK;CAC3C,IAAI,CAAC,OAEH,MAAM,IAAI,gBACR,iBAFY,yBAAyB,OAAO,KAAK,kBAE1B,gDACzB;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,SAA2C;CAC3E,OAAO,QAAQ,MAAM,KAAK,KAAK,QAAQ,SAAS,KAAK,KAAK,KAAA;AAC5D;AAEA,SAAS,qBAAqB,OAA+C;CAM3E,OALa,OACT,KAAK,CAAC,CACP,QAAQ,qBAAqB,GAAG,CAAC,CACjC,QAAQ,OAAO,GAAG,CAAC,CACnB,QAAQ,UAAU,EAAE,KACR,KAAA;AACjB;AAEA,SAAS,QAA2C,OAAsB;CACxE,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,KAAK,GAC7C,IAAI,UAAU,KAAA,GAAW,IAAI,OAAO;CAEtC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,iBAAiB,SAA+B;CAC9D,MAAM,QAAQ,oBAAoB,OAAO;CACzC,MAAM,YAAY,QAAQ;EACxB,GAAG;EACH,MAAM,KAAA;EACN,aAAa,KAAA;EACb,MAAM,QAAQ,OAAO,CAAC,GAAG,QAAQ,IAAI,CAAC,CAAC,KAAK,IAAI,KAAA;EAChD,OAAO,QAAQ;GAAE,GAAG,QAAQ;GAAO,SAAS;EAAM,CAAC;CACrD,CAAC;CACD,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,gBAAgB,SAAS,CAAC,CAAC,CAAC,OAAO,KAAK;AAC7E;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACtEA,SAAgB,kBAAkB,OAMZ;CACpB,IAAI,MAAM,aAAa,WAAW,GAChC,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,+DAC3C;CAEF,MAAM,aAAa,MAAM,aAAa,QAAQ,MAAM,CAAC,EAAE,UAAU;CACjE,MAAM,kBAAkB,MAAM,aAAa,SAAS,WAAW;CAC/D,IAAI,WAAW,WAAW,GAGxB,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,yCAAyC,MAAM,aAAa,OAAO,8BAA8B,MAAM,aAAa,EAAE,EAAE,oBAAoB,iBAAiB,EACxM;CAEF,MAAM,iBAAiB,WAAW,QAAQ,MAAM,EAAE,SAAS,CAAC,CAAC;CAC7D,MAAM,iBAAiB,iBAAiB,WAAW;CAGnD,MAAM,gBAAgB,oBAAoB,KAAK,mBAAmB,WAAW;CAC7E,OAAO;EACL,QAAQ,MAAM;EACd,cAAc,MAAM;EACpB;EACA;EACA;EACA,OAAO;EACP,OAAO;EACP,GAAI,MAAM,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,eAAe,MAAM,cAAc;CACpF;AACF;AA+BA,MAAM,4BAAY,IAAI,IAAI;CACxB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAUD,MAAM,6CAA6B,IAAI,IAAI;CACzC;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAM,kBAAkB;AACxB,MAAM,oBAAoB;AAE1B,SAAS,OAAO,GAAW,WAAsC;CAC/D,OAAO,IAAI,IACT,EACG,YAAY,CAAC,CACb,MAAM,YAAY,CAAC,CACnB,QAAQ,MAAM,EAAE,SAAS,KAAK,CAAC,UAAU,IAAI,CAAC,KAAK,CAAC,WAAW,IAAI,CAAC,CAAC,CAC1E;AACF;;;;;;;AAQA,SAAS,YAAY,iBAAyB,eAA+B;CAC3E,MAAM,MAAM,OAAO,iBAAiB,0BAA0B;CAC9D,IAAI,IAAI,SAAS,GAAG,OAAO;CAC3B,MAAM,OAAO,OAAO,aAAa;CACjC,IAAI,MAAM;CACV,KAAK,MAAM,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC,GAAG;CACtC,OAAO,MAAM,IAAI;AACnB;AAYA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,GAAG,MAAM;EAC1B,KAAK,EAAE,WAAW,GAAA,CAAI,KAAK,CAAC,CAAC,SAAS,mBAAmB;EAKzD,IAAI,QAAQ,YACV,SACA,GAAG,EAAE,QAAQ,GAAG,GAAG,EAAE,KAAK,IAAI,EAAE,WAAW,GAAA,CAAI,MAAM,GAAG,GAAI,GAC9D;EACA,IAAI,IAAI,YAAY,EAAE,QAAQ,IAAI,SAAS,YAAY,MAAM,EAAE,KAAK,YAAY,GAC9E,QAAQ,KAAK,IAAI,OAAO,CAAC;EAE3B,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY;GACrB;GACA,UAAU,aAAa,EAAE,QAAQ,EAAE,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,CAAC;CACD,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,KAAK,MAAM,KAAK,WAAW;EAEzB,IAAI,EAAE,WAAW,YAAY;EAM7B,MAAM,QAAQ,EAAE,WAAW,GAAA,CAAI,KAAK;EACpC,IAAI,KAAK,SAAS,mBAAmB;EASrC,MAAM,QAAQ,YAAY,SAAS,GAAG,EAAE,MAAM,GAAG,MAAM;EACvD,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY,EAAE;GACvB;GACA,UAAU,sBAAsB,EAAE,MAAM,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS;EACX,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,MAAM,MAAM;EAC7B,MAAM,QAAQ,YAAY,IAAI,OAAO,IAAI;EACzC,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,QAAQ;GACjB;GACA,UAAU,cAAc,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACxE,SAAS;EACX,CAAC;CACH,CAAC;CACD,OAAO;AACT;;;;;;;;AASA,eAAsB,iBACpB,MACA,OACA,kBAC4B;CAC5B,IAAI,KAAK,aAAa,WAAW,GAC/B,MAAM,IAAI,MACR,2BAA2B,KAAK,OAAO,4CACzC;CAMF,MAAM,aAA0B,CAAC;CACjC,KAAK,aAAa,SAAS,KAAK,MAAM;EACpC,MAAM,KAAK,IAAI,eAAe;EAC9B,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,eAAe,OAAO,OAC/B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;CAElE,CAAC;CACD,WAAW,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;CAE3C,MAAM,2BAAW,IAAI,IAAuB;CAC5C,MAAM,4BAAY,IAAI,IAAY;CAClC,KAAK,MAAM,KAAK,YAAY;EAC1B,IAAI,SAAS,IAAI,EAAE,QAAQ,KAAK,UAAU,IAAI,EAAE,OAAO,GAAG;EAC1D,SAAS,IAAI,EAAE,UAAU,CAAC;EAC1B,UAAU,IAAI,EAAE,OAAO;CACzB;CAEA,MAAM,eAAmC,CAAC;CAC1C,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,aAAa,QAAQ,KAAK;EACjD,MAAM,MAAM,KAAK,aAAa;EAC9B,MAAM,QAAQ,SAAS,IAAI,CAAC;EAC5B,MAAM,WAAqB,CAAC;EAC5B,IAAI,UAA0B;EAC9B,IAAI;EAEJ,IAAI,OAAO;GACT,SAAS,KAAK,MAAM,QAAQ;GAC5B,IAAI,MAAM,YAAY,MACpB,IAAI;IACF,MAAM,IAAI,MAAM,iBAAiB,KAAK,MAAM,OAAO;IACnD,UAAU,EAAE;IACZ,SAAS,KAAK,gBAAgB,EAAE,UAAU,SAAS,OAAO,KAAK,EAAE,QAAQ;GAC3E,SAAS,KAAK;IAIZ,mBACE,eAAe,kBACX,6CAA6C,IAAI,IAAI,MAAM,GAAG,GAAG,MACjE,wBAAwB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAC7E,SAAS,KAAK,6BAA6B,kBAAkB;GAC/D;QAEA,SAAS,KAAK,6DAA6D;EAE/E,OAAO;GACL,MAAM,KAAK,IAAI,eAAe;GAC9B,MAAM,OAAO,OAAO,QAAQ,gCAAgC;GAC5D,SAAS,KAAK,eAAe,KAAK,0BAA0B;EAC9D;EAEA,MAAM,sBAAsB,UAAU,KAAA;EACtC,MAAM,aAAa,qBAAqB,KAAA;EACxC,MAAM,YAAY,uBAAuB,CAAC,cAAc,YAAY;EACpE,aAAa,KAAK;GAChB,OAAO,IAAI;GACX,OAAO,IAAI;GACX;GACA;GACA;GACA,GAAI,aAAa;IAAE,YAAY;IAAe;GAAiB,IAAI,CAAC;GACpE;EACF,CAAC;CACH;CAMA,MAAM,cAAc,iBAAiB;CACrC,MAAM,gBAAkD,cACpD;EACE,UAAU,YAAY;EACtB,SAAS,YAAY;EACrB,aAAa,CACX,sHACA,GAAG,YAAY,WACjB;EACA,gBAAgB,4BAA4B;GAAE,QAAQ,KAAK;GAAQ;EAAa,CAAC;CACnF,IACA,KAAA;CAEJ,OAAO,kBAAkB;EACvB,QAAQ,KAAK;EACb;EACA,GAAI,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,cAAc;CACzD,CAAC;AACH;;;;;;;;AAgCA,SAAgB,yBAAyB,KAAmD;CAC1F,MAAM,eAAe,cAAoE;EACvF,IAAI,cAAc,QAAQ,OAAO,cAAc,UAAU,OAAO;EAChE,MAAM,EAAE,SAAS,WAAW;EAC5B,IAAI,OAAO,YAAY,WAAW,OAAO;EACzC,OAAO;GAAE;GAAS,QAAQ,OAAO,WAAW,WAAW,SAAS;EAAG;CACrE;CAEA,MAAM,QAAQ,IAAI,MAAM,aAAa;CACrC,IAAI,OACF,IAAI;EACF,MAAM,SAAS,YAAY,KAAK,MAAM,MAAM,EAAE,CAAC;EAC/C,IAAI,QAAQ,OAAO;CACrB,QAAQ,CAER;CAIF,MAAM,QAAQ,IAAI,QAAQ,GAAG;CAC7B,IAAI,UAAU,IAAI;EAChB,MAAM,YAAY,YAAY,qBAAqB,IAAI,MAAM,KAAK,CAAC,CAAC;EACpE,IAAI,WAAW,OAAO;CACxB;CACA,MAAM,IAAI,gBAAgB,uBAAuB,GAAG;AACtD;;;;;;;AAQA,SAAgB,4BACd,MACA,OAAkC,CAAC,GACf;CACpB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,cAAc,KAAK,eAAe;CACxC,MAAM,aAAa,KAAK,cAAc,IAAI,WAAW;CACrD,MAAM,OAAO,KAAK;CAClB,MAAM,SAAS,OAAO,UAA2C;EAE/D,IAAI;GACF,MAAM,MAAM,OAAO,KAAK;EAC1B,QAAQ,CAER;CACF;CACA,MAAM,UAA8B,OAAO,aAAa,YAAY;EAClE,MAAM,UAAU;GACd;GACA,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IACE,MAAM;IACN,SAAS,gBAAgB,YAAY,MAAM,IACzC,YAAY,WAAW,aAAa,YAAY,SAAS,MAAM,GAChE,wBAAwB,QAAQ,MAAM,GAAG,eAAe;GAC3D,CACF;GACA,aAAa;GACb,WAAW;EACb;EACA,IAAI;EACJ,KAAK,IAAI,UAAU,GAAG,UAAU,aAAa,WAAW;GACtD,MAAM,UAAU,KAAK,IAAI;GACzB,MAAM,OAAO;IACX,SAAS,WAAW;IACpB,UAAU,KAAK;IACf;IACA,UAAU;IACV,SAAS;IACT,cAAc;IACd,WAAW;IACX,WAAW;IACX,aAAa;IACb,gBAAgB,CAAC;GACnB,CAAC;GACD,IAAI;IACF,MAAM,OAAO,MAAM,WAAW,YAAY;KACxC,SAAS;KACT,OAAO,KAAK,aAAa;KACzB,OAAO;KACP;KACA,eACE,KAAK,oBAAoB,KAAA,IACrB,KAAA,IACA,2BAA2B,SAAS,EAAE,iBAAiB,KAAK,gBAAgB,CAAC;KACnF,MAAM;MACJ,GAAG,KAAK;MACR,eAAe,YAAY;MAC3B,SAAS,OAAO,OAAO;KACzB;KACA,QAAQ,KAAK;KACb,UAAU,QAAQ,WAAW,KAAK,KAAK,SAAS;MAAE;MAAQ,gBAAgB;KAAO,CAAC;KAClF,SAAS;KACT,kBAAkB;IACpB,CAAC;IACD,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;IAChC,MAAM,OAAO,KAAK;IAMlB,kBAAkB,OAAO,KAAK,aAAa;KACzC,iBAAiB;KACjB,SAAS,uCAAuC,YAAY;IAC9D,CAAC;IACD,MAAM,MAAM,KAAK;IACjB,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc;KACd,gBAAgB,CAAC;IACnB,CAAC;IACD,OAAO,yBAAyB,GAAG;GACrC,SAAS,KAAK;IACZ,UAAU;IACV,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;KAC7D,gBAAgB,CAAC;IACnB,CAAC;IAGD,IAAI,eAAe,wBAAwB,MAAM;GACnD;EACF;EACA,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;CACtE;CACA,QAAQ,cAAc;EACpB,UAAU;EACV,SAAS;GAAE,MAAM;GAA2B,SAAS;EAAM;EAC3D,aAAa,CACX,uGACF;CACF;CACA,OAAO;AACT;;;ACpmBA,SAAS,aAAa,UAAsC;CAC1D,IAAI,CAAC,UAAU,OAAO;CACtB,IAAI,SAAS,SAAS,MAAM,GAAG,OAAO;CACtC,IAAI,SAAS,WAAW,OAAO,GAAG,OAAO;CACzC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,qBAAqB,QAAoD;CACvF,MAAM,YAAwB,CAAC;CAC/B,MAAM,YAAgC,CAAC;CACvC,MAAM,YAAsB,CAAC;CAC7B,MAAM,4BAAY,IAAI,IAAY;CAElC,KAAK,MAAM,MAAM,QACf,IAAI,GAAG,SAAS,aAAa;EAC3B,MAAM,OAAQ,GAAyB;EACvC,IAAI,QAAQ,CAAC,UAAU,IAAI,IAAI,GAAG;GAChC,UAAU,IAAI,IAAI;GAClB,UAAU,KAAK,IAAI;EACrB;CACF,OAAO,IAAI,GAAG,SAAS,YAAY;EACjC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,MAAM,aAAa,EAAE,QAAQ;GAC7B,MAAM,EAAE,QAAQ,EAAE,OAAO,EAAE;GAC3B,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,OAAO,IAAI,GAAG,SAAS,oBAAoB;EACzC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,IAAI,EAAE;GACN,OAAO,EAAE;GACT,QAAQ,EAAE,UAAU;GACpB,GAAI,EAAE,YAAY,KAAA,IAAY,EAAE,SAAS,EAAE,QAAQ,IAAI,CAAC;EAC1D,CAAC;CACH;CAGF,OAAO;EAAE;EAAW;EAAW;CAAU;AAC3C"}
|
|
1
|
+
{"version":3,"file":"produced-state-CxmbFxFd.js","names":[],"sources":["../src/agent-profile.ts","../src/completion-verifier.ts","../src/produced-state.ts"],"sourcesContent":["import { createHash } from 'node:crypto'\nimport type { AgentProfile } from '@tangle-network/agent-interface'\nimport { type HarnessType, harnessSupportsModel } from '@tangle-network/agent-interface'\nimport { ValidationError } from './errors'\nimport { canonicalString } from './ledger-core/canonical'\n\nexport type { AgentProfile, HarnessType } from '@tangle-network/agent-interface'\n\n/**\n * The agentic coding harnesses an eval sweeps by default — the ones we care about\n * ranking. This is the SINGLE source of that list; consumers import it instead of\n * re-declaring their own (a re-declared list is how the fleet drifts). Pass an\n * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known\n * harness) to widen beyond these.\n */\nexport const CODING_HARNESSES: readonly HarnessType[] = [\n 'opencode',\n 'claude-code',\n 'codex',\n 'kimi-code',\n]\n\nexport interface ProfileAxisSpec {\n /** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the\n * harness and model vary. `model.default` is the fallback model. */\n base: AgentProfile\n /** Harnesses to cross. Default: {@link CODING_HARNESSES}. */\n harnesses?: readonly HarnessType[]\n /** Models to cross. Default: `[base.model.default]` — one model, i.e. today's\n * single-model behaviour, so omitting this never changes an existing run. */\n models?: readonly string[]\n /** Force every (harness, model) pair verbatim, even ones the harness can't run —\n * for deliberately testing failure modes. Default (false): SNAP instead — a\n * vendor-locked harness runs only the swept models in its family, or its native\n * default when it supports none, so no harness is dropped and none gets a\n * guaranteed-failing foreign-model cell. */\n keepIncompatible?: boolean\n}\n\n/** Model sentinel for a vendor-locked harness that supports none of the swept models:\n * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness\n * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi\n * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship\n * table that would rot as router catalogs change. */\nexport const HARNESS_NATIVE_MODEL = 'default'\n\n/**\n * Expand a base profile across the harness × model matrix into the `AgentProfile[]`\n * that `runProfileMatrix` / `selfImprove` score — the ONE place \"which harnesses ×\n * which models do we evaluate\" lives, so no product hand-rolls its own harness list\n * or column→profile mapping (the pattern that let those copies drift and silently\n * break the harness pivot).\n *\n * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,\n * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both\n * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row\n * and results join back by harness/model via {@link harnessAxisOf} with no\n * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or\n * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every\n * requested harness runs; `keepIncompatible` forces every pair verbatim.\n *\n * Omit `harnesses`/`models` to sweep the full default set — the \"turn it on for\n * everything we care about\" switch, identical in shape whether one harness or all.\n */\nexport function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[] {\n const harnesses = spec.harnesses ?? CODING_HARNESSES\n if (harnesses.length === 0) throw new ValidationError('expandProfileAxes: no harnesses to sweep')\n const baseModel = spec.base.model?.default\n const models = spec.models ?? (baseModel ? [baseModel] : [])\n if (models.length === 0) {\n throw new ValidationError(\n 'expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied',\n )\n }\n const out: AgentProfile[] = []\n const seen = new Set<string>()\n for (const harness of harnesses) {\n // A universal (router-backed) harness — opencode/pi/claudish — runs every swept\n // model. A vendor-locked harness — codex/claude-code/kimi-code — runs only the\n // swept models in its own family; when it supports NONE of them it snaps to its\n // native default (the `HARNESS_NATIVE_MODEL` sentinel it resolves at runtime)\n // rather than being dropped, so every requested harness still appears in the\n // sweep on a model it can actually run — e.g. sweeping `deepseek/x` puts opencode\n // on deepseek and kimi-code on its own Kimi model, a real head-to-head.\n // `keepIncompatible` forces every (harness, model) pair verbatim (failure-mode runs).\n const supported = spec.keepIncompatible\n ? models\n : models.filter((model) => harnessSupportsModel(harness, model))\n const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL]\n for (const model of effective) {\n const profile: AgentProfile = {\n ...spec.base,\n harness,\n name: `${spec.base.name ?? 'agent'}/${harness}/${model}`,\n model: { ...spec.base.model, default: model },\n metadata: { ...(spec.base.metadata ?? {}), harness, harnessModel: model },\n }\n const id = agentProfileId(profile)\n if (seen.has(id)) continue\n seen.add(id)\n out.push(profile)\n }\n }\n if (out.length === 0) {\n // Unreachable in normal use — snapping guarantees ≥1 cell per harness — but keep a\n // fail-closed guard so a future refactor can't silently produce an empty sweep.\n throw new ValidationError(\n `expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(', ')}], models=[${models.join(', ')}]).`,\n )\n }\n return out\n}\n\n/**\n * Read the (harness, model) a matrix cell ran under, off a profile or a result row's\n * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile\n * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by\n * this instead of recomputing an id (recomputing the wrong key is what broke the pivot\n * in the hand-rolled copies).\n */\nexport function harnessAxisOf(\n profile: Pick<AgentProfile, 'metadata'>,\n): { harness: HarnessType; model: string } | undefined {\n const m = profile.metadata as Record<string, unknown> | undefined\n const harness = m?.harness\n const model = m?.harnessModel\n if (typeof harness === 'string' && typeof model === 'string') {\n return { harness: harness as HarnessType, model }\n }\n return undefined\n}\n\n/**\n * Collision-resistant, path-safe, human-readable profile id for eval artifacts.\n * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix\n * keys, and directory names where two profiles must not collapse onto one row.\n * The suffix is the first 64 bits of the behaviour hash, enough for ordinary\n * eval matrices while keeping filenames readable.\n */\nexport function agentProfileId(profile: AgentProfile): string {\n const label = pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? 'profile'\n return `${label}-${agentProfileHash(profile).slice(0, 16)}`\n}\n\n/**\n * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete\n * model id because run records reject bare/missing model aliases.\n */\nexport function agentProfileModelId(profile: AgentProfile): string {\n const model = profile.model?.default?.trim()\n if (!model) {\n const label = agentProfileDisplayLabel(profile) ?? 'unnamed profile'\n throw new ValidationError(\n `AgentProfile \"${label}\" has no model.default — cannot record eval run`,\n )\n }\n return model\n}\n\nfunction agentProfileDisplayLabel(profile: AgentProfile): string | undefined {\n return profile.name?.trim() || profile.version?.trim() || undefined\n}\n\nfunction pathSafeProfileLabel(label: string | undefined): string | undefined {\n const safe = label\n ?.trim()\n .replace(/[^A-Za-z0-9._-]+/g, '-')\n .replace(/-+/g, '-')\n .replace(/^-|-$/g, '')\n return safe || undefined\n}\n\nfunction compact<T extends Record<string, unknown>>(input: T): Partial<T> {\n const out: Record<string, unknown> = {}\n for (const [key, value] of Object.entries(input)) {\n if (value !== undefined) out[key] = value\n }\n return out as Partial<T>\n}\n\n/**\n * Deterministic behaviour identity for the canonical\n * `@tangle-network/agent-interface` AgentProfile.\n *\n * `name` and `description` are labels and do not affect the hash. Profile\n * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,\n * and extensions do affect the hash. Resource array order is hash-bearing\n * because mount order can change agent behaviour. Undefined fields are treated\n * as absent; explicit `null` fields remain hash-bearing.\n */\nexport function agentProfileHash(profile: AgentProfile): string {\n const model = agentProfileModelId(profile)\n const behaviour = compact({\n ...profile,\n name: undefined,\n description: undefined,\n tags: profile.tags ? [...profile.tags].sort() : undefined,\n model: compact({ ...profile.model, default: model }),\n })\n return createHash('sha256').update(canonicalString(behaviour)).digest('hex')\n}\n","/**\n * Completion verifier — the task-completion oracle.\n *\n * Answers the only eval question that is not a proxy: did the agent actually\n * COMPLETE the task — produce every required deliverable, persisted and\n * correct — rather than describe what should be done. A fluent transcript\n * that never produces the artifact scores zero here.\n *\n * Per requirement, a two-stage check:\n * 1. Structural — a produced item (vault artifact / approved proposal /\n * tool call) of the right kind is matched against the requirement and\n * carries non-empty content. Deterministic; no LLM.\n * 2. Correctness — only if structurally present AND the matched item\n * carries content, one targeted check decides whether that item\n * actually fulfils the requirement. A hallucinated artifact fails here;\n * an absent one already failed stage 1.\n *\n * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —\n * checker failures — are excluded from the denominator, never scored as\n * zeros). Quality dimensions are meaningless on an incomplete task — callers\n * gate on `fullyComplete` / `completionRate` before scoring quality.\n */\n\nimport { randomUUID } from 'node:crypto'\nimport type { ChatClient, ChatRequest } from './analyst/chat-client'\nimport type { Artifact } from './artifact-validator'\nimport { CostLedger, type CostLedgerHandle } from './cost-ledger'\nimport { assertServedModel, ModelSubstitutionError } from './integrity/served-model'\nimport { recoverTruncatedJson } from './json-recovery'\nimport { JudgeParseError } from './judges'\nimport {\n costReceiptFromLlm,\n costReceiptFromLlmError,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport { packageVersion } from './package-version'\nimport type { RawProviderEvent, RawProviderSink } from './trace/raw-provider-sink'\nimport {\n certificationEvidenceDigest,\n type DefaultVerdict,\n type VerdictCertification,\n} from './verdict'\nimport type { CheckerIdentity, VerificationStrategySource } from './verification-strategy'\n\n/** What kind of produced state can satisfy a requirement structurally. */\nexport type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any'\n\nexport interface CompletionRequirement {\n /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */\n reqId: string\n /** Human-readable description of the required deliverable. */\n title: string\n /** Optional kind/category hint, matched against a produced item's kind. */\n category?: string\n /** What produced state satisfies this requirement. Defaults to 'any'. */\n satisfiedBy?: SatisfiedBy\n}\n\nexport interface TaskGold {\n taskId: string\n requirements: CompletionRequirement[]\n}\n\nexport interface ProducedProposal {\n id: string\n title: string\n status: 'pending' | 'approved' | 'rejected'\n /** Optional persisted body — when present, enables a correctness check. */\n content?: string\n}\n\n/** Everything observable about what a run actually produced. */\nexport interface ProducedState {\n /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */\n artifacts: Artifact[]\n /** Proposals / filings the agent created. */\n proposals: ProducedProposal[]\n /** Names of tools the agent invoked. */\n toolCalls: string[]\n}\n\nexport interface RequirementCheck {\n reqId: string\n title: string\n /** A produced item of the right kind matched the requirement, non-empty. */\n structurallyPresent: boolean\n /**\n * Whether the matched item actually fulfils the requirement. `null` when\n * not structurally present, when the matched item carries no content\n * to assess, or when the correctness check itself failed (`unmeasured`).\n */\n correct: boolean | null\n /** structurallyPresent && !unmeasured && correct !== false. */\n satisfied: boolean\n /**\n * Set when the correctness check itself errored (LLM call failure or an\n * unparseable response after retry). The requirement's fulfilment is\n * UNKNOWN — `correct` stays null, `satisfied` is false, and\n * `completionVerdict` excludes the row from `completionRate`'s\n * denominator. Never folded into a zero: a synthetic zero is\n * indistinguishable from a real failure (see `JudgeParseError`).\n */\n unmeasured?: true\n /** Why the correctness check could not be measured (present iff `unmeasured`). */\n unmeasuredReason?: string\n /** Human-readable evidence for the verdict. */\n evidence: string[]\n}\n\n/** Extends the substrate verdict spine: `valid` = `fullyComplete` and\n * `score` = `completionRate` — derived in `completionVerdict()`, the one\n * place those equalities hold by construction. */\nexport interface CompletionVerdict extends DefaultVerdict {\n taskId: string\n requirements: RequirementCheck[]\n /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */\n completionRate: number\n /** Every measurable requirement satisfied (false when anything is unmeasured). */\n fullyComplete: boolean\n /** Requirements whose correctness check errored — reported, never scored as zero. */\n unmeasuredCount: number\n}\n\n/**\n * Construct a `CompletionVerdict` from the per-requirement checks, deriving\n * `completionRate` / `fullyComplete` and the spine fields (`valid` =\n * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero\n * requirements — a verdict over nothing is a misconfiguration, mirroring\n * `verifyCompletion`'s gold-spec guard.\n */\nexport function completionVerdict(input: {\n taskId: string\n requirements: RequirementCheck[]\n /** What certified the correctness stage, when anything did. Omitted =\n * an uncertified verdict — the honest default for a bare checker. */\n certification?: VerdictCertification\n}): CompletionVerdict {\n if (input.requirements.length === 0) {\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`,\n )\n }\n const measurable = input.requirements.filter((r) => !r.unmeasured)\n const unmeasuredCount = input.requirements.length - measurable.length\n if (measurable.length === 0) {\n // Every check errored: this is an infrastructure failure, not a scored\n // run. A 0-rate verdict here would be a fabricated measurement.\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? 'unknown reason'})`,\n )\n }\n const satisfiedCount = measurable.filter((r) => r.satisfied).length\n const completionRate = satisfiedCount / measurable.length\n // A run with unmeasured rows can still report a rate over what WAS\n // measured, but must not claim full completion.\n const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length\n return {\n taskId: input.taskId,\n requirements: input.requirements,\n completionRate,\n fullyComplete,\n unmeasuredCount,\n valid: fullyComplete,\n score: completionRate,\n ...(input.certification === undefined ? {} : { certification: input.certification }),\n }\n}\n\n/**\n * What a correctness checker declares about itself so the completion\n * verdict can carry a certification: which strategy member it discharges,\n * its exact identity, and the steps its answers rest on unverified.\n */\nexport interface CorrectnessCheckerAttestation {\n strategy: VerificationStrategySource\n checker: CheckerIdentity\n assumptions: string[]\n}\n\n/**\n * Decides whether a produced item's content actually fulfils a requirement.\n * Injected so the structural verifier stays pure and unit-testable; the\n * production implementation is `createLlmCorrectnessChecker`.\n *\n * `attestation` is optional metadata on the function value: a checker that\n * carries one yields certified completion verdicts; a bare function yields\n * the same verdict uncertified. A plain arrow function remains a valid\n * checker.\n */\nexport interface CorrectnessChecker {\n (\n requirement: CompletionRequirement,\n content: string,\n ): Promise<{ correct: boolean; reason: string }>\n attestation?: CorrectnessCheckerAttestation\n}\n\nconst STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'of',\n 'for',\n 'and',\n 'or',\n 'to',\n 'in',\n 'on',\n 'with',\n 'by',\n])\n\n// Deliverable-FORM vocabulary — words that name the SHAPE of an output, not its\n// domain content. A correct \"swap-comparison view persisted as a ui/** artifact\"\n// is an OpenUI JSON whose body says nothing about \"artifact\" / \"persisted\" /\n// \"view\"; the discriminative tokens are the domain nouns (swap, comparison).\n// Stripped from the REQUIREMENT side of structural recall so a deliverable is\n// matched on what it IS about, not on the boilerplate describing its form. The\n// correctness checker strips the same class via TITLE_STOPWORDS. Anti-game holds:\n// the distinctive domain tokens remain, so an off-topic item still fails.\nconst REQUIREMENT_FORM_STOPWORDS = new Set([\n 'generated',\n 'generate',\n 'view',\n 'render',\n 'rendered',\n 'persisted',\n 'persist',\n 'artifact',\n 'file',\n 'document',\n 'note',\n 'proposal',\n 'deliverable',\n 'output',\n 'created',\n 'create',\n 'produce',\n 'produced',\n 'flag',\n])\n\nconst MATCH_THRESHOLD = 0.5\nconst MIN_CONTENT_CHARS = 50\n\nfunction tokens(s: string, extraStop?: Set<string>): Set<string> {\n return new Set(\n s\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)),\n )\n}\n\n/**\n * Recall of the requirement's tokens within a candidate's identifying text.\n * Recall, not Jaccard — a candidate's path/id legitimately carries extra\n * tokens the requirement does not name. The requirement side drops\n * deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.\n */\nfunction tokenRecall(requirementText: string, candidateText: string): number {\n const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS)\n if (req.size === 0) return 0\n const cand = tokens(candidateText)\n let hit = 0\n for (const t of req) if (cand.has(t)) hit++\n return hit / req.size\n}\n\ninterface Candidate {\n reqIndex: number\n /** Unique key for a produced item — each item satisfies at most one requirement. */\n itemKey: string\n score: number\n evidence: string\n /** Content to correctness-check, or null when the matched item has none. */\n content: string | null\n}\n\nfunction artifactCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n artifacts: Artifact[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n artifacts.forEach((a, i) => {\n if ((a.content ?? '').trim().length < MIN_CONTENT_CHARS) return\n // Match against the artifact CONTENT too, not just its path + kind — a\n // generated view / note whose path is generic still satisfies a requirement\n // when its body covers it (e.g. an OpenUI comparison grounded in the on-file\n // figures). Bounded slice keeps the recall text cheap; MATCH_THRESHOLD holds.\n let score = tokenRecall(\n reqText,\n `${a.path ?? ''} ${a.kind} ${(a.content ?? '').slice(0, 4000)}`,\n )\n if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) {\n score = Math.max(score, 1)\n }\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `artifact:${i}`,\n score,\n evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,\n content: a.content ?? null,\n })\n })\n return out\n}\n\nfunction proposalCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n proposals: ProducedProposal[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n for (const p of proposals) {\n // Pending or rejected work is not a completed deliverable.\n if (p.status !== 'approved') continue\n // A proposal needs an assessable BODY to be a deliverable. A bare title is\n // not completion: correctness cannot be judged on it, so a title-only match\n // would auto-pass the oracle (structurallyPresent && correct===null →\n // satisfied) with no verifiable content. Tool calls are the only\n // legitimately content-less deliverable (`toolCallCandidates`).\n const body = (p.content ?? '').trim()\n if (body.length < MIN_CONTENT_CHARS) continue\n // Match against the body as well as the (often short) title — a refusal /\n // flag / analysis proposal whose title is a label still satisfies a\n // descriptively-worded requirement when its content covers it. MATCH_THRESHOLD\n // + the requirement's distinctive tokens keep an off-topic proposal out;\n // correctness (a SEMANTIC checker, NOT this lexical pass) then judges\n // polarity/fulfilment, so a negation that merely contains the tokens fails.\n // Structural and correctness must use different evidence or the two-stage\n // check collapses to one lexical gate.\n const score = tokenRecall(reqText, `${p.title} ${body}`)\n if (score < MATCH_THRESHOLD) continue\n out.push({\n reqIndex,\n itemKey: `proposal:${p.id}`,\n score,\n evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,\n content: body,\n })\n }\n return out\n}\n\nfunction toolCallCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n toolCalls: string[],\n): Candidate[] {\n const out: Candidate[] = []\n toolCalls.forEach((name, i) => {\n const score = tokenRecall(req.title, name)\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `tool:${i}`,\n score,\n evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,\n content: null,\n })\n })\n return out\n}\n\n/**\n * Verify whether a run completed the task. `checkCorrectness` is injected —\n * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.\n *\n * Throws on a gold spec with no requirements: an eval task that requires\n * nothing is a misconfiguration, not a vacuously-complete task.\n */\nexport async function verifyCompletion(\n gold: TaskGold,\n state: ProducedState,\n checkCorrectness: CorrectnessChecker,\n): Promise<CompletionVerdict> {\n if (gold.requirements.length === 0) {\n throw new Error(\n `verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`,\n )\n }\n\n // Collect every above-threshold (requirement, produced-item) candidate, then\n // assign greedily by descending score: each requirement and each produced\n // item is used at most once. One deliverable fulfils one requirement.\n const candidates: Candidate[] = []\n gold.requirements.forEach((req, i) => {\n const by = req.satisfiedBy ?? 'any'\n if (by === 'artifact' || by === 'any') {\n candidates.push(...artifactCandidates(req, i, state.artifacts))\n }\n if (by === 'proposal' || by === 'any') {\n candidates.push(...proposalCandidates(req, i, state.proposals))\n }\n if (by === 'tool-call' || by === 'any') {\n candidates.push(...toolCallCandidates(req, i, state.toolCalls))\n }\n })\n candidates.sort((a, b) => b.score - a.score)\n\n const assigned = new Map<number, Candidate>()\n const itemTaken = new Set<string>()\n for (const c of candidates) {\n if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue\n assigned.set(c.reqIndex, c)\n itemTaken.add(c.itemKey)\n }\n\n const requirements: RequirementCheck[] = []\n for (let i = 0; i < gold.requirements.length; i++) {\n const req = gold.requirements[i]!\n const match = assigned.get(i)\n const evidence: string[] = []\n let correct: boolean | null = null\n let unmeasuredReason: string | undefined\n\n if (match) {\n evidence.push(match.evidence)\n if (match.content !== null) {\n try {\n const r = await checkCorrectness(req, match.content)\n correct = r.correct\n evidence.push(`correctness: ${r.correct ? 'pass' : 'fail'} — ${r.reason}`)\n } catch (err) {\n // The CHECKER failed, not the requirement. Recording this as a\n // zero would fabricate a model failure out of an infrastructure\n // one; the requirement is unmeasured and leaves the denominator.\n unmeasuredReason =\n err instanceof JudgeParseError\n ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}`\n : `checker call failed: ${err instanceof Error ? err.message : String(err)}`\n evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`)\n }\n } else {\n evidence.push('correctness: not assessed — matched item carries no content')\n }\n } else {\n const by = req.satisfiedBy ?? 'any'\n const kind = by === 'any' ? 'artifact/proposal/tool-call' : by\n evidence.push(`no produced ${kind} matched this requirement`)\n }\n\n const structurallyPresent = match !== undefined\n const unmeasured = unmeasuredReason !== undefined\n const satisfied = structurallyPresent && !unmeasured && correct !== false\n requirements.push({\n reqId: req.reqId,\n title: req.title,\n structurallyPresent,\n correct,\n satisfied,\n ...(unmeasured ? { unmeasured: true as const, unmeasuredReason } : {}),\n evidence,\n })\n }\n\n // The verdict blends the deterministic structural stage with the injected\n // correctness stage, so the certification names the checker's own member\n // and carries the structural stage's lexical nature as an assumption. A\n // checker without an attestation yields an uncertified verdict.\n const attestation = checkCorrectness.attestation\n const certification: VerdictCertification | undefined = attestation\n ? {\n strategy: attestation.strategy,\n checker: attestation.checker,\n assumptions: [\n 'structural matching is lexical token recall over produced items — the correctness stage only sees items it matched',\n ...attestation.assumptions,\n ],\n evidenceDigest: certificationEvidenceDigest({ taskId: gold.taskId, requirements }),\n }\n : undefined\n\n return completionVerdict({\n taskId: gold.taskId,\n requirements,\n ...(certification === undefined ? {} : { certification }),\n })\n}\n\nexport interface LlmCorrectnessCheckerOpts {\n model?: string\n /** Optional ledger for direct use. */\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /** Max chars of artifact content sent to the checker. */\n maxContentChars?: number\n /**\n * Checker LLM calls per requirement before giving up (parse failures and\n * call errors both consume attempts). The failure then surfaces as an\n * `unmeasured` requirement, never a zero.\n */\n maxAttempts?: number\n /**\n * Forensic capture of every checker request/response/error — without it a\n * checker failure is unauditable (the agent-turn raws never contain the\n * checker's own calls). Same sink contract as `LlmClient`.\n */\n rawSink?: RawProviderSink\n}\n\n/**\n * Parse the correctness checker's model response. Tolerates a response\n * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the\n * verdict boolean usually lands in the first few tokens, so a recovered\n * prefix with a boolean `correct` is a real measurement, not a guess.\n * Fails loud (JudgeParseError) when no boolean verdict is recoverable.\n */\nexport function parseCorrectnessResponse(raw: string): { correct: boolean; reason: string } {\n const readVerdict = (candidate: unknown): { correct: boolean; reason: string } | null => {\n if (candidate === null || typeof candidate !== 'object') return null\n const { correct, reason } = candidate as { correct?: unknown; reason?: unknown }\n if (typeof correct !== 'boolean') return null\n return { correct, reason: typeof reason === 'string' ? reason : '' }\n }\n\n const match = raw.match(/\\{[\\s\\S]*\\}/)\n if (match) {\n try {\n const strict = readVerdict(JSON.parse(match[0]))\n if (strict) return strict\n } catch {\n // fall through to truncation recovery\n }\n }\n // The strict path needs a closing `}`; a cap-hit response has none. Take\n // everything from the first `{` and auto-close it.\n const start = raw.indexOf('{')\n if (start !== -1) {\n const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)))\n if (recovered) return recovered\n }\n throw new JudgeParseError('correctness-checker', raw)\n}\n\n/**\n * Production `CorrectnessChecker` — one LLM call per matched artifact,\n * deterministic (temperature 0), structured JSON out. Judges fulfilment\n * only: a plan, a gesture, or a description of what should be done does not\n * fulfil a requirement — the artifact must BE the deliverable.\n */\nexport function createLlmCorrectnessChecker(\n chat: ChatClient,\n opts: LlmCorrectnessCheckerOpts = {},\n): CorrectnessChecker {\n const model = opts.model ?? 'claude-sonnet-4-6'\n const maxContentChars = opts.maxContentChars ?? 8000\n const maxAttempts = opts.maxAttempts ?? 2\n const costLedger = opts.costLedger ?? new CostLedger()\n const sink = opts.rawSink\n const record = async (event: RawProviderEvent): Promise<void> => {\n // Forensic capture is best-effort; the verdict is the system of record.\n try {\n await sink?.record(event)\n } catch {\n // Intentionally swallowed.\n }\n }\n const checker: CorrectnessChecker = async (requirement, content) => {\n const request = {\n model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}.',\n },\n {\n role: 'user' as const,\n content: `Requirement: ${requirement.title}\\n${\n requirement.category ? `Category: ${requirement.category}\\n` : ''\n }\\nProduced artifact:\\n${content.slice(0, maxContentChars)}`,\n },\n ],\n temperature: 0,\n maxTokens: 200,\n } satisfies ChatRequest\n let lastErr: unknown\n for (let attempt = 0; attempt < maxAttempts; attempt++) {\n const started = Date.now()\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'request',\n timestamp: started,\n requestBody: request,\n redactedFields: [],\n })\n try {\n const paid = await costLedger.runPaidCall({\n channel: 'verifier',\n phase: opts.costPhase ?? 'completion.correctness',\n actor: 'correctness-checker',\n model,\n maximumCharge:\n chat.maximumAttempts === undefined\n ? undefined\n : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),\n tags: {\n ...opts.costTags,\n requirementId: requirement.reqId,\n attempt: String(attempt),\n },\n signal: opts.signal,\n execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),\n receipt: costReceiptFromLlm,\n receiptFromError: costReceiptFromLlmError,\n })\n if (!paid.succeeded) throw paid.error\n const resp = paid.value\n // Hold the transport to its own word: a verdict produced by a model\n // other than the one requested is not this model's verdict. A\n // transport that echoes no id cannot be made to prove identity here —\n // callers needing that proof enable it at the client\n // (`LlmClientOptions.assertServedModel`) or gate on `assertModelsServed`.\n assertServedModel(model, resp.servedModel, {\n allowUnreported: true,\n context: `correctness checker for requirement ${requirement.reqId}`,\n })\n const raw = resp.content\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'response',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n responseBody: resp,\n redactedFields: [],\n })\n return parseCorrectnessResponse(raw)\n } catch (err) {\n lastErr = err\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n // Substitution is a verdict about identity, not a transient fault:\n // another attempt reaches the same wrong model and spends more tokens.\n if (err instanceof ModelSubstitutionError) throw err\n }\n }\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n }\n checker.attestation = {\n strategy: 'judge',\n checker: { name: 'llm-correctness-checker', version: model },\n assumptions: [\n 'served-model identity is accepted when the transport reports none (assertServedModel allowUnreported)',\n ],\n }\n return checker\n}\n\n/** Stopwords for requirement-title tokenization — drops the imperative verbs\n * ('review', 'update', …) common to deliverable titles so recall keys on the\n * substantive nouns, not the boilerplate ask. */\nconst TITLE_STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'and',\n 'or',\n 'for',\n 'to',\n 'of',\n 'in',\n 'on',\n 'with',\n 'review',\n 'update',\n 'new',\n 'proposed',\n])\n\n/**\n * Deterministic `CorrectnessChecker` — the no-LLM counterpart to\n * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its\n * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`\n * of the requirement title's significant tokens. No network.\n *\n * Polarity-blind: token recall credits a negation that contains the\n * requirement's tokens (\"I will NOT produce the comparison\" recalls every token\n * of \"produce the comparison\"). The structural match stage is ALSO lexical, so\n * pairing the two collapses to a single gameable gate. Use this only as an\n * opt-in structural pre-filter or for tasks whose requirements have no polarity\n * to invert; for produced-state grading the correctness checker MUST be semantic\n * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.\n */\nexport function createTokenRecallChecker(\n opts: { minRecall?: number; minContentLength?: number } = {},\n): CorrectnessChecker {\n const minRecall = opts.minRecall ?? 0.5\n const minLen = opts.minContentLength ?? 120\n const checker: CorrectnessChecker = async (requirement, content) => {\n const body = content.trim()\n if (body.length < minLen)\n return {\n correct: false,\n reason: `content too thin (${body.length} chars) to be the deliverable`,\n }\n const titleTokens = requirement.title\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t))\n if (titleTokens.length === 0)\n return {\n correct: true,\n reason: 'requirement title has no significant tokens — structural match accepted',\n }\n const lower = body.toLowerCase()\n const hits = titleTokens.filter((t) => lower.includes(t)).length\n const recall = hits / titleTokens.length\n return recall >= minRecall\n ? {\n correct: true,\n reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`,\n }\n : {\n correct: false,\n reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`,\n }\n }\n // 'schema': token recall checks shape, not meaning — the member whose\n // documented failure mode (\"a well-formed wrong answer passes\") is\n // exactly this checker's polarity blindness.\n checker.attestation = {\n strategy: 'schema',\n checker: { name: 'token-recall-checker', version: packageVersion() },\n assumptions: ['polarity-blind: a negation that recalls the requirement tokens passes'],\n }\n return checker\n}\n","/**\n * Produced-state extraction — normalize a run's runtime event stream into the\n * typed `ProducedState` the completion oracle consumes.\n *\n * `ProducedState` answers \"what did the agent actually produce\" — vault\n * artifacts, proposals, tool calls. The runtime emits these as a stream of\n * events; this module is the single normalization point from that stream to\n * the shape `verifyCompletion` expects.\n *\n * Input is structurally typed (`RuntimeEventLike`) so this module does not\n * depend on agent-runtime — agent-runtime's `RuntimeStreamEvent` satisfies it\n * structurally. The `content` on `ArtifactEventLike` and the whole\n * `proposal_created` variant are the runtime-side enrichments this contract\n * requires; the runtime emits them, this module consumes them.\n */\n\nimport type { Artifact } from './artifact-validator'\nimport type { ProducedProposal, ProducedState } from './completion-verifier'\n\n/** A tool the agent invoked. */\nexport interface ToolCallEventLike {\n type: 'tool_call'\n toolName: string\n}\n\n/**\n * An artifact the agent produced. `content` is the enriched field — the\n * runtime's base `artifact` event carries only metadata; the completion\n * oracle needs the body to verify the deliverable, so the runtime emits it.\n */\nexport interface ArtifactEventLike {\n type: 'artifact'\n artifactId: string\n name?: string\n mimeType?: string\n uri?: string\n content?: string\n}\n\n/** A proposal / filing the agent created. */\nexport interface ProposalEventLike {\n type: 'proposal_created'\n proposalId: string\n title: string\n status?: 'pending' | 'approved' | 'rejected'\n // body of the proposal (e.g. a submit_proposal `description`). When present,\n // the completion oracle correctness-checks it like artifact content; absent,\n // the proposal is graded presence-only.\n content?: string\n}\n\n/**\n * The subset of runtime stream events `extractProducedState` consumes.\n * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;\n * the `{ type: string }` catch-all keeps the input permissive so callers can\n * pass the whole unfiltered telemetry stream — unrecognized events are skipped.\n */\nexport type RuntimeEventLike =\n | ToolCallEventLike\n | ArtifactEventLike\n | ProposalEventLike\n | { type: string }\n\nfunction artifactKind(mimeType: string | undefined): string {\n if (!mimeType) return 'file'\n if (mimeType.includes('json')) return 'json'\n if (mimeType.startsWith('text/')) return 'text'\n return 'file'\n}\n\n/**\n * Normalize a run's runtime event stream into `ProducedState`.\n *\n * Pure and total — unrecognized event types are skipped. `toolCalls` is\n * deduplicated by name in first-seen order (completion cares about a tool's\n * presence, not its call count). An artifact with neither a name nor a uri\n * still yields an entry keyed by its `artifactId` so it is never silently\n * dropped; an artifact with no `content` yields empty content, which the\n * completion oracle's structural check then rejects on its own.\n */\nexport function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState {\n const artifacts: Artifact[] = []\n const proposals: ProducedProposal[] = []\n const toolCalls: string[] = []\n const seenTools = new Set<string>()\n\n for (const ev of events) {\n if (ev.type === 'tool_call') {\n const name = (ev as ToolCallEventLike).toolName\n if (name && !seenTools.has(name)) {\n seenTools.add(name)\n toolCalls.push(name)\n }\n } else if (ev.type === 'artifact') {\n const a = ev as ArtifactEventLike\n artifacts.push({\n kind: artifactKind(a.mimeType),\n path: a.name ?? a.uri ?? a.artifactId,\n content: a.content ?? '',\n })\n } else if (ev.type === 'proposal_created') {\n const p = ev as ProposalEventLike\n proposals.push({\n id: p.proposalId,\n title: p.title,\n status: p.status ?? 'pending',\n ...(p.content !== undefined ? { content: p.content } : {}),\n })\n }\n }\n\n return { artifacts, proposals, toolCalls }\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAa,mBAA2C;CACtD;CACA;CACA;CACA;AACF;;;;;;AAwBA,MAAa,uBAAuB;;;;;;;;;;;;;;;;;;;AAoBpC,SAAgB,kBAAkB,MAAuC;CACvE,MAAM,YAAY,KAAK,aAAa;CACpC,IAAI,UAAU,WAAW,GAAG,MAAM,IAAI,gBAAgB,0CAA0C;CAChG,MAAM,YAAY,KAAK,KAAK,OAAO;CACnC,MAAM,SAAS,KAAK,WAAW,YAAY,CAAC,SAAS,IAAI,CAAC;CAC1D,IAAI,OAAO,WAAW,GACpB,MAAM,IAAI,gBACR,kGACF;CAEF,MAAM,MAAsB,CAAC;CAC7B,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,WAAW,WAAW;EAS/B,MAAM,YAAY,KAAK,mBACnB,SACA,OAAO,QAAQ,UAAU,qBAAqB,SAAS,KAAK,CAAC;EACjE,MAAM,YAAY,UAAU,SAAS,IAAI,YAAY,CAAC,oBAAoB;EAC1E,KAAK,MAAM,SAAS,WAAW;GAC7B,MAAM,UAAwB;IAC5B,GAAG,KAAK;IACR;IACA,MAAM,GAAG,KAAK,KAAK,QAAQ,QAAQ,GAAG,QAAQ,GAAG;IACjD,OAAO;KAAE,GAAG,KAAK,KAAK;KAAO,SAAS;IAAM;IAC5C,UAAU;KAAE,GAAI,KAAK,KAAK,YAAY,CAAC;KAAI;KAAS,cAAc;IAAM;GAC1E;GACA,MAAM,KAAK,eAAe,OAAO;GACjC,IAAI,KAAK,IAAI,EAAE,GAAG;GAClB,KAAK,IAAI,EAAE;GACX,IAAI,KAAK,OAAO;EAClB;CACF;CACA,IAAI,IAAI,WAAW,GAGjB,MAAM,IAAI,gBACR,uDAAuD,UAAU,KAAK,IAAI,EAAE,aAAa,OAAO,KAAK,IAAI,EAAE,IAC7G;CAEF,OAAO;AACT;;;;;;;;AASA,SAAgB,cACd,SACqD;CACrD,MAAM,IAAI,QAAQ;CAClB,MAAM,UAAU,GAAG;CACnB,MAAM,QAAQ,GAAG;CACjB,IAAI,OAAO,YAAY,YAAY,OAAO,UAAU,UAClD,OAAO;EAAW;EAAwB;CAAM;AAGpD;;;;;;;;AASA,SAAgB,eAAe,SAA+B;CAE5D,OAAO,GADO,qBAAqB,yBAAyB,OAAO,CAAC,KAAK,UACzD,GAAG,iBAAiB,OAAO,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1D;;;;;AAMA,SAAgB,oBAAoB,SAA+B;CACjE,MAAM,QAAQ,QAAQ,OAAO,SAAS,KAAK;CAC3C,IAAI,CAAC,OAEH,MAAM,IAAI,gBACR,iBAFY,yBAAyB,OAAO,KAAK,kBAE1B,gDACzB;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,SAA2C;CAC3E,OAAO,QAAQ,MAAM,KAAK,KAAK,QAAQ,SAAS,KAAK,KAAK,KAAA;AAC5D;AAEA,SAAS,qBAAqB,OAA+C;CAM3E,OALa,OACT,KAAK,CAAC,CACP,QAAQ,qBAAqB,GAAG,CAAC,CACjC,QAAQ,OAAO,GAAG,CAAC,CACnB,QAAQ,UAAU,EAAE,KACR,KAAA;AACjB;AAEA,SAAS,QAA2C,OAAsB;CACxE,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,KAAK,GAC7C,IAAI,UAAU,KAAA,GAAW,IAAI,OAAO;CAEtC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,iBAAiB,SAA+B;CAC9D,MAAM,QAAQ,oBAAoB,OAAO;CACzC,MAAM,YAAY,QAAQ;EACxB,GAAG;EACH,MAAM,KAAA;EACN,aAAa,KAAA;EACb,MAAM,QAAQ,OAAO,CAAC,GAAG,QAAQ,IAAI,CAAC,CAAC,KAAK,IAAI,KAAA;EAChD,OAAO,QAAQ;GAAE,GAAG,QAAQ;GAAO,SAAS;EAAM,CAAC;CACrD,CAAC;CACD,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,gBAAgB,SAAS,CAAC,CAAC,CAAC,OAAO,KAAK;AAC7E;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACtEA,SAAgB,kBAAkB,OAMZ;CACpB,IAAI,MAAM,aAAa,WAAW,GAChC,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,+DAC3C;CAEF,MAAM,aAAa,MAAM,aAAa,QAAQ,MAAM,CAAC,EAAE,UAAU;CACjE,MAAM,kBAAkB,MAAM,aAAa,SAAS,WAAW;CAC/D,IAAI,WAAW,WAAW,GAGxB,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,yCAAyC,MAAM,aAAa,OAAO,8BAA8B,MAAM,aAAa,EAAE,EAAE,oBAAoB,iBAAiB,EACxM;CAEF,MAAM,iBAAiB,WAAW,QAAQ,MAAM,EAAE,SAAS,CAAC,CAAC;CAC7D,MAAM,iBAAiB,iBAAiB,WAAW;CAGnD,MAAM,gBAAgB,oBAAoB,KAAK,mBAAmB,WAAW;CAC7E,OAAO;EACL,QAAQ,MAAM;EACd,cAAc,MAAM;EACpB;EACA;EACA;EACA,OAAO;EACP,OAAO;EACP,GAAI,MAAM,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,eAAe,MAAM,cAAc;CACpF;AACF;AA+BA,MAAM,4BAAY,IAAI,IAAI;CACxB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAUD,MAAM,6CAA6B,IAAI,IAAI;CACzC;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAM,kBAAkB;AACxB,MAAM,oBAAoB;AAE1B,SAAS,OAAO,GAAW,WAAsC;CAC/D,OAAO,IAAI,IACT,EACG,YAAY,CAAC,CACb,MAAM,YAAY,CAAC,CACnB,QAAQ,MAAM,EAAE,SAAS,KAAK,CAAC,UAAU,IAAI,CAAC,KAAK,CAAC,WAAW,IAAI,CAAC,CAAC,CAC1E;AACF;;;;;;;AAQA,SAAS,YAAY,iBAAyB,eAA+B;CAC3E,MAAM,MAAM,OAAO,iBAAiB,0BAA0B;CAC9D,IAAI,IAAI,SAAS,GAAG,OAAO;CAC3B,MAAM,OAAO,OAAO,aAAa;CACjC,IAAI,MAAM;CACV,KAAK,MAAM,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC,GAAG;CACtC,OAAO,MAAM,IAAI;AACnB;AAYA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,GAAG,MAAM;EAC1B,KAAK,EAAE,WAAW,GAAA,CAAI,KAAK,CAAC,CAAC,SAAS,mBAAmB;EAKzD,IAAI,QAAQ,YACV,SACA,GAAG,EAAE,QAAQ,GAAG,GAAG,EAAE,KAAK,IAAI,EAAE,WAAW,GAAA,CAAI,MAAM,GAAG,GAAI,GAC9D;EACA,IAAI,IAAI,YAAY,EAAE,QAAQ,IAAI,SAAS,YAAY,MAAM,EAAE,KAAK,YAAY,GAC9E,QAAQ,KAAK,IAAI,OAAO,CAAC;EAE3B,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY;GACrB;GACA,UAAU,aAAa,EAAE,QAAQ,EAAE,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,CAAC;CACD,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,KAAK,MAAM,KAAK,WAAW;EAEzB,IAAI,EAAE,WAAW,YAAY;EAM7B,MAAM,QAAQ,EAAE,WAAW,GAAA,CAAI,KAAK;EACpC,IAAI,KAAK,SAAS,mBAAmB;EASrC,MAAM,QAAQ,YAAY,SAAS,GAAG,EAAE,MAAM,GAAG,MAAM;EACvD,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY,EAAE;GACvB;GACA,UAAU,sBAAsB,EAAE,MAAM,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS;EACX,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,MAAM,MAAM;EAC7B,MAAM,QAAQ,YAAY,IAAI,OAAO,IAAI;EACzC,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,QAAQ;GACjB;GACA,UAAU,cAAc,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACxE,SAAS;EACX,CAAC;CACH,CAAC;CACD,OAAO;AACT;;;;;;;;AASA,eAAsB,iBACpB,MACA,OACA,kBAC4B;CAC5B,IAAI,KAAK,aAAa,WAAW,GAC/B,MAAM,IAAI,MACR,2BAA2B,KAAK,OAAO,4CACzC;CAMF,MAAM,aAA0B,CAAC;CACjC,KAAK,aAAa,SAAS,KAAK,MAAM;EACpC,MAAM,KAAK,IAAI,eAAe;EAC9B,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,eAAe,OAAO,OAC/B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;CAElE,CAAC;CACD,WAAW,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;CAE3C,MAAM,2BAAW,IAAI,IAAuB;CAC5C,MAAM,4BAAY,IAAI,IAAY;CAClC,KAAK,MAAM,KAAK,YAAY;EAC1B,IAAI,SAAS,IAAI,EAAE,QAAQ,KAAK,UAAU,IAAI,EAAE,OAAO,GAAG;EAC1D,SAAS,IAAI,EAAE,UAAU,CAAC;EAC1B,UAAU,IAAI,EAAE,OAAO;CACzB;CAEA,MAAM,eAAmC,CAAC;CAC1C,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,aAAa,QAAQ,KAAK;EACjD,MAAM,MAAM,KAAK,aAAa;EAC9B,MAAM,QAAQ,SAAS,IAAI,CAAC;EAC5B,MAAM,WAAqB,CAAC;EAC5B,IAAI,UAA0B;EAC9B,IAAI;EAEJ,IAAI,OAAO;GACT,SAAS,KAAK,MAAM,QAAQ;GAC5B,IAAI,MAAM,YAAY,MACpB,IAAI;IACF,MAAM,IAAI,MAAM,iBAAiB,KAAK,MAAM,OAAO;IACnD,UAAU,EAAE;IACZ,SAAS,KAAK,gBAAgB,EAAE,UAAU,SAAS,OAAO,KAAK,EAAE,QAAQ;GAC3E,SAAS,KAAK;IAIZ,mBACE,eAAe,kBACX,6CAA6C,IAAI,IAAI,MAAM,GAAG,GAAG,MACjE,wBAAwB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAC7E,SAAS,KAAK,6BAA6B,kBAAkB;GAC/D;QAEA,SAAS,KAAK,6DAA6D;EAE/E,OAAO;GACL,MAAM,KAAK,IAAI,eAAe;GAC9B,MAAM,OAAO,OAAO,QAAQ,gCAAgC;GAC5D,SAAS,KAAK,eAAe,KAAK,0BAA0B;EAC9D;EAEA,MAAM,sBAAsB,UAAU,KAAA;EACtC,MAAM,aAAa,qBAAqB,KAAA;EACxC,MAAM,YAAY,uBAAuB,CAAC,cAAc,YAAY;EACpE,aAAa,KAAK;GAChB,OAAO,IAAI;GACX,OAAO,IAAI;GACX;GACA;GACA;GACA,GAAI,aAAa;IAAE,YAAY;IAAe;GAAiB,IAAI,CAAC;GACpE;EACF,CAAC;CACH;CAMA,MAAM,cAAc,iBAAiB;CACrC,MAAM,gBAAkD,cACpD;EACE,UAAU,YAAY;EACtB,SAAS,YAAY;EACrB,aAAa,CACX,sHACA,GAAG,YAAY,WACjB;EACA,gBAAgB,4BAA4B;GAAE,QAAQ,KAAK;GAAQ;EAAa,CAAC;CACnF,IACA,KAAA;CAEJ,OAAO,kBAAkB;EACvB,QAAQ,KAAK;EACb;EACA,GAAI,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,cAAc;CACzD,CAAC;AACH;;;;;;;;AAgCA,SAAgB,yBAAyB,KAAmD;CAC1F,MAAM,eAAe,cAAoE;EACvF,IAAI,cAAc,QAAQ,OAAO,cAAc,UAAU,OAAO;EAChE,MAAM,EAAE,SAAS,WAAW;EAC5B,IAAI,OAAO,YAAY,WAAW,OAAO;EACzC,OAAO;GAAE;GAAS,QAAQ,OAAO,WAAW,WAAW,SAAS;EAAG;CACrE;CAEA,MAAM,QAAQ,IAAI,MAAM,aAAa;CACrC,IAAI,OACF,IAAI;EACF,MAAM,SAAS,YAAY,KAAK,MAAM,MAAM,EAAE,CAAC;EAC/C,IAAI,QAAQ,OAAO;CACrB,QAAQ,CAER;CAIF,MAAM,QAAQ,IAAI,QAAQ,GAAG;CAC7B,IAAI,UAAU,IAAI;EAChB,MAAM,YAAY,YAAY,qBAAqB,IAAI,MAAM,KAAK,CAAC,CAAC;EACpE,IAAI,WAAW,OAAO;CACxB;CACA,MAAM,IAAI,gBAAgB,uBAAuB,GAAG;AACtD;;;;;;;AAQA,SAAgB,4BACd,MACA,OAAkC,CAAC,GACf;CACpB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,cAAc,KAAK,eAAe;CACxC,MAAM,aAAa,KAAK,cAAc,IAAI,WAAW;CACrD,MAAM,OAAO,KAAK;CAClB,MAAM,SAAS,OAAO,UAA2C;EAE/D,IAAI;GACF,MAAM,MAAM,OAAO,KAAK;EAC1B,QAAQ,CAER;CACF;CACA,MAAM,UAA8B,OAAO,aAAa,YAAY;EAClE,MAAM,UAAU;GACd;GACA,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IACE,MAAM;IACN,SAAS,gBAAgB,YAAY,MAAM,IACzC,YAAY,WAAW,aAAa,YAAY,SAAS,MAAM,GAChE,wBAAwB,QAAQ,MAAM,GAAG,eAAe;GAC3D,CACF;GACA,aAAa;GACb,WAAW;EACb;EACA,IAAI;EACJ,KAAK,IAAI,UAAU,GAAG,UAAU,aAAa,WAAW;GACtD,MAAM,UAAU,KAAK,IAAI;GACzB,MAAM,OAAO;IACX,SAAS,WAAW;IACpB,UAAU,KAAK;IACf;IACA,UAAU;IACV,SAAS;IACT,cAAc;IACd,WAAW;IACX,WAAW;IACX,aAAa;IACb,gBAAgB,CAAC;GACnB,CAAC;GACD,IAAI;IACF,MAAM,OAAO,MAAM,WAAW,YAAY;KACxC,SAAS;KACT,OAAO,KAAK,aAAa;KACzB,OAAO;KACP;KACA,eACE,KAAK,oBAAoB,KAAA,IACrB,KAAA,IACA,2BAA2B,SAAS,EAAE,iBAAiB,KAAK,gBAAgB,CAAC;KACnF,MAAM;MACJ,GAAG,KAAK;MACR,eAAe,YAAY;MAC3B,SAAS,OAAO,OAAO;KACzB;KACA,QAAQ,KAAK;KACb,UAAU,QAAQ,WAAW,KAAK,KAAK,SAAS;MAAE;MAAQ,gBAAgB;KAAO,CAAC;KAClF,SAAS;KACT,kBAAkB;IACpB,CAAC;IACD,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;IAChC,MAAM,OAAO,KAAK;IAMlB,kBAAkB,OAAO,KAAK,aAAa;KACzC,iBAAiB;KACjB,SAAS,uCAAuC,YAAY;IAC9D,CAAC;IACD,MAAM,MAAM,KAAK;IACjB,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc;KACd,gBAAgB,CAAC;IACnB,CAAC;IACD,OAAO,yBAAyB,GAAG;GACrC,SAAS,KAAK;IACZ,UAAU;IACV,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;KAC7D,gBAAgB,CAAC;IACnB,CAAC;IAGD,IAAI,eAAe,wBAAwB,MAAM;GACnD;EACF;EACA,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;CACtE;CACA,QAAQ,cAAc;EACpB,UAAU;EACV,SAAS;GAAE,MAAM;GAA2B,SAAS;EAAM;EAC3D,aAAa,CACX,uGACF;CACF;CACA,OAAO;AACT;;;ACpmBA,SAAS,aAAa,UAAsC;CAC1D,IAAI,CAAC,UAAU,OAAO;CACtB,IAAI,SAAS,SAAS,MAAM,GAAG,OAAO;CACtC,IAAI,SAAS,WAAW,OAAO,GAAG,OAAO;CACzC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,qBAAqB,QAAoD;CACvF,MAAM,YAAwB,CAAC;CAC/B,MAAM,YAAgC,CAAC;CACvC,MAAM,YAAsB,CAAC;CAC7B,MAAM,4BAAY,IAAI,IAAY;CAElC,KAAK,MAAM,MAAM,QACf,IAAI,GAAG,SAAS,aAAa;EAC3B,MAAM,OAAQ,GAAyB;EACvC,IAAI,QAAQ,CAAC,UAAU,IAAI,IAAI,GAAG;GAChC,UAAU,IAAI,IAAI;GAClB,UAAU,KAAK,IAAI;EACrB;CACF,OAAO,IAAI,GAAG,SAAS,YAAY;EACjC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,MAAM,aAAa,EAAE,QAAQ;GAC7B,MAAM,EAAE,QAAQ,EAAE,OAAO,EAAE;GAC3B,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,OAAO,IAAI,GAAG,SAAS,oBAAoB;EACzC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,IAAI,EAAE;GACN,OAAO,EAAE;GACT,QAAQ,EAAE,UAAU;GACpB,GAAI,EAAE,YAAY,KAAA,IAAY,EAAE,SAAS,EAAE,QAAQ,IAAI,CAAC;EAC1D,CAAC;CACH;CAGF,OAAO;EAAE;EAAW;EAAW;CAAU;AAC3C"}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { R as Scenario, h as GateContext, p as Gate, v as GateResult } from "./types-JHMOqZI4.js";
|
|
2
1
|
import { d as PairedBootstrapResult, i as PairedMcNemarEvidence, r as PairedDecisionStatistic, t as PairedDecisionMethod } from "./paired-promotion-decision-CGzg0cI_.js";
|
|
2
|
+
import { R as Scenario, h as GateContext, p as Gate, v as GateResult } from "./types-nokrtr7M.js";
|
|
3
3
|
import { i as Direction } from "./power-preflight-Ptse_Kq7.js";
|
|
4
4
|
//#region src/campaign/gates/promotion-policy.d.ts
|
|
5
5
|
/** Where an objective's per-cell scalar comes from. `composite` reads the
|
|
@@ -131,4 +131,4 @@ interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
|
|
|
131
131
|
declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
|
|
132
132
|
//#endregion
|
|
133
133
|
export { ObjectiveSource as a, PromotionPolicy as c, paretoSignificanceGate as d, EvidenceVector as i, buildEvidenceVector as l, AxisVerdict as n, ParetoSignificanceGateOptions as o, BuildEvidenceVectorOptions as r, PromotionObjective as s, AxisEvidence as t, paretoPolicy as u };
|
|
134
|
-
//# sourceMappingURL=promotion-policy-
|
|
134
|
+
//# sourceMappingURL=promotion-policy-BBBcz5_3.d.ts.map
|