@tangle-network/agent-eval 0.145.6 → 0.145.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (121) hide show
  1. package/CHANGELOG.md +10 -0
  2. package/README.md +4 -0
  3. package/dist/analyst/index.d.ts +10 -10
  4. package/dist/analyst/index.js +1 -1
  5. package/dist/{backend-integrity-CsVin_Wb.d.ts → backend-integrity-BffcHGdm.d.ts} +2 -2
  6. package/dist/{backend-integrity-CsVin_Wb.d.ts.map → backend-integrity-BffcHGdm.d.ts.map} +1 -1
  7. package/dist/{benchmark-Ceoan7vk.d.ts → benchmark-DbiZbPdH.d.ts} +3 -3
  8. package/dist/{benchmark-Ceoan7vk.d.ts.map → benchmark-DbiZbPdH.d.ts.map} +1 -1
  9. package/dist/{benchmark-command-DspwA7cv.js → benchmark-command-C7L0rQsD.js} +2 -2
  10. package/dist/{benchmark-command-DspwA7cv.js.map → benchmark-command-C7L0rQsD.js.map} +1 -1
  11. package/dist/benchmarks/index.d.ts +5 -5
  12. package/dist/benchmarks/index.js +2 -2
  13. package/dist/campaign/index.d.ts +7 -7
  14. package/dist/campaign/index.js +3 -3
  15. package/dist/{campaign-jTOvqnse.js → campaign-DRcGtxEf.js} +35 -11
  16. package/dist/campaign-DRcGtxEf.js.map +1 -0
  17. package/dist/{capture-fetch-DDvpjVRU.d.ts → capture-fetch-hykwHfDI.d.ts} +2 -2
  18. package/dist/{capture-fetch-DDvpjVRU.d.ts.map → capture-fetch-hykwHfDI.d.ts.map} +1 -1
  19. package/dist/cli.js +1 -1
  20. package/dist/{client-DCVe0CwG.d.ts → client-D_TIV9pJ.d.ts} +4 -4
  21. package/dist/{client-DCVe0CwG.d.ts.map → client-D_TIV9pJ.d.ts.map} +1 -1
  22. package/dist/contract/index.d.ts +11 -11
  23. package/dist/contract/index.js +4 -4
  24. package/dist/{default-registry-CgzxnJsj.d.ts → default-registry-cpHtP2Og.d.ts} +6 -6
  25. package/dist/{default-registry-CgzxnJsj.d.ts.map → default-registry-cpHtP2Og.d.ts.map} +1 -1
  26. package/dist/{define-agent-eval-DLhZ8IHi.d.ts → define-agent-eval-DBOpAM_g.d.ts} +6 -6
  27. package/dist/{define-agent-eval-DLhZ8IHi.d.ts.map → define-agent-eval-DBOpAM_g.d.ts.map} +1 -1
  28. package/dist/{define-agent-eval-CViDh2P9.js → define-agent-eval-m5lM7XGE.js} +4 -4
  29. package/dist/{define-agent-eval-CViDh2P9.js.map → define-agent-eval-m5lM7XGE.js.map} +1 -1
  30. package/dist/{engine-FRZ3RLvT.d.ts → engine-DJqRKbhs.d.ts} +7 -7
  31. package/dist/{engine-FRZ3RLvT.d.ts.map → engine-DJqRKbhs.d.ts.map} +1 -1
  32. package/dist/{eval-campaign-UB-usSQ2.js → eval-campaign-aNqpefCS.js} +2 -2
  33. package/dist/{eval-campaign-UB-usSQ2.js.map → eval-campaign-aNqpefCS.js.map} +1 -1
  34. package/dist/{exact-types-BH1twmAJ.d.ts → exact-types-CzbhVDr2.d.ts} +2 -2
  35. package/dist/{exact-types-BH1twmAJ.d.ts.map → exact-types-CzbhVDr2.d.ts.map} +1 -1
  36. package/dist/experiment/index.d.ts +3 -3
  37. package/dist/{feedback-trajectory-juOozjAc.d.ts → feedback-trajectory-D9vYSob_.d.ts} +3 -3
  38. package/dist/{feedback-trajectory-juOozjAc.d.ts.map → feedback-trajectory-D9vYSob_.d.ts.map} +1 -1
  39. package/dist/hosted/index.d.ts +3 -3
  40. package/dist/index-BKShPTwZ.d.ts +1 -0
  41. package/dist/{index-COQYtuRF.d.ts → index-BNPtkBPf.d.ts} +2 -2
  42. package/dist/{index-COQYtuRF.d.ts.map → index-BNPtkBPf.d.ts.map} +1 -1
  43. package/dist/{index-DF3ynqmJ.d.ts → index-D7gl974A.d.ts} +11 -11
  44. package/dist/{index-DF3ynqmJ.d.ts.map → index-D7gl974A.d.ts.map} +1 -1
  45. package/dist/{index-PgqfwhsM.d.ts → index-YrUFx3FU.d.ts} +5 -5
  46. package/dist/{index-PgqfwhsM.d.ts.map → index-YrUFx3FU.d.ts.map} +1 -1
  47. package/dist/index.d.ts +22 -22
  48. package/dist/index.js +9 -9
  49. package/dist/{insight-report-CRi-Ufrj.d.ts → insight-report-BeT8KCgI.d.ts} +3 -3
  50. package/dist/{insight-report-CRi-Ufrj.d.ts.map → insight-report-BeT8KCgI.d.ts.map} +1 -1
  51. package/dist/{llm-judge-DuYa4SEA.js → llm-judge-BO1LGjdn.js} +2 -2
  52. package/dist/{llm-judge-DuYa4SEA.js.map → llm-judge-BO1LGjdn.js.map} +1 -1
  53. package/dist/meta-eval/index.d.ts +1 -1
  54. package/dist/{mint-Dj9Ww_3I.js → mint-BV6tLVWl.js} +2 -2
  55. package/dist/{mint-Dj9Ww_3I.js.map → mint-BV6tLVWl.js.map} +1 -1
  56. package/dist/multishot/index.d.ts +1 -1
  57. package/dist/openapi.json +1 -1
  58. package/dist/{pre-registration-DeRvl9sE.d.ts → pre-registration-zFSLEiFU.d.ts} +2 -2
  59. package/dist/{pre-registration-DeRvl9sE.d.ts.map → pre-registration-zFSLEiFU.d.ts.map} +1 -1
  60. package/dist/{produced-state-D6j7qy1Q.js → produced-state-C6vSgOpE.js} +2 -2
  61. package/dist/{produced-state-D6j7qy1Q.js.map → produced-state-C6vSgOpE.js.map} +1 -1
  62. package/dist/{promotion-policy-CsMZJOB-.d.ts → promotion-policy-u3wj6w3U.d.ts} +2 -2
  63. package/dist/{promotion-policy-CsMZJOB-.d.ts.map → promotion-policy-u3wj6w3U.d.ts.map} +1 -1
  64. package/dist/{provenance-8L-_4xiL.d.ts → provenance-BOtMtoiZ.d.ts} +5 -5
  65. package/dist/{provenance-8L-_4xiL.d.ts.map → provenance-BOtMtoiZ.d.ts.map} +1 -1
  66. package/dist/{registry-oJeeI4-a.d.ts → registry-B_1Frl8a.d.ts} +3 -3
  67. package/dist/{registry-oJeeI4-a.d.ts.map → registry-B_1Frl8a.d.ts.map} +1 -1
  68. package/dist/{release-confidence-BFRE5WSp.d.ts → release-confidence-4XrqlpFD.d.ts} +3 -3
  69. package/dist/{release-confidence-BFRE5WSp.d.ts.map → release-confidence-4XrqlpFD.d.ts.map} +1 -1
  70. package/dist/{release-confidence-CxDuiAev.js → release-confidence-BknrpBnO.js} +2 -2
  71. package/dist/{release-confidence-CxDuiAev.js.map → release-confidence-BknrpBnO.js.map} +1 -1
  72. package/dist/reporting.d.ts +3 -3
  73. package/dist/reporting.js +1 -1
  74. package/dist/{researcher-DaN4GST-.d.ts → researcher-Du-oniHp.d.ts} +3 -3
  75. package/dist/{researcher-DaN4GST-.d.ts.map → researcher-Du-oniHp.d.ts.map} +1 -1
  76. package/dist/{reward-hacking-DSSTuI9r.d.ts → reward-hacking-Bu8ev6PR.d.ts} +2 -2
  77. package/dist/{reward-hacking-DSSTuI9r.d.ts.map → reward-hacking-Bu8ev6PR.d.ts.map} +1 -1
  78. package/dist/{reward-hacking-DNgjilrV.js → reward-hacking-DFo2FU5J.js} +2 -2
  79. package/dist/{reward-hacking-DNgjilrV.js.map → reward-hacking-DFo2FU5J.js.map} +1 -1
  80. package/dist/rl.d.ts +5 -5
  81. package/dist/rl.js +4 -4
  82. package/dist/rollout/index.d.ts +1 -1
  83. package/dist/rollout/index.js +2 -2
  84. package/dist/{rollout-BWtw0I_6.js → rollout-ytVQ7WT8.js} +2 -2
  85. package/dist/{rollout-BWtw0I_6.js.map → rollout-ytVQ7WT8.js.map} +1 -1
  86. package/dist/{rubric-predictive-validity-BgxtKe4G.d.ts → rubric-predictive-validity-C7LnNvF2.d.ts} +2 -2
  87. package/dist/{rubric-predictive-validity-BgxtKe4G.d.ts.map → rubric-predictive-validity-C7LnNvF2.d.ts.map} +1 -1
  88. package/dist/{run-record-BvHPVS-i.js → run-record-D2lDdSAz.js} +5 -3
  89. package/dist/run-record-D2lDdSAz.js.map +1 -0
  90. package/dist/{run-record-CKiihE6f.d.ts → run-record-DVV82Gwh.d.ts} +8 -3
  91. package/dist/run-record-DVV82Gwh.d.ts.map +1 -0
  92. package/dist/{skillopt-optimization-method-CFDSzD-7.d.ts → skillopt-optimization-method-BPQwXkkY.d.ts} +5 -5
  93. package/dist/{skillopt-optimization-method-CFDSzD-7.d.ts.map → skillopt-optimization-method-BPQwXkkY.d.ts.map} +1 -1
  94. package/dist/{skillopt-optimization-method-BTls2l-Z.js → skillopt-optimization-method-Ceic2bfU.js} +2 -2
  95. package/dist/{skillopt-optimization-method-BTls2l-Z.js.map → skillopt-optimization-method-Ceic2bfU.js.map} +1 -1
  96. package/dist/{statistical-heldout-C4De2tRI.d.ts → statistical-heldout-CBGPSZ5X.d.ts} +3 -3
  97. package/dist/{statistical-heldout-C4De2tRI.d.ts.map → statistical-heldout-CBGPSZ5X.d.ts.map} +1 -1
  98. package/dist/{store-tool-spans-CggeC1LB.d.ts → store-tool-spans-C9c4R7ca.d.ts} +4 -4
  99. package/dist/{store-tool-spans-CggeC1LB.d.ts.map → store-tool-spans-C9c4R7ca.d.ts.map} +1 -1
  100. package/dist/{summary-report-CaL-Hnxt.d.ts → summary-report-B__Y5ub3.d.ts} +2 -2
  101. package/dist/{summary-report-CaL-Hnxt.d.ts.map → summary-report-B__Y5ub3.d.ts.map} +1 -1
  102. package/dist/{tool-groups-BdcoEgtV.d.ts → tool-groups-CZotz-e_.d.ts} +3 -3
  103. package/dist/tool-groups-CZotz-e_.d.ts.map +1 -0
  104. package/dist/trace-repair/index.d.ts +2 -2
  105. package/dist/traces.d.ts +6 -6
  106. package/dist/traces.js +1 -1
  107. package/dist/{types-BxLccGMf.d.ts → types-BjsNDR49.d.ts} +3 -3
  108. package/dist/{types-BxLccGMf.d.ts.map → types-BjsNDR49.d.ts.map} +1 -1
  109. package/dist/{types-BRsxjg7z.d.ts → types-C4bSVIr7.d.ts} +3 -3
  110. package/dist/{types-BRsxjg7z.d.ts.map → types-C4bSVIr7.d.ts.map} +1 -1
  111. package/dist/{types-DLQx4mKU.d.ts → types-vXyshMwx.d.ts} +2 -2
  112. package/dist/{types-DLQx4mKU.d.ts.map → types-vXyshMwx.d.ts.map} +1 -1
  113. package/dist/wire/index.d.ts +1 -1
  114. package/docs/eval-surface-map.md +2 -1
  115. package/docs/wire-protocol.md +4 -0
  116. package/package.json +1 -1
  117. package/dist/campaign-jTOvqnse.js.map +0 -1
  118. package/dist/index-Ba3YrbAL.d.ts +0 -1
  119. package/dist/run-record-BvHPVS-i.js.map +0 -1
  120. package/dist/run-record-CKiihE6f.d.ts.map +0 -1
  121. package/dist/tool-groups-BdcoEgtV.d.ts.map +0 -1
@@ -3,7 +3,7 @@ import { s as TraceStore } from "../store-CT9YIIve.js";
3
3
  import { a as ContinuousCalibrationResult, n as CandidateScore, o as GoldenItem, r as ContinuousAgreement, t as CalibrationResult } from "../judge-calibration-C5CbMYce.js";
4
4
  import { n as SeriesConvergenceResult, o as CorpusAgreementReport, t as SeriesConvergenceOptions } from "../series-convergence-D1cL1f-4.js";
5
5
  import { a as OutcomeFilter, i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "../outcome-store-BYHIuO0e.js";
6
- import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "../rubric-predictive-validity-BgxtKe4G.js";
6
+ import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "../rubric-predictive-validity-C7LnNvF2.js";
7
7
  //#region src/meta-eval/correlation-study.d.ts
8
8
  interface EvalMetricSpec {
9
9
  id: string;
@@ -1,6 +1,6 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
2
  import { a as scoreOrigin, i as rolloutRewardFields } from "./reward-nw2xZGZG.js";
3
- import { o as runTaskScore } from "./run-record-BvHPVS-i.js";
3
+ import { s as runTaskScore } from "./run-record-D2lDdSAz.js";
4
4
  import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
5
5
  import { i as ROLLOUT_SCHEMA, s as assertMinted } from "./schema-C6DW4ZHR.js";
6
6
  //#region src/rollout/mint.ts
@@ -315,4 +315,4 @@ async function mintRolloutRows(records, store, options = {}) {
315
315
  //#endregion
316
316
  export { unmintableReasons as n, mintRolloutRows as t };
317
317
 
318
- //# sourceMappingURL=mint-Dj9Ww_3I.js.map
318
+ //# sourceMappingURL=mint-BV6tLVWl.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"mint-Dj9Ww_3I.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
1
+ {"version":3,"file":"mint-BV6tLVWl.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
@@ -1,5 +1,5 @@
1
1
  import { p as CostProvenance } from "../cost-ledger-DbQdN3nO.js";
2
- import { w as JudgeScore } from "../types-BxLccGMf.js";
2
+ import { w as JudgeScore } from "../types-BjsNDR49.js";
3
3
  import { o as MatrixResult } from "../index-CvjYbU0D.js";
4
4
  import { AgentProfile } from "@tangle-network/agent-interface";
5
5
  //#region src/multishot/types.d.ts
package/dist/openapi.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "openapi": "3.1.0",
3
3
  "info": {
4
4
  "title": "@tangle-network/agent-eval — wire protocol",
5
- "version": "0.145.6",
5
+ "version": "0.145.7",
6
6
  "description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
7
7
  "contact": {
8
8
  "name": "Tangle Network",
@@ -1,4 +1,4 @@
1
- import { a as RunRecord } from "./run-record-CKiihE6f.js";
1
+ import { a as RunRecord } from "./run-record-DVV82Gwh.js";
2
2
  import { d as PairedBootstrapResult, u as PairedBootstrapOptions } from "./paired-promotion-decision-CGzg0cI_.js";
3
3
  //#region src/statistics/paired-binary.d.ts
4
4
  /** A binomial proportion estimate with a confidence interval. */
@@ -574,4 +574,4 @@ declare function evaluateHypothesis(manifest: SignedManifest, observed: {
574
574
  }): Promise<HypothesisResult>;
575
575
  //#endregion
576
576
  export { RiskDifferenceResult as A, EProcessOptions as C, ExactRiskDifferenceResult as D, eProcess as E, pairedRiskDifference as F, pairedRiskDifferenceExact as I, pairedRiskDifferenceScore as L, isBinaryOutcomeVector as M, mcnemar as N, McNemarResult as O, pairedBinaryScale as P, passAtK as R, EProcess as S, EProcessStep as T, PairedCorrectness as _, evaluateHypothesis as a, pairArms as b, verifyManifest as c, MatchedRunRecordPair as d, PairArmsOptions as f, PairedArmsComparison as g, PairedArmRow as h, canonicalize as i, ScoreRiskDifferenceResult as j, ProportionInterval as k, ComparePairedArmsOptions as l, PairRunRecordsResult as m, HypothesisResult as n, hashJson as o, PairArmsResult as p, SignedManifest as r, signManifest as s, HypothesisManifest as t, MatchedPair as u, PairedMetricDelta as v, EProcessState as w, pairRunRecords as x, comparePairedArms as y, wilson as z };
577
- //# sourceMappingURL=pre-registration-DeRvl9sE.d.ts.map
577
+ //# sourceMappingURL=pre-registration-zFSLEiFU.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"pre-registration-DeRvl9sE.d.ts","names":[],"sources":["../src/statistics/paired-binary.ts","../src/statistics/sequential-eprocess.ts","../src/paired-arms.ts","../src/pre-registration.ts"],"mappings":";;;;UAgBiB;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;iBA2CzD,sBAAsB,QAAQ;;UAU7B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;iBAkBc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;UAkCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+Bc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;UAyEc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8Dc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;;;;;;;;;;iBA6Ea,kBACd,QAAQ,mBACR,OAAO;;;;;;;;;;;iBA0BO,QAAQ,WAAW,WAAW;;;UCjgB7B;;;EAGf;;;;EAIA;;;;EAIA;;UAGe;;EAEf;;EAEA;;EAEA;;UAGe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;UC/BrC;;;EAGf;;;;;EAKA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;EAEf;;EAEA;;;UAIe;EACf;;;;EAIA;EACA,UAAU;EACV,WAAW;;UAGI;;EAEf,OAAO;;;EAGP,kBAAkB;;EAElB,mBAAmB;;;;;;;;;;;;;;;;;;;;iBAqBL,SAAS,eAAe,gBAAgB,MAAM,kBAAkB;;UA8G/D;;EAEf;;EAEA;;EAEA,SAAS;;EAET,gBAAgB;;;UAID;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,aAAa;;EAEb;IAAY;IAAW;;;UAGR,iCAAiC;;;;EAIhD;;EAEA,YAAY;;UAGG;EACf;EACA;EACA;;;EAGA,aAAa;EACb,cAAc;;;;;;;;;;;;;;;iBAgBA,kBACd,eAAe,gBACf,MAAM,2BACL;UAmEc;EACf;EACA;EACA,UAAU;EACV,WAAW;;UAGI;EACf,OAAO;EACP,kBAAkB;EAClB,mBAAmB;;;;;;;;;iBAeL,eACd,uBAAuB,aACvB,wBAAwB,cACvB;;;;;;;;;;;;;UC1Wc;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;;;;;;;;;;KAWU;UAEK,uBAAuB;;EAEtC;;;;;;;;;EASA,OAAO;;UAGQ;EACf,UAAU;EACV;EACA;EACA;;;EAGA;;EAEA,kBAAkB;EAGlB;;;;;;;;;;;iBAYc,aAAa;;;;;;;;;;;;;;;;;;;;;;iBA8BP,SAAS,GAAG,KAAK,IAAI;;;;;;;;;;iBAkBrB,aAAa,GAAG,qBAAqB,QAAQ;;;;;;;iBAW7C,eAAe,GAAG,iBAAiB;;;;;iBAWnC,mBACpB,UAAU,gBACV;EAAY;EAAW;EAAgB;IACtC,QAAQ"}
1
+ {"version":3,"file":"pre-registration-zFSLEiFU.d.ts","names":[],"sources":["../src/statistics/paired-binary.ts","../src/statistics/sequential-eprocess.ts","../src/paired-arms.ts","../src/pre-registration.ts"],"mappings":";;;;UAgBiB;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;iBA2CzD,sBAAsB,QAAQ;;UAU7B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;iBAkBc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;UAkCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+Bc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;UAyEc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8Dc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;;;;;;;;;;iBA6Ea,kBACd,QAAQ,mBACR,OAAO;;;;;;;;;;;iBA0BO,QAAQ,WAAW,WAAW;;;UCjgB7B;;;EAGf;;;;EAIA;;;;EAIA;;UAGe;;EAEf;;EAEA;;EAEA;;UAGe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;UC/BrC;;;EAGf;;;;;EAKA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;EAEf;;EAEA;;;UAIe;EACf;;;;EAIA;EACA,UAAU;EACV,WAAW;;UAGI;;EAEf,OAAO;;;EAGP,kBAAkB;;EAElB,mBAAmB;;;;;;;;;;;;;;;;;;;;iBAqBL,SAAS,eAAe,gBAAgB,MAAM,kBAAkB;;UA8G/D;;EAEf;;EAEA;;EAEA,SAAS;;EAET,gBAAgB;;;UAID;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,aAAa;;EAEb;IAAY;IAAW;;;UAGR,iCAAiC;;;;EAIhD;;EAEA,YAAY;;UAGG;EACf;EACA;EACA;;;EAGA,aAAa;EACb,cAAc;;;;;;;;;;;;;;;iBAgBA,kBACd,eAAe,gBACf,MAAM,2BACL;UAmEc;EACf;EACA;EACA,UAAU;EACV,WAAW;;UAGI;EACf,OAAO;EACP,kBAAkB;EAClB,mBAAmB;;;;;;;;;iBAeL,eACd,uBAAuB,aACvB,wBAAwB,cACvB;;;;;;;;;;;;;UC1Wc;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;;;;;;;;;;KAWU;UAEK,uBAAuB;;EAEtC;;;;;;;;;EASA,OAAO;;UAGQ;EACf,UAAU;EACV;EACA;EACA;;;EAGA;;EAEA,kBAAkB;EAGlB;;;;;;;;;;;iBAYc,aAAa;;;;;;;;;;;;;;;;;;;;;;iBA8BP,SAAS,GAAG,KAAK,IAAI;;;;;;;;;;iBAkBrB,aAAa,GAAG,qBAAqB,QAAQ;;;;;;;iBAW7C,eAAe,GAAG,iBAAiB;;;;;iBAWnC,mBACpB,UAAU,gBACV;EAAY;EAAW;EAAgB;IACtC,QAAQ"}
@@ -1,6 +1,6 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
2
  import { t as canonicalize } from "./pre-registration-DakwTRXk.js";
3
- import { I as recoverTruncatedJson, n as JudgeParseError } from "./llm-judge-DuYa4SEA.js";
3
+ import { I as recoverTruncatedJson, n as JudgeParseError } from "./llm-judge-BO1LGjdn.js";
4
4
  import { i as CostLedger } from "./cost-ledger-BSe92yAV.js";
5
5
  import { t as certificationEvidenceDigest } from "./verdict-BndeTAh_.js";
6
6
  import { f as maximumChargeForLlmRequest, g as assertServedModel, h as ModelSubstitutionError, l as costReceiptFromLlm, u as costReceiptFromLlmError } from "./llm-client-d0-2TT1g.js";
@@ -583,4 +583,4 @@ function extractProducedState(events) {
583
583
  //#endregion
584
584
  export { CODING_HARNESSES as a, agentProfileId as c, harnessAxisOf as d, verifyCompletion as i, agentProfileModelId as l, completionVerdict as n, HARNESS_NATIVE_MODEL as o, createLlmCorrectnessChecker as r, agentProfileHash as s, extractProducedState as t, expandProfileAxes as u };
585
585
 
586
- //# sourceMappingURL=produced-state-D6j7qy1Q.js.map
586
+ //# sourceMappingURL=produced-state-C6vSgOpE.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"produced-state-D6j7qy1Q.js","names":[],"sources":["../src/agent-profile.ts","../src/completion-verifier.ts","../src/produced-state.ts"],"sourcesContent":["import { createHash } from 'node:crypto'\nimport type { AgentProfile } from '@tangle-network/agent-interface'\nimport { type HarnessType, harnessSupportsModel } from '@tangle-network/agent-interface'\nimport { ValidationError } from './errors'\nimport { canonicalize } from './pre-registration'\n\nexport type { AgentProfile, HarnessType } from '@tangle-network/agent-interface'\n\n/**\n * The agentic coding harnesses an eval sweeps by default — the ones we care about\n * ranking. This is the SINGLE source of that list; consumers import it instead of\n * re-declaring their own (a re-declared list is how the fleet drifts). Pass an\n * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known\n * harness) to widen beyond these.\n */\nexport const CODING_HARNESSES: readonly HarnessType[] = [\n 'opencode',\n 'claude-code',\n 'codex',\n 'kimi-code',\n]\n\nexport interface ProfileAxisSpec {\n /** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the\n * harness and model vary. `model.default` is the fallback model. */\n base: AgentProfile\n /** Harnesses to cross. Default: {@link CODING_HARNESSES}. */\n harnesses?: readonly HarnessType[]\n /** Models to cross. Default: `[base.model.default]` — one model, i.e. today's\n * single-model behaviour, so omitting this never changes an existing run. */\n models?: readonly string[]\n /** Force every (harness, model) pair verbatim, even ones the harness can't run —\n * for deliberately testing failure modes. Default (false): SNAP instead — a\n * vendor-locked harness runs only the swept models in its family, or its native\n * default when it supports none, so no harness is dropped and none gets a\n * guaranteed-failing foreign-model cell. */\n keepIncompatible?: boolean\n}\n\n/** Model sentinel for a vendor-locked harness that supports none of the swept models:\n * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness\n * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi\n * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship\n * table that would rot as router catalogs change. */\nexport const HARNESS_NATIVE_MODEL = 'default'\n\n/**\n * Expand a base profile across the harness × model matrix into the `AgentProfile[]`\n * that `runProfileMatrix` / `selfImprove` score — the ONE place \"which harnesses ×\n * which models do we evaluate\" lives, so no product hand-rolls its own harness list\n * or column→profile mapping (the pattern that let those copies drift and silently\n * break the harness pivot).\n *\n * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,\n * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both\n * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row\n * and results join back by harness/model via {@link harnessAxisOf} with no\n * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or\n * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every\n * requested harness runs; `keepIncompatible` forces every pair verbatim.\n *\n * Omit `harnesses`/`models` to sweep the full default set — the \"turn it on for\n * everything we care about\" switch, identical in shape whether one harness or all.\n */\nexport function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[] {\n const harnesses = spec.harnesses ?? CODING_HARNESSES\n if (harnesses.length === 0) throw new ValidationError('expandProfileAxes: no harnesses to sweep')\n const baseModel = spec.base.model?.default\n const models = spec.models ?? (baseModel ? [baseModel] : [])\n if (models.length === 0) {\n throw new ValidationError(\n 'expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied',\n )\n }\n const out: AgentProfile[] = []\n const seen = new Set<string>()\n for (const harness of harnesses) {\n // A universal (router-backed) harness — opencode/pi/claudish — runs every swept\n // model. A vendor-locked harness — codex/claude-code/kimi-code — runs only the\n // swept models in its own family; when it supports NONE of them it snaps to its\n // native default (the `HARNESS_NATIVE_MODEL` sentinel it resolves at runtime)\n // rather than being dropped, so every requested harness still appears in the\n // sweep on a model it can actually run — e.g. sweeping `deepseek/x` puts opencode\n // on deepseek and kimi-code on its own Kimi model, a real head-to-head.\n // `keepIncompatible` forces every (harness, model) pair verbatim (failure-mode runs).\n const supported = spec.keepIncompatible\n ? models\n : models.filter((model) => harnessSupportsModel(harness, model))\n const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL]\n for (const model of effective) {\n const profile: AgentProfile = {\n ...spec.base,\n harness,\n name: `${spec.base.name ?? 'agent'}/${harness}/${model}`,\n model: { ...spec.base.model, default: model },\n metadata: { ...(spec.base.metadata ?? {}), harness, harnessModel: model },\n }\n const id = agentProfileId(profile)\n if (seen.has(id)) continue\n seen.add(id)\n out.push(profile)\n }\n }\n if (out.length === 0) {\n // Unreachable in normal use — snapping guarantees ≥1 cell per harness — but keep a\n // fail-closed guard so a future refactor can't silently produce an empty sweep.\n throw new ValidationError(\n `expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(', ')}], models=[${models.join(', ')}]).`,\n )\n }\n return out\n}\n\n/**\n * Read the (harness, model) a matrix cell ran under, off a profile or a result row's\n * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile\n * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by\n * this instead of recomputing an id (recomputing the wrong key is what broke the pivot\n * in the hand-rolled copies).\n */\nexport function harnessAxisOf(\n profile: Pick<AgentProfile, 'metadata'>,\n): { harness: HarnessType; model: string } | undefined {\n const m = profile.metadata as Record<string, unknown> | undefined\n const harness = m?.harness\n const model = m?.harnessModel\n if (typeof harness === 'string' && typeof model === 'string') {\n return { harness: harness as HarnessType, model }\n }\n return undefined\n}\n\n/**\n * Collision-resistant, path-safe, human-readable profile id for eval artifacts.\n * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix\n * keys, and directory names where two profiles must not collapse onto one row.\n * The suffix is the first 64 bits of the behaviour hash, enough for ordinary\n * eval matrices while keeping filenames readable.\n */\nexport function agentProfileId(profile: AgentProfile): string {\n const label = pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? 'profile'\n return `${label}-${agentProfileHash(profile).slice(0, 16)}`\n}\n\n/**\n * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete\n * model id because run records reject bare/missing model aliases.\n */\nexport function agentProfileModelId(profile: AgentProfile): string {\n const model = profile.model?.default?.trim()\n if (!model) {\n const label = agentProfileDisplayLabel(profile) ?? 'unnamed profile'\n throw new ValidationError(\n `AgentProfile \"${label}\" has no model.default — cannot record eval run`,\n )\n }\n return model\n}\n\nfunction agentProfileDisplayLabel(profile: AgentProfile): string | undefined {\n return profile.name?.trim() || profile.version?.trim() || undefined\n}\n\nfunction pathSafeProfileLabel(label: string | undefined): string | undefined {\n const safe = label\n ?.trim()\n .replace(/[^A-Za-z0-9._-]+/g, '-')\n .replace(/-+/g, '-')\n .replace(/^-|-$/g, '')\n return safe || undefined\n}\n\nfunction compact<T extends Record<string, unknown>>(input: T): Partial<T> {\n const out: Record<string, unknown> = {}\n for (const [key, value] of Object.entries(input)) {\n if (value !== undefined) out[key] = value\n }\n return out as Partial<T>\n}\n\n/**\n * Deterministic behaviour identity for the canonical\n * `@tangle-network/agent-interface` AgentProfile.\n *\n * `name` and `description` are labels and do not affect the hash. Profile\n * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,\n * and extensions do affect the hash. Resource array order is hash-bearing\n * because mount order can change agent behaviour. Undefined fields are treated\n * as absent; explicit `null` fields remain hash-bearing.\n */\nexport function agentProfileHash(profile: AgentProfile): string {\n const model = agentProfileModelId(profile)\n const behaviour = {\n ...profile,\n name: undefined,\n description: undefined,\n tags: profile.tags ? [...profile.tags].sort() : undefined,\n model: compact({ ...profile.model, default: model }),\n }\n return createHash('sha256')\n .update(JSON.stringify(canonicalize(behaviour)))\n .digest('hex')\n}\n","/**\n * Completion verifier — the task-completion oracle.\n *\n * Answers the only eval question that is not a proxy: did the agent actually\n * COMPLETE the task — produce every required deliverable, persisted and\n * correct — rather than describe what should be done. A fluent transcript\n * that never produces the artifact scores zero here.\n *\n * Per requirement, a two-stage check:\n * 1. Structural — a produced item (vault artifact / approved proposal /\n * tool call) of the right kind is matched against the requirement and\n * carries non-empty content. Deterministic; no LLM.\n * 2. Correctness — only if structurally present AND the matched item\n * carries content, one targeted check decides whether that item\n * actually fulfils the requirement. A hallucinated artifact fails here;\n * an absent one already failed stage 1.\n *\n * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —\n * checker failures — are excluded from the denominator, never scored as\n * zeros). Quality dimensions are meaningless on an incomplete task — callers\n * gate on `fullyComplete` / `completionRate` before scoring quality.\n */\n\nimport { randomUUID } from 'node:crypto'\nimport type { ChatClient, ChatRequest } from './analyst/chat-client'\nimport type { Artifact } from './artifact-validator'\nimport { CostLedger, type CostLedgerHandle } from './cost-ledger'\nimport { assertServedModel, ModelSubstitutionError } from './integrity/served-model'\nimport { recoverTruncatedJson } from './json-recovery'\nimport { JudgeParseError } from './judges'\nimport {\n costReceiptFromLlm,\n costReceiptFromLlmError,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport { packageVersion } from './package-version'\nimport type { RawProviderEvent, RawProviderSink } from './trace/raw-provider-sink'\nimport {\n certificationEvidenceDigest,\n type DefaultVerdict,\n type VerdictCertification,\n} from './verdict'\nimport type { CheckerIdentity, VerificationStrategySource } from './verification-strategy'\n\n/** What kind of produced state can satisfy a requirement structurally. */\nexport type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any'\n\nexport interface CompletionRequirement {\n /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */\n reqId: string\n /** Human-readable description of the required deliverable. */\n title: string\n /** Optional kind/category hint, matched against a produced item's kind. */\n category?: string\n /** What produced state satisfies this requirement. Defaults to 'any'. */\n satisfiedBy?: SatisfiedBy\n}\n\nexport interface TaskGold {\n taskId: string\n requirements: CompletionRequirement[]\n}\n\nexport interface ProducedProposal {\n id: string\n title: string\n status: 'pending' | 'approved' | 'rejected'\n /** Optional persisted body — when present, enables a correctness check. */\n content?: string\n}\n\n/** Everything observable about what a run actually produced. */\nexport interface ProducedState {\n /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */\n artifacts: Artifact[]\n /** Proposals / filings the agent created. */\n proposals: ProducedProposal[]\n /** Names of tools the agent invoked. */\n toolCalls: string[]\n}\n\nexport interface RequirementCheck {\n reqId: string\n title: string\n /** A produced item of the right kind matched the requirement, non-empty. */\n structurallyPresent: boolean\n /**\n * Whether the matched item actually fulfils the requirement. `null` when\n * not structurally present, when the matched item carries no content\n * to assess, or when the correctness check itself failed (`unmeasured`).\n */\n correct: boolean | null\n /** structurallyPresent && !unmeasured && correct !== false. */\n satisfied: boolean\n /**\n * Set when the correctness check itself errored (LLM call failure or an\n * unparseable response after retry). The requirement's fulfilment is\n * UNKNOWN — `correct` stays null, `satisfied` is false, and\n * `completionVerdict` excludes the row from `completionRate`'s\n * denominator. Never folded into a zero: a synthetic zero is\n * indistinguishable from a real failure (see `JudgeParseError`).\n */\n unmeasured?: true\n /** Why the correctness check could not be measured (present iff `unmeasured`). */\n unmeasuredReason?: string\n /** Human-readable evidence for the verdict. */\n evidence: string[]\n}\n\n/** Extends the substrate verdict spine: `valid` = `fullyComplete` and\n * `score` = `completionRate` — derived in `completionVerdict()`, the one\n * place those equalities hold by construction. */\nexport interface CompletionVerdict extends DefaultVerdict {\n taskId: string\n requirements: RequirementCheck[]\n /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */\n completionRate: number\n /** Every measurable requirement satisfied (false when anything is unmeasured). */\n fullyComplete: boolean\n /** Requirements whose correctness check errored — reported, never scored as zero. */\n unmeasuredCount: number\n}\n\n/**\n * Construct a `CompletionVerdict` from the per-requirement checks, deriving\n * `completionRate` / `fullyComplete` and the spine fields (`valid` =\n * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero\n * requirements — a verdict over nothing is a misconfiguration, mirroring\n * `verifyCompletion`'s gold-spec guard.\n */\nexport function completionVerdict(input: {\n taskId: string\n requirements: RequirementCheck[]\n /** What certified the correctness stage, when anything did. Omitted =\n * an uncertified verdict — the honest default for a bare checker. */\n certification?: VerdictCertification\n}): CompletionVerdict {\n if (input.requirements.length === 0) {\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`,\n )\n }\n const measurable = input.requirements.filter((r) => !r.unmeasured)\n const unmeasuredCount = input.requirements.length - measurable.length\n if (measurable.length === 0) {\n // Every check errored: this is an infrastructure failure, not a scored\n // run. A 0-rate verdict here would be a fabricated measurement.\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? 'unknown reason'})`,\n )\n }\n const satisfiedCount = measurable.filter((r) => r.satisfied).length\n const completionRate = satisfiedCount / measurable.length\n // A run with unmeasured rows can still report a rate over what WAS\n // measured, but must not claim full completion.\n const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length\n return {\n taskId: input.taskId,\n requirements: input.requirements,\n completionRate,\n fullyComplete,\n unmeasuredCount,\n valid: fullyComplete,\n score: completionRate,\n ...(input.certification === undefined ? {} : { certification: input.certification }),\n }\n}\n\n/**\n * What a correctness checker declares about itself so the completion\n * verdict can carry a certification: which strategy member it discharges,\n * its exact identity, and the steps its answers rest on unverified.\n */\nexport interface CorrectnessCheckerAttestation {\n strategy: VerificationStrategySource\n checker: CheckerIdentity\n assumptions: string[]\n}\n\n/**\n * Decides whether a produced item's content actually fulfils a requirement.\n * Injected so the structural verifier stays pure and unit-testable; the\n * production implementation is `createLlmCorrectnessChecker`.\n *\n * `attestation` is optional metadata on the function value: a checker that\n * carries one yields certified completion verdicts; a bare function yields\n * the same verdict uncertified. A plain arrow function remains a valid\n * checker.\n */\nexport interface CorrectnessChecker {\n (\n requirement: CompletionRequirement,\n content: string,\n ): Promise<{ correct: boolean; reason: string }>\n attestation?: CorrectnessCheckerAttestation\n}\n\nconst STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'of',\n 'for',\n 'and',\n 'or',\n 'to',\n 'in',\n 'on',\n 'with',\n 'by',\n])\n\n// Deliverable-FORM vocabulary — words that name the SHAPE of an output, not its\n// domain content. A correct \"swap-comparison view persisted as a ui/** artifact\"\n// is an OpenUI JSON whose body says nothing about \"artifact\" / \"persisted\" /\n// \"view\"; the discriminative tokens are the domain nouns (swap, comparison).\n// Stripped from the REQUIREMENT side of structural recall so a deliverable is\n// matched on what it IS about, not on the boilerplate describing its form. The\n// correctness checker strips the same class via TITLE_STOPWORDS. Anti-game holds:\n// the distinctive domain tokens remain, so an off-topic item still fails.\nconst REQUIREMENT_FORM_STOPWORDS = new Set([\n 'generated',\n 'generate',\n 'view',\n 'render',\n 'rendered',\n 'persisted',\n 'persist',\n 'artifact',\n 'file',\n 'document',\n 'note',\n 'proposal',\n 'deliverable',\n 'output',\n 'created',\n 'create',\n 'produce',\n 'produced',\n 'flag',\n])\n\nconst MATCH_THRESHOLD = 0.5\nconst MIN_CONTENT_CHARS = 50\n\nfunction tokens(s: string, extraStop?: Set<string>): Set<string> {\n return new Set(\n s\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)),\n )\n}\n\n/**\n * Recall of the requirement's tokens within a candidate's identifying text.\n * Recall, not Jaccard — a candidate's path/id legitimately carries extra\n * tokens the requirement does not name. The requirement side drops\n * deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.\n */\nfunction tokenRecall(requirementText: string, candidateText: string): number {\n const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS)\n if (req.size === 0) return 0\n const cand = tokens(candidateText)\n let hit = 0\n for (const t of req) if (cand.has(t)) hit++\n return hit / req.size\n}\n\ninterface Candidate {\n reqIndex: number\n /** Unique key for a produced item — each item satisfies at most one requirement. */\n itemKey: string\n score: number\n evidence: string\n /** Content to correctness-check, or null when the matched item has none. */\n content: string | null\n}\n\nfunction artifactCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n artifacts: Artifact[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n artifacts.forEach((a, i) => {\n if ((a.content ?? '').trim().length < MIN_CONTENT_CHARS) return\n // Match against the artifact CONTENT too, not just its path + kind — a\n // generated view / note whose path is generic still satisfies a requirement\n // when its body covers it (e.g. an OpenUI comparison grounded in the on-file\n // figures). Bounded slice keeps the recall text cheap; MATCH_THRESHOLD holds.\n let score = tokenRecall(\n reqText,\n `${a.path ?? ''} ${a.kind} ${(a.content ?? '').slice(0, 4000)}`,\n )\n if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) {\n score = Math.max(score, 1)\n }\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `artifact:${i}`,\n score,\n evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,\n content: a.content ?? null,\n })\n })\n return out\n}\n\nfunction proposalCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n proposals: ProducedProposal[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n for (const p of proposals) {\n // Pending or rejected work is not a completed deliverable.\n if (p.status !== 'approved') continue\n // A proposal needs an assessable BODY to be a deliverable. A bare title is\n // not completion: correctness cannot be judged on it, so a title-only match\n // would auto-pass the oracle (structurallyPresent && correct===null →\n // satisfied) with no verifiable content. Tool calls are the only\n // legitimately content-less deliverable (`toolCallCandidates`).\n const body = (p.content ?? '').trim()\n if (body.length < MIN_CONTENT_CHARS) continue\n // Match against the body as well as the (often short) title — a refusal /\n // flag / analysis proposal whose title is a label still satisfies a\n // descriptively-worded requirement when its content covers it. MATCH_THRESHOLD\n // + the requirement's distinctive tokens keep an off-topic proposal out;\n // correctness (a SEMANTIC checker, NOT this lexical pass) then judges\n // polarity/fulfilment, so a negation that merely contains the tokens fails.\n // Structural and correctness must use different evidence or the two-stage\n // check collapses to one lexical gate.\n const score = tokenRecall(reqText, `${p.title} ${body}`)\n if (score < MATCH_THRESHOLD) continue\n out.push({\n reqIndex,\n itemKey: `proposal:${p.id}`,\n score,\n evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,\n content: body,\n })\n }\n return out\n}\n\nfunction toolCallCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n toolCalls: string[],\n): Candidate[] {\n const out: Candidate[] = []\n toolCalls.forEach((name, i) => {\n const score = tokenRecall(req.title, name)\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `tool:${i}`,\n score,\n evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,\n content: null,\n })\n })\n return out\n}\n\n/**\n * Verify whether a run completed the task. `checkCorrectness` is injected —\n * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.\n *\n * Throws on a gold spec with no requirements: an eval task that requires\n * nothing is a misconfiguration, not a vacuously-complete task.\n */\nexport async function verifyCompletion(\n gold: TaskGold,\n state: ProducedState,\n checkCorrectness: CorrectnessChecker,\n): Promise<CompletionVerdict> {\n if (gold.requirements.length === 0) {\n throw new Error(\n `verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`,\n )\n }\n\n // Collect every above-threshold (requirement, produced-item) candidate, then\n // assign greedily by descending score: each requirement and each produced\n // item is used at most once. One deliverable fulfils one requirement.\n const candidates: Candidate[] = []\n gold.requirements.forEach((req, i) => {\n const by = req.satisfiedBy ?? 'any'\n if (by === 'artifact' || by === 'any') {\n candidates.push(...artifactCandidates(req, i, state.artifacts))\n }\n if (by === 'proposal' || by === 'any') {\n candidates.push(...proposalCandidates(req, i, state.proposals))\n }\n if (by === 'tool-call' || by === 'any') {\n candidates.push(...toolCallCandidates(req, i, state.toolCalls))\n }\n })\n candidates.sort((a, b) => b.score - a.score)\n\n const assigned = new Map<number, Candidate>()\n const itemTaken = new Set<string>()\n for (const c of candidates) {\n if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue\n assigned.set(c.reqIndex, c)\n itemTaken.add(c.itemKey)\n }\n\n const requirements: RequirementCheck[] = []\n for (let i = 0; i < gold.requirements.length; i++) {\n const req = gold.requirements[i]!\n const match = assigned.get(i)\n const evidence: string[] = []\n let correct: boolean | null = null\n let unmeasuredReason: string | undefined\n\n if (match) {\n evidence.push(match.evidence)\n if (match.content !== null) {\n try {\n const r = await checkCorrectness(req, match.content)\n correct = r.correct\n evidence.push(`correctness: ${r.correct ? 'pass' : 'fail'} — ${r.reason}`)\n } catch (err) {\n // The CHECKER failed, not the requirement. Recording this as a\n // zero would fabricate a model failure out of an infrastructure\n // one; the requirement is unmeasured and leaves the denominator.\n unmeasuredReason =\n err instanceof JudgeParseError\n ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}`\n : `checker call failed: ${err instanceof Error ? err.message : String(err)}`\n evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`)\n }\n } else {\n evidence.push('correctness: not assessed — matched item carries no content')\n }\n } else {\n const by = req.satisfiedBy ?? 'any'\n const kind = by === 'any' ? 'artifact/proposal/tool-call' : by\n evidence.push(`no produced ${kind} matched this requirement`)\n }\n\n const structurallyPresent = match !== undefined\n const unmeasured = unmeasuredReason !== undefined\n const satisfied = structurallyPresent && !unmeasured && correct !== false\n requirements.push({\n reqId: req.reqId,\n title: req.title,\n structurallyPresent,\n correct,\n satisfied,\n ...(unmeasured ? { unmeasured: true as const, unmeasuredReason } : {}),\n evidence,\n })\n }\n\n // The verdict blends the deterministic structural stage with the injected\n // correctness stage, so the certification names the checker's own member\n // and carries the structural stage's lexical nature as an assumption. A\n // checker without an attestation yields an uncertified verdict.\n const attestation = checkCorrectness.attestation\n const certification: VerdictCertification | undefined = attestation\n ? {\n strategy: attestation.strategy,\n checker: attestation.checker,\n assumptions: [\n 'structural matching is lexical token recall over produced items — the correctness stage only sees items it matched',\n ...attestation.assumptions,\n ],\n evidenceDigest: certificationEvidenceDigest({ taskId: gold.taskId, requirements }),\n }\n : undefined\n\n return completionVerdict({\n taskId: gold.taskId,\n requirements,\n ...(certification === undefined ? {} : { certification }),\n })\n}\n\nexport interface LlmCorrectnessCheckerOpts {\n model?: string\n /** Optional ledger for direct use. */\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /** Max chars of artifact content sent to the checker. */\n maxContentChars?: number\n /**\n * Checker LLM calls per requirement before giving up (parse failures and\n * call errors both consume attempts). The failure then surfaces as an\n * `unmeasured` requirement, never a zero.\n */\n maxAttempts?: number\n /**\n * Forensic capture of every checker request/response/error — without it a\n * checker failure is unauditable (the agent-turn raws never contain the\n * checker's own calls). Same sink contract as `LlmClient`.\n */\n rawSink?: RawProviderSink\n}\n\n/**\n * Parse the correctness checker's model response. Tolerates a response\n * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the\n * verdict boolean usually lands in the first few tokens, so a recovered\n * prefix with a boolean `correct` is a real measurement, not a guess.\n * Fails loud (JudgeParseError) when no boolean verdict is recoverable.\n */\nexport function parseCorrectnessResponse(raw: string): { correct: boolean; reason: string } {\n const readVerdict = (candidate: unknown): { correct: boolean; reason: string } | null => {\n if (candidate === null || typeof candidate !== 'object') return null\n const { correct, reason } = candidate as { correct?: unknown; reason?: unknown }\n if (typeof correct !== 'boolean') return null\n return { correct, reason: typeof reason === 'string' ? reason : '' }\n }\n\n const match = raw.match(/\\{[\\s\\S]*\\}/)\n if (match) {\n try {\n const strict = readVerdict(JSON.parse(match[0]))\n if (strict) return strict\n } catch {\n // fall through to truncation recovery\n }\n }\n // The strict path needs a closing `}`; a cap-hit response has none. Take\n // everything from the first `{` and auto-close it.\n const start = raw.indexOf('{')\n if (start !== -1) {\n const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)))\n if (recovered) return recovered\n }\n throw new JudgeParseError('correctness-checker', raw)\n}\n\n/**\n * Production `CorrectnessChecker` — one LLM call per matched artifact,\n * deterministic (temperature 0), structured JSON out. Judges fulfilment\n * only: a plan, a gesture, or a description of what should be done does not\n * fulfil a requirement — the artifact must BE the deliverable.\n */\nexport function createLlmCorrectnessChecker(\n chat: ChatClient,\n opts: LlmCorrectnessCheckerOpts = {},\n): CorrectnessChecker {\n const model = opts.model ?? 'claude-sonnet-4-6'\n const maxContentChars = opts.maxContentChars ?? 8000\n const maxAttempts = opts.maxAttempts ?? 2\n const costLedger = opts.costLedger ?? new CostLedger()\n const sink = opts.rawSink\n const record = async (event: RawProviderEvent): Promise<void> => {\n // Forensic capture is best-effort; the verdict is the system of record.\n try {\n await sink?.record(event)\n } catch {\n // Intentionally swallowed.\n }\n }\n const checker: CorrectnessChecker = async (requirement, content) => {\n const request = {\n model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}.',\n },\n {\n role: 'user' as const,\n content: `Requirement: ${requirement.title}\\n${\n requirement.category ? `Category: ${requirement.category}\\n` : ''\n }\\nProduced artifact:\\n${content.slice(0, maxContentChars)}`,\n },\n ],\n temperature: 0,\n maxTokens: 200,\n } satisfies ChatRequest\n let lastErr: unknown\n for (let attempt = 0; attempt < maxAttempts; attempt++) {\n const started = Date.now()\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'request',\n timestamp: started,\n requestBody: request,\n redactedFields: [],\n })\n try {\n const paid = await costLedger.runPaidCall({\n channel: 'verifier',\n phase: opts.costPhase ?? 'completion.correctness',\n actor: 'correctness-checker',\n model,\n maximumCharge:\n chat.maximumAttempts === undefined\n ? undefined\n : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),\n tags: {\n ...opts.costTags,\n requirementId: requirement.reqId,\n attempt: String(attempt),\n },\n signal: opts.signal,\n execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),\n receipt: costReceiptFromLlm,\n receiptFromError: costReceiptFromLlmError,\n })\n if (!paid.succeeded) throw paid.error\n const resp = paid.value\n // Hold the transport to its own word: a verdict produced by a model\n // other than the one requested is not this model's verdict. A\n // transport that echoes no id cannot be made to prove identity here —\n // callers needing that proof enable it at the client\n // (`LlmClientOptions.assertServedModel`) or gate on `assertModelsServed`.\n assertServedModel(model, resp.servedModel, {\n allowUnreported: true,\n context: `correctness checker for requirement ${requirement.reqId}`,\n })\n const raw = resp.content\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'response',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n responseBody: resp,\n redactedFields: [],\n })\n return parseCorrectnessResponse(raw)\n } catch (err) {\n lastErr = err\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n // Substitution is a verdict about identity, not a transient fault:\n // another attempt reaches the same wrong model and spends more tokens.\n if (err instanceof ModelSubstitutionError) throw err\n }\n }\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n }\n checker.attestation = {\n strategy: 'judge',\n checker: { name: 'llm-correctness-checker', version: model },\n assumptions: [\n 'served-model identity is accepted when the transport reports none (assertServedModel allowUnreported)',\n ],\n }\n return checker\n}\n\n/** Stopwords for requirement-title tokenization — drops the imperative verbs\n * ('review', 'update', …) common to deliverable titles so recall keys on the\n * substantive nouns, not the boilerplate ask. */\nconst TITLE_STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'and',\n 'or',\n 'for',\n 'to',\n 'of',\n 'in',\n 'on',\n 'with',\n 'review',\n 'update',\n 'new',\n 'proposed',\n])\n\n/**\n * Deterministic `CorrectnessChecker` — the no-LLM counterpart to\n * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its\n * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`\n * of the requirement title's significant tokens. No network.\n *\n * Polarity-blind: token recall credits a negation that contains the\n * requirement's tokens (\"I will NOT produce the comparison\" recalls every token\n * of \"produce the comparison\"). The structural match stage is ALSO lexical, so\n * pairing the two collapses to a single gameable gate. Use this only as an\n * opt-in structural pre-filter or for tasks whose requirements have no polarity\n * to invert; for produced-state grading the correctness checker MUST be semantic\n * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.\n */\nexport function createTokenRecallChecker(\n opts: { minRecall?: number; minContentLength?: number } = {},\n): CorrectnessChecker {\n const minRecall = opts.minRecall ?? 0.5\n const minLen = opts.minContentLength ?? 120\n const checker: CorrectnessChecker = async (requirement, content) => {\n const body = content.trim()\n if (body.length < minLen)\n return {\n correct: false,\n reason: `content too thin (${body.length} chars) to be the deliverable`,\n }\n const titleTokens = requirement.title\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t))\n if (titleTokens.length === 0)\n return {\n correct: true,\n reason: 'requirement title has no significant tokens — structural match accepted',\n }\n const lower = body.toLowerCase()\n const hits = titleTokens.filter((t) => lower.includes(t)).length\n const recall = hits / titleTokens.length\n return recall >= minRecall\n ? {\n correct: true,\n reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`,\n }\n : {\n correct: false,\n reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`,\n }\n }\n // 'schema': token recall checks shape, not meaning — the member whose\n // documented failure mode (\"a well-formed wrong answer passes\") is\n // exactly this checker's polarity blindness.\n checker.attestation = {\n strategy: 'schema',\n checker: { name: 'token-recall-checker', version: packageVersion() },\n assumptions: ['polarity-blind: a negation that recalls the requirement tokens passes'],\n }\n return checker\n}\n","/**\n * Produced-state extraction — normalize a run's runtime event stream into the\n * typed `ProducedState` the completion oracle consumes.\n *\n * `ProducedState` answers \"what did the agent actually produce\" — vault\n * artifacts, proposals, tool calls. The runtime emits these as a stream of\n * events; this module is the single normalization point from that stream to\n * the shape `verifyCompletion` expects.\n *\n * Input is structurally typed (`RuntimeEventLike`) so this module does not\n * depend on agent-runtime — agent-runtime's `RuntimeStreamEvent` satisfies it\n * structurally. The `content` on `ArtifactEventLike` and the whole\n * `proposal_created` variant are the runtime-side enrichments this contract\n * requires; the runtime emits them, this module consumes them.\n */\n\nimport type { Artifact } from './artifact-validator'\nimport type { ProducedProposal, ProducedState } from './completion-verifier'\n\n/** A tool the agent invoked. */\nexport interface ToolCallEventLike {\n type: 'tool_call'\n toolName: string\n}\n\n/**\n * An artifact the agent produced. `content` is the enriched field — the\n * runtime's base `artifact` event carries only metadata; the completion\n * oracle needs the body to verify the deliverable, so the runtime emits it.\n */\nexport interface ArtifactEventLike {\n type: 'artifact'\n artifactId: string\n name?: string\n mimeType?: string\n uri?: string\n content?: string\n}\n\n/** A proposal / filing the agent created. */\nexport interface ProposalEventLike {\n type: 'proposal_created'\n proposalId: string\n title: string\n status?: 'pending' | 'approved' | 'rejected'\n // body of the proposal (e.g. a submit_proposal `description`). When present,\n // the completion oracle correctness-checks it like artifact content; absent,\n // the proposal is graded presence-only.\n content?: string\n}\n\n/**\n * The subset of runtime stream events `extractProducedState` consumes.\n * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;\n * the `{ type: string }` catch-all keeps the input permissive so callers can\n * pass the whole unfiltered telemetry stream — unrecognized events are skipped.\n */\nexport type RuntimeEventLike =\n | ToolCallEventLike\n | ArtifactEventLike\n | ProposalEventLike\n | { type: string }\n\nfunction artifactKind(mimeType: string | undefined): string {\n if (!mimeType) return 'file'\n if (mimeType.includes('json')) return 'json'\n if (mimeType.startsWith('text/')) return 'text'\n return 'file'\n}\n\n/**\n * Normalize a run's runtime event stream into `ProducedState`.\n *\n * Pure and total — unrecognized event types are skipped. `toolCalls` is\n * deduplicated by name in first-seen order (completion cares about a tool's\n * presence, not its call count). An artifact with neither a name nor a uri\n * still yields an entry keyed by its `artifactId` so it is never silently\n * dropped; an artifact with no `content` yields empty content, which the\n * completion oracle's structural check then rejects on its own.\n */\nexport function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState {\n const artifacts: Artifact[] = []\n const proposals: ProducedProposal[] = []\n const toolCalls: string[] = []\n const seenTools = new Set<string>()\n\n for (const ev of events) {\n if (ev.type === 'tool_call') {\n const name = (ev as ToolCallEventLike).toolName\n if (name && !seenTools.has(name)) {\n seenTools.add(name)\n toolCalls.push(name)\n }\n } else if (ev.type === 'artifact') {\n const a = ev as ArtifactEventLike\n artifacts.push({\n kind: artifactKind(a.mimeType),\n path: a.name ?? a.uri ?? a.artifactId,\n content: a.content ?? '',\n })\n } else if (ev.type === 'proposal_created') {\n const p = ev as ProposalEventLike\n proposals.push({\n id: p.proposalId,\n title: p.title,\n status: p.status ?? 'pending',\n ...(p.content !== undefined ? { content: p.content } : {}),\n })\n }\n }\n\n return { artifacts, proposals, toolCalls }\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAa,mBAA2C;CACtD;CACA;CACA;CACA;AACF;;;;;;AAwBA,MAAa,uBAAuB;;;;;;;;;;;;;;;;;;;AAoBpC,SAAgB,kBAAkB,MAAuC;CACvE,MAAM,YAAY,KAAK,aAAa;CACpC,IAAI,UAAU,WAAW,GAAG,MAAM,IAAI,gBAAgB,0CAA0C;CAChG,MAAM,YAAY,KAAK,KAAK,OAAO;CACnC,MAAM,SAAS,KAAK,WAAW,YAAY,CAAC,SAAS,IAAI,CAAC;CAC1D,IAAI,OAAO,WAAW,GACpB,MAAM,IAAI,gBACR,kGACF;CAEF,MAAM,MAAsB,CAAC;CAC7B,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,WAAW,WAAW;EAS/B,MAAM,YAAY,KAAK,mBACnB,SACA,OAAO,QAAQ,UAAU,qBAAqB,SAAS,KAAK,CAAC;EACjE,MAAM,YAAY,UAAU,SAAS,IAAI,YAAY,CAAC,oBAAoB;EAC1E,KAAK,MAAM,SAAS,WAAW;GAC7B,MAAM,UAAwB;IAC5B,GAAG,KAAK;IACR;IACA,MAAM,GAAG,KAAK,KAAK,QAAQ,QAAQ,GAAG,QAAQ,GAAG;IACjD,OAAO;KAAE,GAAG,KAAK,KAAK;KAAO,SAAS;IAAM;IAC5C,UAAU;KAAE,GAAI,KAAK,KAAK,YAAY,CAAC;KAAI;KAAS,cAAc;IAAM;GAC1E;GACA,MAAM,KAAK,eAAe,OAAO;GACjC,IAAI,KAAK,IAAI,EAAE,GAAG;GAClB,KAAK,IAAI,EAAE;GACX,IAAI,KAAK,OAAO;EAClB;CACF;CACA,IAAI,IAAI,WAAW,GAGjB,MAAM,IAAI,gBACR,uDAAuD,UAAU,KAAK,IAAI,EAAE,aAAa,OAAO,KAAK,IAAI,EAAE,IAC7G;CAEF,OAAO;AACT;;;;;;;;AASA,SAAgB,cACd,SACqD;CACrD,MAAM,IAAI,QAAQ;CAClB,MAAM,UAAU,GAAG;CACnB,MAAM,QAAQ,GAAG;CACjB,IAAI,OAAO,YAAY,YAAY,OAAO,UAAU,UAClD,OAAO;EAAW;EAAwB;CAAM;AAGpD;;;;;;;;AASA,SAAgB,eAAe,SAA+B;CAE5D,OAAO,GADO,qBAAqB,yBAAyB,OAAO,CAAC,KAAK,UACzD,GAAG,iBAAiB,OAAO,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1D;;;;;AAMA,SAAgB,oBAAoB,SAA+B;CACjE,MAAM,QAAQ,QAAQ,OAAO,SAAS,KAAK;CAC3C,IAAI,CAAC,OAEH,MAAM,IAAI,gBACR,iBAFY,yBAAyB,OAAO,KAAK,kBAE1B,gDACzB;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,SAA2C;CAC3E,OAAO,QAAQ,MAAM,KAAK,KAAK,QAAQ,SAAS,KAAK,KAAK,KAAA;AAC5D;AAEA,SAAS,qBAAqB,OAA+C;CAM3E,OALa,OACT,KAAK,CAAC,CACP,QAAQ,qBAAqB,GAAG,CAAC,CACjC,QAAQ,OAAO,GAAG,CAAC,CACnB,QAAQ,UAAU,EAAE,KACR,KAAA;AACjB;AAEA,SAAS,QAA2C,OAAsB;CACxE,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,KAAK,GAC7C,IAAI,UAAU,KAAA,GAAW,IAAI,OAAO;CAEtC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,iBAAiB,SAA+B;CAC9D,MAAM,QAAQ,oBAAoB,OAAO;CACzC,MAAM,YAAY;EAChB,GAAG;EACH,MAAM,KAAA;EACN,aAAa,KAAA;EACb,MAAM,QAAQ,OAAO,CAAC,GAAG,QAAQ,IAAI,CAAC,CAAC,KAAK,IAAI,KAAA;EAChD,OAAO,QAAQ;GAAE,GAAG,QAAQ;GAAO,SAAS;EAAM,CAAC;CACrD;CACA,OAAO,WAAW,QAAQ,CAAC,CACxB,OAAO,KAAK,UAAU,aAAa,SAAS,CAAC,CAAC,CAAC,CAC/C,OAAO,KAAK;AACjB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACxEA,SAAgB,kBAAkB,OAMZ;CACpB,IAAI,MAAM,aAAa,WAAW,GAChC,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,+DAC3C;CAEF,MAAM,aAAa,MAAM,aAAa,QAAQ,MAAM,CAAC,EAAE,UAAU;CACjE,MAAM,kBAAkB,MAAM,aAAa,SAAS,WAAW;CAC/D,IAAI,WAAW,WAAW,GAGxB,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,yCAAyC,MAAM,aAAa,OAAO,8BAA8B,MAAM,aAAa,EAAE,EAAE,oBAAoB,iBAAiB,EACxM;CAEF,MAAM,iBAAiB,WAAW,QAAQ,MAAM,EAAE,SAAS,CAAC,CAAC;CAC7D,MAAM,iBAAiB,iBAAiB,WAAW;CAGnD,MAAM,gBAAgB,oBAAoB,KAAK,mBAAmB,WAAW;CAC7E,OAAO;EACL,QAAQ,MAAM;EACd,cAAc,MAAM;EACpB;EACA;EACA;EACA,OAAO;EACP,OAAO;EACP,GAAI,MAAM,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,eAAe,MAAM,cAAc;CACpF;AACF;AA+BA,MAAM,4BAAY,IAAI,IAAI;CACxB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAUD,MAAM,6CAA6B,IAAI,IAAI;CACzC;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAM,kBAAkB;AACxB,MAAM,oBAAoB;AAE1B,SAAS,OAAO,GAAW,WAAsC;CAC/D,OAAO,IAAI,IACT,EACG,YAAY,CAAC,CACb,MAAM,YAAY,CAAC,CACnB,QAAQ,MAAM,EAAE,SAAS,KAAK,CAAC,UAAU,IAAI,CAAC,KAAK,CAAC,WAAW,IAAI,CAAC,CAAC,CAC1E;AACF;;;;;;;AAQA,SAAS,YAAY,iBAAyB,eAA+B;CAC3E,MAAM,MAAM,OAAO,iBAAiB,0BAA0B;CAC9D,IAAI,IAAI,SAAS,GAAG,OAAO;CAC3B,MAAM,OAAO,OAAO,aAAa;CACjC,IAAI,MAAM;CACV,KAAK,MAAM,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC,GAAG;CACtC,OAAO,MAAM,IAAI;AACnB;AAYA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,GAAG,MAAM;EAC1B,KAAK,EAAE,WAAW,GAAA,CAAI,KAAK,CAAC,CAAC,SAAS,mBAAmB;EAKzD,IAAI,QAAQ,YACV,SACA,GAAG,EAAE,QAAQ,GAAG,GAAG,EAAE,KAAK,IAAI,EAAE,WAAW,GAAA,CAAI,MAAM,GAAG,GAAI,GAC9D;EACA,IAAI,IAAI,YAAY,EAAE,QAAQ,IAAI,SAAS,YAAY,MAAM,EAAE,KAAK,YAAY,GAC9E,QAAQ,KAAK,IAAI,OAAO,CAAC;EAE3B,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY;GACrB;GACA,UAAU,aAAa,EAAE,QAAQ,EAAE,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,CAAC;CACD,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,KAAK,MAAM,KAAK,WAAW;EAEzB,IAAI,EAAE,WAAW,YAAY;EAM7B,MAAM,QAAQ,EAAE,WAAW,GAAA,CAAI,KAAK;EACpC,IAAI,KAAK,SAAS,mBAAmB;EASrC,MAAM,QAAQ,YAAY,SAAS,GAAG,EAAE,MAAM,GAAG,MAAM;EACvD,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY,EAAE;GACvB;GACA,UAAU,sBAAsB,EAAE,MAAM,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS;EACX,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,MAAM,MAAM;EAC7B,MAAM,QAAQ,YAAY,IAAI,OAAO,IAAI;EACzC,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,QAAQ;GACjB;GACA,UAAU,cAAc,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACxE,SAAS;EACX,CAAC;CACH,CAAC;CACD,OAAO;AACT;;;;;;;;AASA,eAAsB,iBACpB,MACA,OACA,kBAC4B;CAC5B,IAAI,KAAK,aAAa,WAAW,GAC/B,MAAM,IAAI,MACR,2BAA2B,KAAK,OAAO,4CACzC;CAMF,MAAM,aAA0B,CAAC;CACjC,KAAK,aAAa,SAAS,KAAK,MAAM;EACpC,MAAM,KAAK,IAAI,eAAe;EAC9B,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,eAAe,OAAO,OAC/B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;CAElE,CAAC;CACD,WAAW,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;CAE3C,MAAM,2BAAW,IAAI,IAAuB;CAC5C,MAAM,4BAAY,IAAI,IAAY;CAClC,KAAK,MAAM,KAAK,YAAY;EAC1B,IAAI,SAAS,IAAI,EAAE,QAAQ,KAAK,UAAU,IAAI,EAAE,OAAO,GAAG;EAC1D,SAAS,IAAI,EAAE,UAAU,CAAC;EAC1B,UAAU,IAAI,EAAE,OAAO;CACzB;CAEA,MAAM,eAAmC,CAAC;CAC1C,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,aAAa,QAAQ,KAAK;EACjD,MAAM,MAAM,KAAK,aAAa;EAC9B,MAAM,QAAQ,SAAS,IAAI,CAAC;EAC5B,MAAM,WAAqB,CAAC;EAC5B,IAAI,UAA0B;EAC9B,IAAI;EAEJ,IAAI,OAAO;GACT,SAAS,KAAK,MAAM,QAAQ;GAC5B,IAAI,MAAM,YAAY,MACpB,IAAI;IACF,MAAM,IAAI,MAAM,iBAAiB,KAAK,MAAM,OAAO;IACnD,UAAU,EAAE;IACZ,SAAS,KAAK,gBAAgB,EAAE,UAAU,SAAS,OAAO,KAAK,EAAE,QAAQ;GAC3E,SAAS,KAAK;IAIZ,mBACE,eAAe,kBACX,6CAA6C,IAAI,IAAI,MAAM,GAAG,GAAG,MACjE,wBAAwB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAC7E,SAAS,KAAK,6BAA6B,kBAAkB;GAC/D;QAEA,SAAS,KAAK,6DAA6D;EAE/E,OAAO;GACL,MAAM,KAAK,IAAI,eAAe;GAC9B,MAAM,OAAO,OAAO,QAAQ,gCAAgC;GAC5D,SAAS,KAAK,eAAe,KAAK,0BAA0B;EAC9D;EAEA,MAAM,sBAAsB,UAAU,KAAA;EACtC,MAAM,aAAa,qBAAqB,KAAA;EACxC,MAAM,YAAY,uBAAuB,CAAC,cAAc,YAAY;EACpE,aAAa,KAAK;GAChB,OAAO,IAAI;GACX,OAAO,IAAI;GACX;GACA;GACA;GACA,GAAI,aAAa;IAAE,YAAY;IAAe;GAAiB,IAAI,CAAC;GACpE;EACF,CAAC;CACH;CAMA,MAAM,cAAc,iBAAiB;CACrC,MAAM,gBAAkD,cACpD;EACE,UAAU,YAAY;EACtB,SAAS,YAAY;EACrB,aAAa,CACX,sHACA,GAAG,YAAY,WACjB;EACA,gBAAgB,4BAA4B;GAAE,QAAQ,KAAK;GAAQ;EAAa,CAAC;CACnF,IACA,KAAA;CAEJ,OAAO,kBAAkB;EACvB,QAAQ,KAAK;EACb;EACA,GAAI,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,cAAc;CACzD,CAAC;AACH;;;;;;;;AAgCA,SAAgB,yBAAyB,KAAmD;CAC1F,MAAM,eAAe,cAAoE;EACvF,IAAI,cAAc,QAAQ,OAAO,cAAc,UAAU,OAAO;EAChE,MAAM,EAAE,SAAS,WAAW;EAC5B,IAAI,OAAO,YAAY,WAAW,OAAO;EACzC,OAAO;GAAE;GAAS,QAAQ,OAAO,WAAW,WAAW,SAAS;EAAG;CACrE;CAEA,MAAM,QAAQ,IAAI,MAAM,aAAa;CACrC,IAAI,OACF,IAAI;EACF,MAAM,SAAS,YAAY,KAAK,MAAM,MAAM,EAAE,CAAC;EAC/C,IAAI,QAAQ,OAAO;CACrB,QAAQ,CAER;CAIF,MAAM,QAAQ,IAAI,QAAQ,GAAG;CAC7B,IAAI,UAAU,IAAI;EAChB,MAAM,YAAY,YAAY,qBAAqB,IAAI,MAAM,KAAK,CAAC,CAAC;EACpE,IAAI,WAAW,OAAO;CACxB;CACA,MAAM,IAAI,gBAAgB,uBAAuB,GAAG;AACtD;;;;;;;AAQA,SAAgB,4BACd,MACA,OAAkC,CAAC,GACf;CACpB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,cAAc,KAAK,eAAe;CACxC,MAAM,aAAa,KAAK,cAAc,IAAI,WAAW;CACrD,MAAM,OAAO,KAAK;CAClB,MAAM,SAAS,OAAO,UAA2C;EAE/D,IAAI;GACF,MAAM,MAAM,OAAO,KAAK;EAC1B,QAAQ,CAER;CACF;CACA,MAAM,UAA8B,OAAO,aAAa,YAAY;EAClE,MAAM,UAAU;GACd;GACA,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IACE,MAAM;IACN,SAAS,gBAAgB,YAAY,MAAM,IACzC,YAAY,WAAW,aAAa,YAAY,SAAS,MAAM,GAChE,wBAAwB,QAAQ,MAAM,GAAG,eAAe;GAC3D,CACF;GACA,aAAa;GACb,WAAW;EACb;EACA,IAAI;EACJ,KAAK,IAAI,UAAU,GAAG,UAAU,aAAa,WAAW;GACtD,MAAM,UAAU,KAAK,IAAI;GACzB,MAAM,OAAO;IACX,SAAS,WAAW;IACpB,UAAU,KAAK;IACf;IACA,UAAU;IACV,SAAS;IACT,cAAc;IACd,WAAW;IACX,WAAW;IACX,aAAa;IACb,gBAAgB,CAAC;GACnB,CAAC;GACD,IAAI;IACF,MAAM,OAAO,MAAM,WAAW,YAAY;KACxC,SAAS;KACT,OAAO,KAAK,aAAa;KACzB,OAAO;KACP;KACA,eACE,KAAK,oBAAoB,KAAA,IACrB,KAAA,IACA,2BAA2B,SAAS,EAAE,iBAAiB,KAAK,gBAAgB,CAAC;KACnF,MAAM;MACJ,GAAG,KAAK;MACR,eAAe,YAAY;MAC3B,SAAS,OAAO,OAAO;KACzB;KACA,QAAQ,KAAK;KACb,UAAU,QAAQ,WAAW,KAAK,KAAK,SAAS;MAAE;MAAQ,gBAAgB;KAAO,CAAC;KAClF,SAAS;KACT,kBAAkB;IACpB,CAAC;IACD,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;IAChC,MAAM,OAAO,KAAK;IAMlB,kBAAkB,OAAO,KAAK,aAAa;KACzC,iBAAiB;KACjB,SAAS,uCAAuC,YAAY;IAC9D,CAAC;IACD,MAAM,MAAM,KAAK;IACjB,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc;KACd,gBAAgB,CAAC;IACnB,CAAC;IACD,OAAO,yBAAyB,GAAG;GACrC,SAAS,KAAK;IACZ,UAAU;IACV,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;KAC7D,gBAAgB,CAAC;IACnB,CAAC;IAGD,IAAI,eAAe,wBAAwB,MAAM;GACnD;EACF;EACA,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;CACtE;CACA,QAAQ,cAAc;EACpB,UAAU;EACV,SAAS;GAAE,MAAM;GAA2B,SAAS;EAAM;EAC3D,aAAa,CACX,uGACF;CACF;CACA,OAAO;AACT;;;ACpmBA,SAAS,aAAa,UAAsC;CAC1D,IAAI,CAAC,UAAU,OAAO;CACtB,IAAI,SAAS,SAAS,MAAM,GAAG,OAAO;CACtC,IAAI,SAAS,WAAW,OAAO,GAAG,OAAO;CACzC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,qBAAqB,QAAoD;CACvF,MAAM,YAAwB,CAAC;CAC/B,MAAM,YAAgC,CAAC;CACvC,MAAM,YAAsB,CAAC;CAC7B,MAAM,4BAAY,IAAI,IAAY;CAElC,KAAK,MAAM,MAAM,QACf,IAAI,GAAG,SAAS,aAAa;EAC3B,MAAM,OAAQ,GAAyB;EACvC,IAAI,QAAQ,CAAC,UAAU,IAAI,IAAI,GAAG;GAChC,UAAU,IAAI,IAAI;GAClB,UAAU,KAAK,IAAI;EACrB;CACF,OAAO,IAAI,GAAG,SAAS,YAAY;EACjC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,MAAM,aAAa,EAAE,QAAQ;GAC7B,MAAM,EAAE,QAAQ,EAAE,OAAO,EAAE;GAC3B,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,OAAO,IAAI,GAAG,SAAS,oBAAoB;EACzC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,IAAI,EAAE;GACN,OAAO,EAAE;GACT,QAAQ,EAAE,UAAU;GACpB,GAAI,EAAE,YAAY,KAAA,IAAY,EAAE,SAAS,EAAE,QAAQ,IAAI,CAAC;EAC1D,CAAC;CACH;CAGF,OAAO;EAAE;EAAW;EAAW;CAAU;AAC3C"}
1
+ {"version":3,"file":"produced-state-C6vSgOpE.js","names":[],"sources":["../src/agent-profile.ts","../src/completion-verifier.ts","../src/produced-state.ts"],"sourcesContent":["import { createHash } from 'node:crypto'\nimport type { AgentProfile } from '@tangle-network/agent-interface'\nimport { type HarnessType, harnessSupportsModel } from '@tangle-network/agent-interface'\nimport { ValidationError } from './errors'\nimport { canonicalize } from './pre-registration'\n\nexport type { AgentProfile, HarnessType } from '@tangle-network/agent-interface'\n\n/**\n * The agentic coding harnesses an eval sweeps by default — the ones we care about\n * ranking. This is the SINGLE source of that list; consumers import it instead of\n * re-declaring their own (a re-declared list is how the fleet drifts). Pass an\n * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known\n * harness) to widen beyond these.\n */\nexport const CODING_HARNESSES: readonly HarnessType[] = [\n 'opencode',\n 'claude-code',\n 'codex',\n 'kimi-code',\n]\n\nexport interface ProfileAxisSpec {\n /** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the\n * harness and model vary. `model.default` is the fallback model. */\n base: AgentProfile\n /** Harnesses to cross. Default: {@link CODING_HARNESSES}. */\n harnesses?: readonly HarnessType[]\n /** Models to cross. Default: `[base.model.default]` — one model, i.e. today's\n * single-model behaviour, so omitting this never changes an existing run. */\n models?: readonly string[]\n /** Force every (harness, model) pair verbatim, even ones the harness can't run —\n * for deliberately testing failure modes. Default (false): SNAP instead — a\n * vendor-locked harness runs only the swept models in its family, or its native\n * default when it supports none, so no harness is dropped and none gets a\n * guaranteed-failing foreign-model cell. */\n keepIncompatible?: boolean\n}\n\n/** Model sentinel for a vendor-locked harness that supports none of the swept models:\n * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness\n * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi\n * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship\n * table that would rot as router catalogs change. */\nexport const HARNESS_NATIVE_MODEL = 'default'\n\n/**\n * Expand a base profile across the harness × model matrix into the `AgentProfile[]`\n * that `runProfileMatrix` / `selfImprove` score — the ONE place \"which harnesses ×\n * which models do we evaluate\" lives, so no product hand-rolls its own harness list\n * or column→profile mapping (the pattern that let those copies drift and silently\n * break the harness pivot).\n *\n * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,\n * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both\n * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row\n * and results join back by harness/model via {@link harnessAxisOf} with no\n * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or\n * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every\n * requested harness runs; `keepIncompatible` forces every pair verbatim.\n *\n * Omit `harnesses`/`models` to sweep the full default set — the \"turn it on for\n * everything we care about\" switch, identical in shape whether one harness or all.\n */\nexport function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[] {\n const harnesses = spec.harnesses ?? CODING_HARNESSES\n if (harnesses.length === 0) throw new ValidationError('expandProfileAxes: no harnesses to sweep')\n const baseModel = spec.base.model?.default\n const models = spec.models ?? (baseModel ? [baseModel] : [])\n if (models.length === 0) {\n throw new ValidationError(\n 'expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied',\n )\n }\n const out: AgentProfile[] = []\n const seen = new Set<string>()\n for (const harness of harnesses) {\n // A universal (router-backed) harness — opencode/pi/claudish — runs every swept\n // model. A vendor-locked harness — codex/claude-code/kimi-code — runs only the\n // swept models in its own family; when it supports NONE of them it snaps to its\n // native default (the `HARNESS_NATIVE_MODEL` sentinel it resolves at runtime)\n // rather than being dropped, so every requested harness still appears in the\n // sweep on a model it can actually run — e.g. sweeping `deepseek/x` puts opencode\n // on deepseek and kimi-code on its own Kimi model, a real head-to-head.\n // `keepIncompatible` forces every (harness, model) pair verbatim (failure-mode runs).\n const supported = spec.keepIncompatible\n ? models\n : models.filter((model) => harnessSupportsModel(harness, model))\n const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL]\n for (const model of effective) {\n const profile: AgentProfile = {\n ...spec.base,\n harness,\n name: `${spec.base.name ?? 'agent'}/${harness}/${model}`,\n model: { ...spec.base.model, default: model },\n metadata: { ...(spec.base.metadata ?? {}), harness, harnessModel: model },\n }\n const id = agentProfileId(profile)\n if (seen.has(id)) continue\n seen.add(id)\n out.push(profile)\n }\n }\n if (out.length === 0) {\n // Unreachable in normal use — snapping guarantees ≥1 cell per harness — but keep a\n // fail-closed guard so a future refactor can't silently produce an empty sweep.\n throw new ValidationError(\n `expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(', ')}], models=[${models.join(', ')}]).`,\n )\n }\n return out\n}\n\n/**\n * Read the (harness, model) a matrix cell ran under, off a profile or a result row's\n * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile\n * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by\n * this instead of recomputing an id (recomputing the wrong key is what broke the pivot\n * in the hand-rolled copies).\n */\nexport function harnessAxisOf(\n profile: Pick<AgentProfile, 'metadata'>,\n): { harness: HarnessType; model: string } | undefined {\n const m = profile.metadata as Record<string, unknown> | undefined\n const harness = m?.harness\n const model = m?.harnessModel\n if (typeof harness === 'string' && typeof model === 'string') {\n return { harness: harness as HarnessType, model }\n }\n return undefined\n}\n\n/**\n * Collision-resistant, path-safe, human-readable profile id for eval artifacts.\n * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix\n * keys, and directory names where two profiles must not collapse onto one row.\n * The suffix is the first 64 bits of the behaviour hash, enough for ordinary\n * eval matrices while keeping filenames readable.\n */\nexport function agentProfileId(profile: AgentProfile): string {\n const label = pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? 'profile'\n return `${label}-${agentProfileHash(profile).slice(0, 16)}`\n}\n\n/**\n * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete\n * model id because run records reject bare/missing model aliases.\n */\nexport function agentProfileModelId(profile: AgentProfile): string {\n const model = profile.model?.default?.trim()\n if (!model) {\n const label = agentProfileDisplayLabel(profile) ?? 'unnamed profile'\n throw new ValidationError(\n `AgentProfile \"${label}\" has no model.default — cannot record eval run`,\n )\n }\n return model\n}\n\nfunction agentProfileDisplayLabel(profile: AgentProfile): string | undefined {\n return profile.name?.trim() || profile.version?.trim() || undefined\n}\n\nfunction pathSafeProfileLabel(label: string | undefined): string | undefined {\n const safe = label\n ?.trim()\n .replace(/[^A-Za-z0-9._-]+/g, '-')\n .replace(/-+/g, '-')\n .replace(/^-|-$/g, '')\n return safe || undefined\n}\n\nfunction compact<T extends Record<string, unknown>>(input: T): Partial<T> {\n const out: Record<string, unknown> = {}\n for (const [key, value] of Object.entries(input)) {\n if (value !== undefined) out[key] = value\n }\n return out as Partial<T>\n}\n\n/**\n * Deterministic behaviour identity for the canonical\n * `@tangle-network/agent-interface` AgentProfile.\n *\n * `name` and `description` are labels and do not affect the hash. Profile\n * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,\n * and extensions do affect the hash. Resource array order is hash-bearing\n * because mount order can change agent behaviour. Undefined fields are treated\n * as absent; explicit `null` fields remain hash-bearing.\n */\nexport function agentProfileHash(profile: AgentProfile): string {\n const model = agentProfileModelId(profile)\n const behaviour = {\n ...profile,\n name: undefined,\n description: undefined,\n tags: profile.tags ? [...profile.tags].sort() : undefined,\n model: compact({ ...profile.model, default: model }),\n }\n return createHash('sha256')\n .update(JSON.stringify(canonicalize(behaviour)))\n .digest('hex')\n}\n","/**\n * Completion verifier — the task-completion oracle.\n *\n * Answers the only eval question that is not a proxy: did the agent actually\n * COMPLETE the task — produce every required deliverable, persisted and\n * correct — rather than describe what should be done. A fluent transcript\n * that never produces the artifact scores zero here.\n *\n * Per requirement, a two-stage check:\n * 1. Structural — a produced item (vault artifact / approved proposal /\n * tool call) of the right kind is matched against the requirement and\n * carries non-empty content. Deterministic; no LLM.\n * 2. Correctness — only if structurally present AND the matched item\n * carries content, one targeted check decides whether that item\n * actually fulfils the requirement. A hallucinated artifact fails here;\n * an absent one already failed stage 1.\n *\n * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —\n * checker failures — are excluded from the denominator, never scored as\n * zeros). Quality dimensions are meaningless on an incomplete task — callers\n * gate on `fullyComplete` / `completionRate` before scoring quality.\n */\n\nimport { randomUUID } from 'node:crypto'\nimport type { ChatClient, ChatRequest } from './analyst/chat-client'\nimport type { Artifact } from './artifact-validator'\nimport { CostLedger, type CostLedgerHandle } from './cost-ledger'\nimport { assertServedModel, ModelSubstitutionError } from './integrity/served-model'\nimport { recoverTruncatedJson } from './json-recovery'\nimport { JudgeParseError } from './judges'\nimport {\n costReceiptFromLlm,\n costReceiptFromLlmError,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport { packageVersion } from './package-version'\nimport type { RawProviderEvent, RawProviderSink } from './trace/raw-provider-sink'\nimport {\n certificationEvidenceDigest,\n type DefaultVerdict,\n type VerdictCertification,\n} from './verdict'\nimport type { CheckerIdentity, VerificationStrategySource } from './verification-strategy'\n\n/** What kind of produced state can satisfy a requirement structurally. */\nexport type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any'\n\nexport interface CompletionRequirement {\n /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */\n reqId: string\n /** Human-readable description of the required deliverable. */\n title: string\n /** Optional kind/category hint, matched against a produced item's kind. */\n category?: string\n /** What produced state satisfies this requirement. Defaults to 'any'. */\n satisfiedBy?: SatisfiedBy\n}\n\nexport interface TaskGold {\n taskId: string\n requirements: CompletionRequirement[]\n}\n\nexport interface ProducedProposal {\n id: string\n title: string\n status: 'pending' | 'approved' | 'rejected'\n /** Optional persisted body — when present, enables a correctness check. */\n content?: string\n}\n\n/** Everything observable about what a run actually produced. */\nexport interface ProducedState {\n /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */\n artifacts: Artifact[]\n /** Proposals / filings the agent created. */\n proposals: ProducedProposal[]\n /** Names of tools the agent invoked. */\n toolCalls: string[]\n}\n\nexport interface RequirementCheck {\n reqId: string\n title: string\n /** A produced item of the right kind matched the requirement, non-empty. */\n structurallyPresent: boolean\n /**\n * Whether the matched item actually fulfils the requirement. `null` when\n * not structurally present, when the matched item carries no content\n * to assess, or when the correctness check itself failed (`unmeasured`).\n */\n correct: boolean | null\n /** structurallyPresent && !unmeasured && correct !== false. */\n satisfied: boolean\n /**\n * Set when the correctness check itself errored (LLM call failure or an\n * unparseable response after retry). The requirement's fulfilment is\n * UNKNOWN — `correct` stays null, `satisfied` is false, and\n * `completionVerdict` excludes the row from `completionRate`'s\n * denominator. Never folded into a zero: a synthetic zero is\n * indistinguishable from a real failure (see `JudgeParseError`).\n */\n unmeasured?: true\n /** Why the correctness check could not be measured (present iff `unmeasured`). */\n unmeasuredReason?: string\n /** Human-readable evidence for the verdict. */\n evidence: string[]\n}\n\n/** Extends the substrate verdict spine: `valid` = `fullyComplete` and\n * `score` = `completionRate` — derived in `completionVerdict()`, the one\n * place those equalities hold by construction. */\nexport interface CompletionVerdict extends DefaultVerdict {\n taskId: string\n requirements: RequirementCheck[]\n /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */\n completionRate: number\n /** Every measurable requirement satisfied (false when anything is unmeasured). */\n fullyComplete: boolean\n /** Requirements whose correctness check errored — reported, never scored as zero. */\n unmeasuredCount: number\n}\n\n/**\n * Construct a `CompletionVerdict` from the per-requirement checks, deriving\n * `completionRate` / `fullyComplete` and the spine fields (`valid` =\n * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero\n * requirements — a verdict over nothing is a misconfiguration, mirroring\n * `verifyCompletion`'s gold-spec guard.\n */\nexport function completionVerdict(input: {\n taskId: string\n requirements: RequirementCheck[]\n /** What certified the correctness stage, when anything did. Omitted =\n * an uncertified verdict — the honest default for a bare checker. */\n certification?: VerdictCertification\n}): CompletionVerdict {\n if (input.requirements.length === 0) {\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`,\n )\n }\n const measurable = input.requirements.filter((r) => !r.unmeasured)\n const unmeasuredCount = input.requirements.length - measurable.length\n if (measurable.length === 0) {\n // Every check errored: this is an infrastructure failure, not a scored\n // run. A 0-rate verdict here would be a fabricated measurement.\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? 'unknown reason'})`,\n )\n }\n const satisfiedCount = measurable.filter((r) => r.satisfied).length\n const completionRate = satisfiedCount / measurable.length\n // A run with unmeasured rows can still report a rate over what WAS\n // measured, but must not claim full completion.\n const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length\n return {\n taskId: input.taskId,\n requirements: input.requirements,\n completionRate,\n fullyComplete,\n unmeasuredCount,\n valid: fullyComplete,\n score: completionRate,\n ...(input.certification === undefined ? {} : { certification: input.certification }),\n }\n}\n\n/**\n * What a correctness checker declares about itself so the completion\n * verdict can carry a certification: which strategy member it discharges,\n * its exact identity, and the steps its answers rest on unverified.\n */\nexport interface CorrectnessCheckerAttestation {\n strategy: VerificationStrategySource\n checker: CheckerIdentity\n assumptions: string[]\n}\n\n/**\n * Decides whether a produced item's content actually fulfils a requirement.\n * Injected so the structural verifier stays pure and unit-testable; the\n * production implementation is `createLlmCorrectnessChecker`.\n *\n * `attestation` is optional metadata on the function value: a checker that\n * carries one yields certified completion verdicts; a bare function yields\n * the same verdict uncertified. A plain arrow function remains a valid\n * checker.\n */\nexport interface CorrectnessChecker {\n (\n requirement: CompletionRequirement,\n content: string,\n ): Promise<{ correct: boolean; reason: string }>\n attestation?: CorrectnessCheckerAttestation\n}\n\nconst STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'of',\n 'for',\n 'and',\n 'or',\n 'to',\n 'in',\n 'on',\n 'with',\n 'by',\n])\n\n// Deliverable-FORM vocabulary — words that name the SHAPE of an output, not its\n// domain content. A correct \"swap-comparison view persisted as a ui/** artifact\"\n// is an OpenUI JSON whose body says nothing about \"artifact\" / \"persisted\" /\n// \"view\"; the discriminative tokens are the domain nouns (swap, comparison).\n// Stripped from the REQUIREMENT side of structural recall so a deliverable is\n// matched on what it IS about, not on the boilerplate describing its form. The\n// correctness checker strips the same class via TITLE_STOPWORDS. Anti-game holds:\n// the distinctive domain tokens remain, so an off-topic item still fails.\nconst REQUIREMENT_FORM_STOPWORDS = new Set([\n 'generated',\n 'generate',\n 'view',\n 'render',\n 'rendered',\n 'persisted',\n 'persist',\n 'artifact',\n 'file',\n 'document',\n 'note',\n 'proposal',\n 'deliverable',\n 'output',\n 'created',\n 'create',\n 'produce',\n 'produced',\n 'flag',\n])\n\nconst MATCH_THRESHOLD = 0.5\nconst MIN_CONTENT_CHARS = 50\n\nfunction tokens(s: string, extraStop?: Set<string>): Set<string> {\n return new Set(\n s\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)),\n )\n}\n\n/**\n * Recall of the requirement's tokens within a candidate's identifying text.\n * Recall, not Jaccard — a candidate's path/id legitimately carries extra\n * tokens the requirement does not name. The requirement side drops\n * deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.\n */\nfunction tokenRecall(requirementText: string, candidateText: string): number {\n const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS)\n if (req.size === 0) return 0\n const cand = tokens(candidateText)\n let hit = 0\n for (const t of req) if (cand.has(t)) hit++\n return hit / req.size\n}\n\ninterface Candidate {\n reqIndex: number\n /** Unique key for a produced item — each item satisfies at most one requirement. */\n itemKey: string\n score: number\n evidence: string\n /** Content to correctness-check, or null when the matched item has none. */\n content: string | null\n}\n\nfunction artifactCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n artifacts: Artifact[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n artifacts.forEach((a, i) => {\n if ((a.content ?? '').trim().length < MIN_CONTENT_CHARS) return\n // Match against the artifact CONTENT too, not just its path + kind — a\n // generated view / note whose path is generic still satisfies a requirement\n // when its body covers it (e.g. an OpenUI comparison grounded in the on-file\n // figures). Bounded slice keeps the recall text cheap; MATCH_THRESHOLD holds.\n let score = tokenRecall(\n reqText,\n `${a.path ?? ''} ${a.kind} ${(a.content ?? '').slice(0, 4000)}`,\n )\n if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) {\n score = Math.max(score, 1)\n }\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `artifact:${i}`,\n score,\n evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,\n content: a.content ?? null,\n })\n })\n return out\n}\n\nfunction proposalCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n proposals: ProducedProposal[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n for (const p of proposals) {\n // Pending or rejected work is not a completed deliverable.\n if (p.status !== 'approved') continue\n // A proposal needs an assessable BODY to be a deliverable. A bare title is\n // not completion: correctness cannot be judged on it, so a title-only match\n // would auto-pass the oracle (structurallyPresent && correct===null →\n // satisfied) with no verifiable content. Tool calls are the only\n // legitimately content-less deliverable (`toolCallCandidates`).\n const body = (p.content ?? '').trim()\n if (body.length < MIN_CONTENT_CHARS) continue\n // Match against the body as well as the (often short) title — a refusal /\n // flag / analysis proposal whose title is a label still satisfies a\n // descriptively-worded requirement when its content covers it. MATCH_THRESHOLD\n // + the requirement's distinctive tokens keep an off-topic proposal out;\n // correctness (a SEMANTIC checker, NOT this lexical pass) then judges\n // polarity/fulfilment, so a negation that merely contains the tokens fails.\n // Structural and correctness must use different evidence or the two-stage\n // check collapses to one lexical gate.\n const score = tokenRecall(reqText, `${p.title} ${body}`)\n if (score < MATCH_THRESHOLD) continue\n out.push({\n reqIndex,\n itemKey: `proposal:${p.id}`,\n score,\n evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,\n content: body,\n })\n }\n return out\n}\n\nfunction toolCallCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n toolCalls: string[],\n): Candidate[] {\n const out: Candidate[] = []\n toolCalls.forEach((name, i) => {\n const score = tokenRecall(req.title, name)\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `tool:${i}`,\n score,\n evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,\n content: null,\n })\n })\n return out\n}\n\n/**\n * Verify whether a run completed the task. `checkCorrectness` is injected —\n * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.\n *\n * Throws on a gold spec with no requirements: an eval task that requires\n * nothing is a misconfiguration, not a vacuously-complete task.\n */\nexport async function verifyCompletion(\n gold: TaskGold,\n state: ProducedState,\n checkCorrectness: CorrectnessChecker,\n): Promise<CompletionVerdict> {\n if (gold.requirements.length === 0) {\n throw new Error(\n `verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`,\n )\n }\n\n // Collect every above-threshold (requirement, produced-item) candidate, then\n // assign greedily by descending score: each requirement and each produced\n // item is used at most once. One deliverable fulfils one requirement.\n const candidates: Candidate[] = []\n gold.requirements.forEach((req, i) => {\n const by = req.satisfiedBy ?? 'any'\n if (by === 'artifact' || by === 'any') {\n candidates.push(...artifactCandidates(req, i, state.artifacts))\n }\n if (by === 'proposal' || by === 'any') {\n candidates.push(...proposalCandidates(req, i, state.proposals))\n }\n if (by === 'tool-call' || by === 'any') {\n candidates.push(...toolCallCandidates(req, i, state.toolCalls))\n }\n })\n candidates.sort((a, b) => b.score - a.score)\n\n const assigned = new Map<number, Candidate>()\n const itemTaken = new Set<string>()\n for (const c of candidates) {\n if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue\n assigned.set(c.reqIndex, c)\n itemTaken.add(c.itemKey)\n }\n\n const requirements: RequirementCheck[] = []\n for (let i = 0; i < gold.requirements.length; i++) {\n const req = gold.requirements[i]!\n const match = assigned.get(i)\n const evidence: string[] = []\n let correct: boolean | null = null\n let unmeasuredReason: string | undefined\n\n if (match) {\n evidence.push(match.evidence)\n if (match.content !== null) {\n try {\n const r = await checkCorrectness(req, match.content)\n correct = r.correct\n evidence.push(`correctness: ${r.correct ? 'pass' : 'fail'} — ${r.reason}`)\n } catch (err) {\n // The CHECKER failed, not the requirement. Recording this as a\n // zero would fabricate a model failure out of an infrastructure\n // one; the requirement is unmeasured and leaves the denominator.\n unmeasuredReason =\n err instanceof JudgeParseError\n ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}`\n : `checker call failed: ${err instanceof Error ? err.message : String(err)}`\n evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`)\n }\n } else {\n evidence.push('correctness: not assessed — matched item carries no content')\n }\n } else {\n const by = req.satisfiedBy ?? 'any'\n const kind = by === 'any' ? 'artifact/proposal/tool-call' : by\n evidence.push(`no produced ${kind} matched this requirement`)\n }\n\n const structurallyPresent = match !== undefined\n const unmeasured = unmeasuredReason !== undefined\n const satisfied = structurallyPresent && !unmeasured && correct !== false\n requirements.push({\n reqId: req.reqId,\n title: req.title,\n structurallyPresent,\n correct,\n satisfied,\n ...(unmeasured ? { unmeasured: true as const, unmeasuredReason } : {}),\n evidence,\n })\n }\n\n // The verdict blends the deterministic structural stage with the injected\n // correctness stage, so the certification names the checker's own member\n // and carries the structural stage's lexical nature as an assumption. A\n // checker without an attestation yields an uncertified verdict.\n const attestation = checkCorrectness.attestation\n const certification: VerdictCertification | undefined = attestation\n ? {\n strategy: attestation.strategy,\n checker: attestation.checker,\n assumptions: [\n 'structural matching is lexical token recall over produced items — the correctness stage only sees items it matched',\n ...attestation.assumptions,\n ],\n evidenceDigest: certificationEvidenceDigest({ taskId: gold.taskId, requirements }),\n }\n : undefined\n\n return completionVerdict({\n taskId: gold.taskId,\n requirements,\n ...(certification === undefined ? {} : { certification }),\n })\n}\n\nexport interface LlmCorrectnessCheckerOpts {\n model?: string\n /** Optional ledger for direct use. */\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /** Max chars of artifact content sent to the checker. */\n maxContentChars?: number\n /**\n * Checker LLM calls per requirement before giving up (parse failures and\n * call errors both consume attempts). The failure then surfaces as an\n * `unmeasured` requirement, never a zero.\n */\n maxAttempts?: number\n /**\n * Forensic capture of every checker request/response/error — without it a\n * checker failure is unauditable (the agent-turn raws never contain the\n * checker's own calls). Same sink contract as `LlmClient`.\n */\n rawSink?: RawProviderSink\n}\n\n/**\n * Parse the correctness checker's model response. Tolerates a response\n * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the\n * verdict boolean usually lands in the first few tokens, so a recovered\n * prefix with a boolean `correct` is a real measurement, not a guess.\n * Fails loud (JudgeParseError) when no boolean verdict is recoverable.\n */\nexport function parseCorrectnessResponse(raw: string): { correct: boolean; reason: string } {\n const readVerdict = (candidate: unknown): { correct: boolean; reason: string } | null => {\n if (candidate === null || typeof candidate !== 'object') return null\n const { correct, reason } = candidate as { correct?: unknown; reason?: unknown }\n if (typeof correct !== 'boolean') return null\n return { correct, reason: typeof reason === 'string' ? reason : '' }\n }\n\n const match = raw.match(/\\{[\\s\\S]*\\}/)\n if (match) {\n try {\n const strict = readVerdict(JSON.parse(match[0]))\n if (strict) return strict\n } catch {\n // fall through to truncation recovery\n }\n }\n // The strict path needs a closing `}`; a cap-hit response has none. Take\n // everything from the first `{` and auto-close it.\n const start = raw.indexOf('{')\n if (start !== -1) {\n const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)))\n if (recovered) return recovered\n }\n throw new JudgeParseError('correctness-checker', raw)\n}\n\n/**\n * Production `CorrectnessChecker` — one LLM call per matched artifact,\n * deterministic (temperature 0), structured JSON out. Judges fulfilment\n * only: a plan, a gesture, or a description of what should be done does not\n * fulfil a requirement — the artifact must BE the deliverable.\n */\nexport function createLlmCorrectnessChecker(\n chat: ChatClient,\n opts: LlmCorrectnessCheckerOpts = {},\n): CorrectnessChecker {\n const model = opts.model ?? 'claude-sonnet-4-6'\n const maxContentChars = opts.maxContentChars ?? 8000\n const maxAttempts = opts.maxAttempts ?? 2\n const costLedger = opts.costLedger ?? new CostLedger()\n const sink = opts.rawSink\n const record = async (event: RawProviderEvent): Promise<void> => {\n // Forensic capture is best-effort; the verdict is the system of record.\n try {\n await sink?.record(event)\n } catch {\n // Intentionally swallowed.\n }\n }\n const checker: CorrectnessChecker = async (requirement, content) => {\n const request = {\n model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}.',\n },\n {\n role: 'user' as const,\n content: `Requirement: ${requirement.title}\\n${\n requirement.category ? `Category: ${requirement.category}\\n` : ''\n }\\nProduced artifact:\\n${content.slice(0, maxContentChars)}`,\n },\n ],\n temperature: 0,\n maxTokens: 200,\n } satisfies ChatRequest\n let lastErr: unknown\n for (let attempt = 0; attempt < maxAttempts; attempt++) {\n const started = Date.now()\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'request',\n timestamp: started,\n requestBody: request,\n redactedFields: [],\n })\n try {\n const paid = await costLedger.runPaidCall({\n channel: 'verifier',\n phase: opts.costPhase ?? 'completion.correctness',\n actor: 'correctness-checker',\n model,\n maximumCharge:\n chat.maximumAttempts === undefined\n ? undefined\n : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),\n tags: {\n ...opts.costTags,\n requirementId: requirement.reqId,\n attempt: String(attempt),\n },\n signal: opts.signal,\n execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),\n receipt: costReceiptFromLlm,\n receiptFromError: costReceiptFromLlmError,\n })\n if (!paid.succeeded) throw paid.error\n const resp = paid.value\n // Hold the transport to its own word: a verdict produced by a model\n // other than the one requested is not this model's verdict. A\n // transport that echoes no id cannot be made to prove identity here —\n // callers needing that proof enable it at the client\n // (`LlmClientOptions.assertServedModel`) or gate on `assertModelsServed`.\n assertServedModel(model, resp.servedModel, {\n allowUnreported: true,\n context: `correctness checker for requirement ${requirement.reqId}`,\n })\n const raw = resp.content\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'response',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n responseBody: resp,\n redactedFields: [],\n })\n return parseCorrectnessResponse(raw)\n } catch (err) {\n lastErr = err\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n // Substitution is a verdict about identity, not a transient fault:\n // another attempt reaches the same wrong model and spends more tokens.\n if (err instanceof ModelSubstitutionError) throw err\n }\n }\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n }\n checker.attestation = {\n strategy: 'judge',\n checker: { name: 'llm-correctness-checker', version: model },\n assumptions: [\n 'served-model identity is accepted when the transport reports none (assertServedModel allowUnreported)',\n ],\n }\n return checker\n}\n\n/** Stopwords for requirement-title tokenization — drops the imperative verbs\n * ('review', 'update', …) common to deliverable titles so recall keys on the\n * substantive nouns, not the boilerplate ask. */\nconst TITLE_STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'and',\n 'or',\n 'for',\n 'to',\n 'of',\n 'in',\n 'on',\n 'with',\n 'review',\n 'update',\n 'new',\n 'proposed',\n])\n\n/**\n * Deterministic `CorrectnessChecker` — the no-LLM counterpart to\n * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its\n * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`\n * of the requirement title's significant tokens. No network.\n *\n * Polarity-blind: token recall credits a negation that contains the\n * requirement's tokens (\"I will NOT produce the comparison\" recalls every token\n * of \"produce the comparison\"). The structural match stage is ALSO lexical, so\n * pairing the two collapses to a single gameable gate. Use this only as an\n * opt-in structural pre-filter or for tasks whose requirements have no polarity\n * to invert; for produced-state grading the correctness checker MUST be semantic\n * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.\n */\nexport function createTokenRecallChecker(\n opts: { minRecall?: number; minContentLength?: number } = {},\n): CorrectnessChecker {\n const minRecall = opts.minRecall ?? 0.5\n const minLen = opts.minContentLength ?? 120\n const checker: CorrectnessChecker = async (requirement, content) => {\n const body = content.trim()\n if (body.length < minLen)\n return {\n correct: false,\n reason: `content too thin (${body.length} chars) to be the deliverable`,\n }\n const titleTokens = requirement.title\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t))\n if (titleTokens.length === 0)\n return {\n correct: true,\n reason: 'requirement title has no significant tokens — structural match accepted',\n }\n const lower = body.toLowerCase()\n const hits = titleTokens.filter((t) => lower.includes(t)).length\n const recall = hits / titleTokens.length\n return recall >= minRecall\n ? {\n correct: true,\n reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`,\n }\n : {\n correct: false,\n reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`,\n }\n }\n // 'schema': token recall checks shape, not meaning — the member whose\n // documented failure mode (\"a well-formed wrong answer passes\") is\n // exactly this checker's polarity blindness.\n checker.attestation = {\n strategy: 'schema',\n checker: { name: 'token-recall-checker', version: packageVersion() },\n assumptions: ['polarity-blind: a negation that recalls the requirement tokens passes'],\n }\n return checker\n}\n","/**\n * Produced-state extraction — normalize a run's runtime event stream into the\n * typed `ProducedState` the completion oracle consumes.\n *\n * `ProducedState` answers \"what did the agent actually produce\" — vault\n * artifacts, proposals, tool calls. The runtime emits these as a stream of\n * events; this module is the single normalization point from that stream to\n * the shape `verifyCompletion` expects.\n *\n * Input is structurally typed (`RuntimeEventLike`) so this module does not\n * depend on agent-runtime — agent-runtime's `RuntimeStreamEvent` satisfies it\n * structurally. The `content` on `ArtifactEventLike` and the whole\n * `proposal_created` variant are the runtime-side enrichments this contract\n * requires; the runtime emits them, this module consumes them.\n */\n\nimport type { Artifact } from './artifact-validator'\nimport type { ProducedProposal, ProducedState } from './completion-verifier'\n\n/** A tool the agent invoked. */\nexport interface ToolCallEventLike {\n type: 'tool_call'\n toolName: string\n}\n\n/**\n * An artifact the agent produced. `content` is the enriched field — the\n * runtime's base `artifact` event carries only metadata; the completion\n * oracle needs the body to verify the deliverable, so the runtime emits it.\n */\nexport interface ArtifactEventLike {\n type: 'artifact'\n artifactId: string\n name?: string\n mimeType?: string\n uri?: string\n content?: string\n}\n\n/** A proposal / filing the agent created. */\nexport interface ProposalEventLike {\n type: 'proposal_created'\n proposalId: string\n title: string\n status?: 'pending' | 'approved' | 'rejected'\n // body of the proposal (e.g. a submit_proposal `description`). When present,\n // the completion oracle correctness-checks it like artifact content; absent,\n // the proposal is graded presence-only.\n content?: string\n}\n\n/**\n * The subset of runtime stream events `extractProducedState` consumes.\n * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;\n * the `{ type: string }` catch-all keeps the input permissive so callers can\n * pass the whole unfiltered telemetry stream — unrecognized events are skipped.\n */\nexport type RuntimeEventLike =\n | ToolCallEventLike\n | ArtifactEventLike\n | ProposalEventLike\n | { type: string }\n\nfunction artifactKind(mimeType: string | undefined): string {\n if (!mimeType) return 'file'\n if (mimeType.includes('json')) return 'json'\n if (mimeType.startsWith('text/')) return 'text'\n return 'file'\n}\n\n/**\n * Normalize a run's runtime event stream into `ProducedState`.\n *\n * Pure and total — unrecognized event types are skipped. `toolCalls` is\n * deduplicated by name in first-seen order (completion cares about a tool's\n * presence, not its call count). An artifact with neither a name nor a uri\n * still yields an entry keyed by its `artifactId` so it is never silently\n * dropped; an artifact with no `content` yields empty content, which the\n * completion oracle's structural check then rejects on its own.\n */\nexport function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState {\n const artifacts: Artifact[] = []\n const proposals: ProducedProposal[] = []\n const toolCalls: string[] = []\n const seenTools = new Set<string>()\n\n for (const ev of events) {\n if (ev.type === 'tool_call') {\n const name = (ev as ToolCallEventLike).toolName\n if (name && !seenTools.has(name)) {\n seenTools.add(name)\n toolCalls.push(name)\n }\n } else if (ev.type === 'artifact') {\n const a = ev as ArtifactEventLike\n artifacts.push({\n kind: artifactKind(a.mimeType),\n path: a.name ?? a.uri ?? a.artifactId,\n content: a.content ?? '',\n })\n } else if (ev.type === 'proposal_created') {\n const p = ev as ProposalEventLike\n proposals.push({\n id: p.proposalId,\n title: p.title,\n status: p.status ?? 'pending',\n ...(p.content !== undefined ? { content: p.content } : {}),\n })\n }\n }\n\n return { artifacts, proposals, toolCalls }\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAa,mBAA2C;CACtD;CACA;CACA;CACA;AACF;;;;;;AAwBA,MAAa,uBAAuB;;;;;;;;;;;;;;;;;;;AAoBpC,SAAgB,kBAAkB,MAAuC;CACvE,MAAM,YAAY,KAAK,aAAa;CACpC,IAAI,UAAU,WAAW,GAAG,MAAM,IAAI,gBAAgB,0CAA0C;CAChG,MAAM,YAAY,KAAK,KAAK,OAAO;CACnC,MAAM,SAAS,KAAK,WAAW,YAAY,CAAC,SAAS,IAAI,CAAC;CAC1D,IAAI,OAAO,WAAW,GACpB,MAAM,IAAI,gBACR,kGACF;CAEF,MAAM,MAAsB,CAAC;CAC7B,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,WAAW,WAAW;EAS/B,MAAM,YAAY,KAAK,mBACnB,SACA,OAAO,QAAQ,UAAU,qBAAqB,SAAS,KAAK,CAAC;EACjE,MAAM,YAAY,UAAU,SAAS,IAAI,YAAY,CAAC,oBAAoB;EAC1E,KAAK,MAAM,SAAS,WAAW;GAC7B,MAAM,UAAwB;IAC5B,GAAG,KAAK;IACR;IACA,MAAM,GAAG,KAAK,KAAK,QAAQ,QAAQ,GAAG,QAAQ,GAAG;IACjD,OAAO;KAAE,GAAG,KAAK,KAAK;KAAO,SAAS;IAAM;IAC5C,UAAU;KAAE,GAAI,KAAK,KAAK,YAAY,CAAC;KAAI;KAAS,cAAc;IAAM;GAC1E;GACA,MAAM,KAAK,eAAe,OAAO;GACjC,IAAI,KAAK,IAAI,EAAE,GAAG;GAClB,KAAK,IAAI,EAAE;GACX,IAAI,KAAK,OAAO;EAClB;CACF;CACA,IAAI,IAAI,WAAW,GAGjB,MAAM,IAAI,gBACR,uDAAuD,UAAU,KAAK,IAAI,EAAE,aAAa,OAAO,KAAK,IAAI,EAAE,IAC7G;CAEF,OAAO;AACT;;;;;;;;AASA,SAAgB,cACd,SACqD;CACrD,MAAM,IAAI,QAAQ;CAClB,MAAM,UAAU,GAAG;CACnB,MAAM,QAAQ,GAAG;CACjB,IAAI,OAAO,YAAY,YAAY,OAAO,UAAU,UAClD,OAAO;EAAW;EAAwB;CAAM;AAGpD;;;;;;;;AASA,SAAgB,eAAe,SAA+B;CAE5D,OAAO,GADO,qBAAqB,yBAAyB,OAAO,CAAC,KAAK,UACzD,GAAG,iBAAiB,OAAO,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1D;;;;;AAMA,SAAgB,oBAAoB,SAA+B;CACjE,MAAM,QAAQ,QAAQ,OAAO,SAAS,KAAK;CAC3C,IAAI,CAAC,OAEH,MAAM,IAAI,gBACR,iBAFY,yBAAyB,OAAO,KAAK,kBAE1B,gDACzB;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,SAA2C;CAC3E,OAAO,QAAQ,MAAM,KAAK,KAAK,QAAQ,SAAS,KAAK,KAAK,KAAA;AAC5D;AAEA,SAAS,qBAAqB,OAA+C;CAM3E,OALa,OACT,KAAK,CAAC,CACP,QAAQ,qBAAqB,GAAG,CAAC,CACjC,QAAQ,OAAO,GAAG,CAAC,CACnB,QAAQ,UAAU,EAAE,KACR,KAAA;AACjB;AAEA,SAAS,QAA2C,OAAsB;CACxE,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,KAAK,GAC7C,IAAI,UAAU,KAAA,GAAW,IAAI,OAAO;CAEtC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,iBAAiB,SAA+B;CAC9D,MAAM,QAAQ,oBAAoB,OAAO;CACzC,MAAM,YAAY;EAChB,GAAG;EACH,MAAM,KAAA;EACN,aAAa,KAAA;EACb,MAAM,QAAQ,OAAO,CAAC,GAAG,QAAQ,IAAI,CAAC,CAAC,KAAK,IAAI,KAAA;EAChD,OAAO,QAAQ;GAAE,GAAG,QAAQ;GAAO,SAAS;EAAM,CAAC;CACrD;CACA,OAAO,WAAW,QAAQ,CAAC,CACxB,OAAO,KAAK,UAAU,aAAa,SAAS,CAAC,CAAC,CAAC,CAC/C,OAAO,KAAK;AACjB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACxEA,SAAgB,kBAAkB,OAMZ;CACpB,IAAI,MAAM,aAAa,WAAW,GAChC,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,+DAC3C;CAEF,MAAM,aAAa,MAAM,aAAa,QAAQ,MAAM,CAAC,EAAE,UAAU;CACjE,MAAM,kBAAkB,MAAM,aAAa,SAAS,WAAW;CAC/D,IAAI,WAAW,WAAW,GAGxB,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,yCAAyC,MAAM,aAAa,OAAO,8BAA8B,MAAM,aAAa,EAAE,EAAE,oBAAoB,iBAAiB,EACxM;CAEF,MAAM,iBAAiB,WAAW,QAAQ,MAAM,EAAE,SAAS,CAAC,CAAC;CAC7D,MAAM,iBAAiB,iBAAiB,WAAW;CAGnD,MAAM,gBAAgB,oBAAoB,KAAK,mBAAmB,WAAW;CAC7E,OAAO;EACL,QAAQ,MAAM;EACd,cAAc,MAAM;EACpB;EACA;EACA;EACA,OAAO;EACP,OAAO;EACP,GAAI,MAAM,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,eAAe,MAAM,cAAc;CACpF;AACF;AA+BA,MAAM,4BAAY,IAAI,IAAI;CACxB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAUD,MAAM,6CAA6B,IAAI,IAAI;CACzC;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAM,kBAAkB;AACxB,MAAM,oBAAoB;AAE1B,SAAS,OAAO,GAAW,WAAsC;CAC/D,OAAO,IAAI,IACT,EACG,YAAY,CAAC,CACb,MAAM,YAAY,CAAC,CACnB,QAAQ,MAAM,EAAE,SAAS,KAAK,CAAC,UAAU,IAAI,CAAC,KAAK,CAAC,WAAW,IAAI,CAAC,CAAC,CAC1E;AACF;;;;;;;AAQA,SAAS,YAAY,iBAAyB,eAA+B;CAC3E,MAAM,MAAM,OAAO,iBAAiB,0BAA0B;CAC9D,IAAI,IAAI,SAAS,GAAG,OAAO;CAC3B,MAAM,OAAO,OAAO,aAAa;CACjC,IAAI,MAAM;CACV,KAAK,MAAM,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC,GAAG;CACtC,OAAO,MAAM,IAAI;AACnB;AAYA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,GAAG,MAAM;EAC1B,KAAK,EAAE,WAAW,GAAA,CAAI,KAAK,CAAC,CAAC,SAAS,mBAAmB;EAKzD,IAAI,QAAQ,YACV,SACA,GAAG,EAAE,QAAQ,GAAG,GAAG,EAAE,KAAK,IAAI,EAAE,WAAW,GAAA,CAAI,MAAM,GAAG,GAAI,GAC9D;EACA,IAAI,IAAI,YAAY,EAAE,QAAQ,IAAI,SAAS,YAAY,MAAM,EAAE,KAAK,YAAY,GAC9E,QAAQ,KAAK,IAAI,OAAO,CAAC;EAE3B,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY;GACrB;GACA,UAAU,aAAa,EAAE,QAAQ,EAAE,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,CAAC;CACD,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,KAAK,MAAM,KAAK,WAAW;EAEzB,IAAI,EAAE,WAAW,YAAY;EAM7B,MAAM,QAAQ,EAAE,WAAW,GAAA,CAAI,KAAK;EACpC,IAAI,KAAK,SAAS,mBAAmB;EASrC,MAAM,QAAQ,YAAY,SAAS,GAAG,EAAE,MAAM,GAAG,MAAM;EACvD,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY,EAAE;GACvB;GACA,UAAU,sBAAsB,EAAE,MAAM,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS;EACX,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,MAAM,MAAM;EAC7B,MAAM,QAAQ,YAAY,IAAI,OAAO,IAAI;EACzC,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,QAAQ;GACjB;GACA,UAAU,cAAc,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACxE,SAAS;EACX,CAAC;CACH,CAAC;CACD,OAAO;AACT;;;;;;;;AASA,eAAsB,iBACpB,MACA,OACA,kBAC4B;CAC5B,IAAI,KAAK,aAAa,WAAW,GAC/B,MAAM,IAAI,MACR,2BAA2B,KAAK,OAAO,4CACzC;CAMF,MAAM,aAA0B,CAAC;CACjC,KAAK,aAAa,SAAS,KAAK,MAAM;EACpC,MAAM,KAAK,IAAI,eAAe;EAC9B,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,eAAe,OAAO,OAC/B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;CAElE,CAAC;CACD,WAAW,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;CAE3C,MAAM,2BAAW,IAAI,IAAuB;CAC5C,MAAM,4BAAY,IAAI,IAAY;CAClC,KAAK,MAAM,KAAK,YAAY;EAC1B,IAAI,SAAS,IAAI,EAAE,QAAQ,KAAK,UAAU,IAAI,EAAE,OAAO,GAAG;EAC1D,SAAS,IAAI,EAAE,UAAU,CAAC;EAC1B,UAAU,IAAI,EAAE,OAAO;CACzB;CAEA,MAAM,eAAmC,CAAC;CAC1C,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,aAAa,QAAQ,KAAK;EACjD,MAAM,MAAM,KAAK,aAAa;EAC9B,MAAM,QAAQ,SAAS,IAAI,CAAC;EAC5B,MAAM,WAAqB,CAAC;EAC5B,IAAI,UAA0B;EAC9B,IAAI;EAEJ,IAAI,OAAO;GACT,SAAS,KAAK,MAAM,QAAQ;GAC5B,IAAI,MAAM,YAAY,MACpB,IAAI;IACF,MAAM,IAAI,MAAM,iBAAiB,KAAK,MAAM,OAAO;IACnD,UAAU,EAAE;IACZ,SAAS,KAAK,gBAAgB,EAAE,UAAU,SAAS,OAAO,KAAK,EAAE,QAAQ;GAC3E,SAAS,KAAK;IAIZ,mBACE,eAAe,kBACX,6CAA6C,IAAI,IAAI,MAAM,GAAG,GAAG,MACjE,wBAAwB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAC7E,SAAS,KAAK,6BAA6B,kBAAkB;GAC/D;QAEA,SAAS,KAAK,6DAA6D;EAE/E,OAAO;GACL,MAAM,KAAK,IAAI,eAAe;GAC9B,MAAM,OAAO,OAAO,QAAQ,gCAAgC;GAC5D,SAAS,KAAK,eAAe,KAAK,0BAA0B;EAC9D;EAEA,MAAM,sBAAsB,UAAU,KAAA;EACtC,MAAM,aAAa,qBAAqB,KAAA;EACxC,MAAM,YAAY,uBAAuB,CAAC,cAAc,YAAY;EACpE,aAAa,KAAK;GAChB,OAAO,IAAI;GACX,OAAO,IAAI;GACX;GACA;GACA;GACA,GAAI,aAAa;IAAE,YAAY;IAAe;GAAiB,IAAI,CAAC;GACpE;EACF,CAAC;CACH;CAMA,MAAM,cAAc,iBAAiB;CACrC,MAAM,gBAAkD,cACpD;EACE,UAAU,YAAY;EACtB,SAAS,YAAY;EACrB,aAAa,CACX,sHACA,GAAG,YAAY,WACjB;EACA,gBAAgB,4BAA4B;GAAE,QAAQ,KAAK;GAAQ;EAAa,CAAC;CACnF,IACA,KAAA;CAEJ,OAAO,kBAAkB;EACvB,QAAQ,KAAK;EACb;EACA,GAAI,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,cAAc;CACzD,CAAC;AACH;;;;;;;;AAgCA,SAAgB,yBAAyB,KAAmD;CAC1F,MAAM,eAAe,cAAoE;EACvF,IAAI,cAAc,QAAQ,OAAO,cAAc,UAAU,OAAO;EAChE,MAAM,EAAE,SAAS,WAAW;EAC5B,IAAI,OAAO,YAAY,WAAW,OAAO;EACzC,OAAO;GAAE;GAAS,QAAQ,OAAO,WAAW,WAAW,SAAS;EAAG;CACrE;CAEA,MAAM,QAAQ,IAAI,MAAM,aAAa;CACrC,IAAI,OACF,IAAI;EACF,MAAM,SAAS,YAAY,KAAK,MAAM,MAAM,EAAE,CAAC;EAC/C,IAAI,QAAQ,OAAO;CACrB,QAAQ,CAER;CAIF,MAAM,QAAQ,IAAI,QAAQ,GAAG;CAC7B,IAAI,UAAU,IAAI;EAChB,MAAM,YAAY,YAAY,qBAAqB,IAAI,MAAM,KAAK,CAAC,CAAC;EACpE,IAAI,WAAW,OAAO;CACxB;CACA,MAAM,IAAI,gBAAgB,uBAAuB,GAAG;AACtD;;;;;;;AAQA,SAAgB,4BACd,MACA,OAAkC,CAAC,GACf;CACpB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,cAAc,KAAK,eAAe;CACxC,MAAM,aAAa,KAAK,cAAc,IAAI,WAAW;CACrD,MAAM,OAAO,KAAK;CAClB,MAAM,SAAS,OAAO,UAA2C;EAE/D,IAAI;GACF,MAAM,MAAM,OAAO,KAAK;EAC1B,QAAQ,CAER;CACF;CACA,MAAM,UAA8B,OAAO,aAAa,YAAY;EAClE,MAAM,UAAU;GACd;GACA,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IACE,MAAM;IACN,SAAS,gBAAgB,YAAY,MAAM,IACzC,YAAY,WAAW,aAAa,YAAY,SAAS,MAAM,GAChE,wBAAwB,QAAQ,MAAM,GAAG,eAAe;GAC3D,CACF;GACA,aAAa;GACb,WAAW;EACb;EACA,IAAI;EACJ,KAAK,IAAI,UAAU,GAAG,UAAU,aAAa,WAAW;GACtD,MAAM,UAAU,KAAK,IAAI;GACzB,MAAM,OAAO;IACX,SAAS,WAAW;IACpB,UAAU,KAAK;IACf;IACA,UAAU;IACV,SAAS;IACT,cAAc;IACd,WAAW;IACX,WAAW;IACX,aAAa;IACb,gBAAgB,CAAC;GACnB,CAAC;GACD,IAAI;IACF,MAAM,OAAO,MAAM,WAAW,YAAY;KACxC,SAAS;KACT,OAAO,KAAK,aAAa;KACzB,OAAO;KACP;KACA,eACE,KAAK,oBAAoB,KAAA,IACrB,KAAA,IACA,2BAA2B,SAAS,EAAE,iBAAiB,KAAK,gBAAgB,CAAC;KACnF,MAAM;MACJ,GAAG,KAAK;MACR,eAAe,YAAY;MAC3B,SAAS,OAAO,OAAO;KACzB;KACA,QAAQ,KAAK;KACb,UAAU,QAAQ,WAAW,KAAK,KAAK,SAAS;MAAE;MAAQ,gBAAgB;KAAO,CAAC;KAClF,SAAS;KACT,kBAAkB;IACpB,CAAC;IACD,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;IAChC,MAAM,OAAO,KAAK;IAMlB,kBAAkB,OAAO,KAAK,aAAa;KACzC,iBAAiB;KACjB,SAAS,uCAAuC,YAAY;IAC9D,CAAC;IACD,MAAM,MAAM,KAAK;IACjB,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc;KACd,gBAAgB,CAAC;IACnB,CAAC;IACD,OAAO,yBAAyB,GAAG;GACrC,SAAS,KAAK;IACZ,UAAU;IACV,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;KAC7D,gBAAgB,CAAC;IACnB,CAAC;IAGD,IAAI,eAAe,wBAAwB,MAAM;GACnD;EACF;EACA,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;CACtE;CACA,QAAQ,cAAc;EACpB,UAAU;EACV,SAAS;GAAE,MAAM;GAA2B,SAAS;EAAM;EAC3D,aAAa,CACX,uGACF;CACF;CACA,OAAO;AACT;;;ACpmBA,SAAS,aAAa,UAAsC;CAC1D,IAAI,CAAC,UAAU,OAAO;CACtB,IAAI,SAAS,SAAS,MAAM,GAAG,OAAO;CACtC,IAAI,SAAS,WAAW,OAAO,GAAG,OAAO;CACzC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,qBAAqB,QAAoD;CACvF,MAAM,YAAwB,CAAC;CAC/B,MAAM,YAAgC,CAAC;CACvC,MAAM,YAAsB,CAAC;CAC7B,MAAM,4BAAY,IAAI,IAAY;CAElC,KAAK,MAAM,MAAM,QACf,IAAI,GAAG,SAAS,aAAa;EAC3B,MAAM,OAAQ,GAAyB;EACvC,IAAI,QAAQ,CAAC,UAAU,IAAI,IAAI,GAAG;GAChC,UAAU,IAAI,IAAI;GAClB,UAAU,KAAK,IAAI;EACrB;CACF,OAAO,IAAI,GAAG,SAAS,YAAY;EACjC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,MAAM,aAAa,EAAE,QAAQ;GAC7B,MAAM,EAAE,QAAQ,EAAE,OAAO,EAAE;GAC3B,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,OAAO,IAAI,GAAG,SAAS,oBAAoB;EACzC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,IAAI,EAAE;GACN,OAAO,EAAE;GACT,QAAQ,EAAE,UAAU;GACpB,GAAI,EAAE,YAAY,KAAA,IAAY,EAAE,SAAS,EAAE,QAAQ,IAAI,CAAC;EAC1D,CAAC;CACH;CAGF,OAAO;EAAE;EAAW;EAAW;CAAU;AAC3C"}
@@ -1,4 +1,4 @@
1
- import { R as Scenario, h as GateContext, p as Gate, v as GateResult } from "./types-BxLccGMf.js";
1
+ import { R as Scenario, h as GateContext, p as Gate, v as GateResult } from "./types-BjsNDR49.js";
2
2
  import { d as PairedBootstrapResult, i as PairedMcNemarEvidence, r as PairedDecisionStatistic, t as PairedDecisionMethod } from "./paired-promotion-decision-CGzg0cI_.js";
3
3
  import { t as Direction } from "./pareto-BqNW3LJR.js";
4
4
  //#region src/campaign/gates/promotion-policy.d.ts
@@ -131,4 +131,4 @@ interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
131
131
  declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
132
132
  //#endregion
133
133
  export { ObjectiveSource as a, PromotionPolicy as c, paretoSignificanceGate as d, EvidenceVector as i, buildEvidenceVector as l, AxisVerdict as n, ParetoSignificanceGateOptions as o, BuildEvidenceVectorOptions as r, PromotionObjective as s, AxisEvidence as t, paretoPolicy as u };
134
- //# sourceMappingURL=promotion-policy-CsMZJOB-.d.ts.map
134
+ //# sourceMappingURL=promotion-policy-u3wj6w3U.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"promotion-policy-CsMZJOB-.d.ts","names":[],"sources":["../src/campaign/gates/promotion-policy.ts"],"mappings":";;;;;;KA0CY;EAAoB;;EAAwB;EAAmB;;UAE1D;;EAEf;EACA,QAAQ;;;EAGR,WAAW;;;;EAIX;;;;;EAKA;;;KAIU;UAEK;EACf;EACA,QAAQ;EACR,WAAW;;;;;;;;EAQX,WAAW;;;;;EAKX;;;EAGA;IAAM;IAAa;;;EAEnB,mBAAmB;;EAEnB,SAAS;;;EAGT;;EAEA;EACA;EACA,gBAAgB;EAChB;EACA;EACA,SAAS;;UAGM;;EAEf,MAAM;;;EAGN;;;EAGA;IAAQ;IAAmB;;;;;;KAMjB,mBAAmB,IAAI,mBAAmB;UAErC;;;;EAIf;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;;iBASc,oBAAoB,WAAW,kBAAkB,UAC/D,KAAK,YAAY,WAAW,YAC5B,YAAY,sBACZ,OAAM,6BACL;;;;;;;;cAwIU,cAAc;UAiFV,sCAAsC;;EAErD,YAAY;;;EAGZ,SAAS;;EAET;;;;;;;iBAQc,uBAAuB,qBAAqB,kBAAkB,WAAW,UACvF,SAAS,gCACR,KAAK,WAAW"}
1
+ {"version":3,"file":"promotion-policy-u3wj6w3U.d.ts","names":[],"sources":["../src/campaign/gates/promotion-policy.ts"],"mappings":";;;;;;KA0CY;EAAoB;;EAAwB;EAAmB;;UAE1D;;EAEf;EACA,QAAQ;;;EAGR,WAAW;;;;EAIX;;;;;EAKA;;;KAIU;UAEK;EACf;EACA,QAAQ;EACR,WAAW;;;;;;;;EAQX,WAAW;;;;;EAKX;;;EAGA;IAAM;IAAa;;;EAEnB,mBAAmB;;EAEnB,SAAS;;;EAGT;;EAEA;EACA;EACA,gBAAgB;EAChB;EACA;EACA,SAAS;;UAGM;;EAEf,MAAM;;;EAGN;;;EAGA;IAAQ;IAAmB;;;;;;KAMjB,mBAAmB,IAAI,mBAAmB;UAErC;;;;EAIf;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;;iBASc,oBAAoB,WAAW,kBAAkB,UAC/D,KAAK,YAAY,WAAW,YAC5B,YAAY,sBACZ,OAAM,6BACL;;;;;;;;cAwIU,cAAc;UAiFV,sCAAsC;;EAErD,YAAY;;;EAGZ,SAAS;;EAET;;;;;;;iBAQc,uBAAuB,qBAAqB,kBAAkB,WAAW,UACvF,SAAS,gCACR,KAAK,WAAW"}
@@ -1,11 +1,11 @@
1
1
  import { c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt, o as CostLedger, p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
2
- import { a as RunRecord } from "./run-record-CKiihE6f.js";
2
+ import { a as RunRecord } from "./run-record-DVV82Gwh.js";
3
3
  import { p as ChatClient } from "./types-Cx3YUh2r.js";
4
- import { _ as ProposalFinding } from "./types-DLQx4mKU.js";
4
+ import { _ as ProposalFinding } from "./types-vXyshMwx.js";
5
5
  import { g as ExternalTextCandidate, o as ExternalOptimizerEvaluationObservation } from "./external-optimizer-contracts-lixrOZdX.js";
6
- import { C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-BxLccGMf.js";
6
+ import { C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-BjsNDR49.js";
7
7
  import { r as DatasetScenario, t as Dataset } from "./dataset-CJjKqQfA.js";
8
- import { g as TraceSpanEvent, t as HostedClient } from "./client-DCVe0CwG.js";
8
+ import { g as TraceSpanEvent, t as HostedClient } from "./client-D_TIV9pJ.js";
9
9
  import { z } from "zod";
10
10
  //#region src/llm-judge.d.ts
11
11
  /** A rubric dimension as a bare key or the full `{ key, description }` shape. A
@@ -1185,4 +1185,4 @@ interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends
1185
1185
  declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
1186
1186
  //#endregion
1187
1187
  export { OptimizationTokenUsage as $, RedTeamCategory as A, llmJudge as At, CanaryReport as B, RunOptimizationOptions as C, fsCampaignStorage as Ct, runEval as D, openAutoPr as Dt, RunEvalOptions as E, OpenAutoPrResult as Et, scoreRedTeamOutput as F, OptimizationMethodComparison as G, CompareOptimizationMethodsOptions as H, CanaryAlert as I, OptimizationMethodProvenance as J, OptimizationMethodInput as K, CanaryEvaluation as L, RedTeamReport as M, redTeamDataset as N, DEFAULT_RED_TEAM_CORPUS as O, LlmJudgeDimension as Ot, redTeamReport as P, OptimizationPackageSource as Q, CanaryKind as R, PremeasuredOptimizationBaseline as S, createRunCostLedger as St, runOptimization as T, OpenAutoPrOptions as Tt, ComparisonCost as U, runCanaries as V, OptimizationMethod as W, OptimizationMethodRunOptions as X, OptimizationMethodResult as Y, OptimizationMethodScore as Z, provenanceSpansPath as _, ExternalOptimizerObservationArtifact as _t, LoopProvenanceBackend as a, RunCampaignOptions as at, RunImprovementLoopResult as b, readExternalOptimizerObservationArtifact as bt, LoopProvenanceOptimizationMethod as c, CampaignRunPlanCell as ct, campaignMeasurementDigest as d, GepaCandidatePopulationArtifact as dt, combineComparisonCosts as et, canonicalDigest as f, GepaCandidatePopulationCandidate as ft, provenanceRecordPath as g, ExternalOptimizerExecutionSummary as gt, loopProvenanceSpans as h, readGepaCandidatePopulationArtifact as ht, LoopProvenanceArgsFromResult as i, CampaignCellFailureReceipt as it, RedTeamFinding as j, RedTeamCase as k, LlmJudgeOptions as kt, LoopProvenanceRecord as l, PlanCampaignRunOptions as lt, loopProvenanceArgsFromResult as m, GepaCandidateSelectionScore as mt, EmitLoopProvenanceArgs as n, costFromLedgerSummary as nt, LoopProvenanceCandidate as o, runCampaign as ot, emitLoopProvenance as p, GepaCandidatePopulationSummary as pt, OptimizationMethodPairwise as q, EmitLoopProvenanceResult as r, optimizationTokenUsageFromSummary as rt, LoopProvenanceEvidence as s, CampaignRunPlan as st, BuildLoopProvenanceArgs as t, compareOptimizationMethods as tt, buildLoopProvenanceRecord as u, planCampaignRun as ut, verifyLoopProvenanceRecord as v, ExternalOptimizerObservationSummary as vt, RunOptimizationResult as w, inMemoryCampaignStorage as wt, runImprovementLoop as x, CampaignStorage as xt, RunImprovementLoopOptions as y, ExternalOptimizerSubmittedCandidate as yt, CanaryOptions as z };
1188
- //# sourceMappingURL=provenance-8L-_4xiL.d.ts.map
1188
+ //# sourceMappingURL=provenance-BOtMtoiZ.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"provenance-8L-_4xiL.d.ts","names":[],"sources":["../src/llm-judge.ts","../src/campaign/auto-pr.ts","../src/campaign/storage.ts","../src/campaign/external-optimizer-observations.ts","../src/campaign/gepa-candidate-population.ts","../src/campaign/cell-cache.ts","../src/campaign/plan-campaign-run.ts","../src/campaign/run-campaign.ts","../src/campaign/presets/compare-optimization-methods.ts","../src/canary.ts","../src/red-team.ts","../src/campaign/presets/run-eval.ts","../src/campaign/presets/run-optimization.ts","../src/campaign/presets/run-improvement-loop.ts","../src/campaign/provenance.ts"],"mappings":";;;;;;;;;;;;KA6CY,6BAA6B;UAExB,gBAAgB,WAAW,kBAAkB,WAAW;;;;EAIvE,MAAM;;;EAGN,aAAa;;EAEb;;EAEA;EACA;EACA;;;EAGA,UAAU;;;;;EAKV;;EAEA,aAAa,UAAU;;;EAGvB,cAAc;IAAS,UAAU;IAAW,UAAU;;;EAEtD,aAAa;EACb;IAAmB;IAAc,QAAQ,EAAE;;;;;;;;;;;;;iBAoB7B,SAAS,qBAAqB,kBAAkB,WAAW,UACzE,cACA,gBACA,MAAM,gBAAgB,WAAW,aAChC,YAAY,WAAW;;;UClFT,kBAAkB,WAAW,kBAAkB;;EAE9D,QAAQ,eAAe,WAAW;;;EAGlC,MAAM;;;EAGN;;EAEA;EACA;;EAEA;;EAEA;;;EAGA;;EAEA,UAAU;IAAqB;IAAgB;IAAgB;;;UAGhD;EACf;EACA;EACA;EACA;;;;;iBAMc,WAAW,WAAW,kBAAkB,UACtD,SAAS,kBAAkB,WAAW,aACrC;;;;;;;;;;;;;;;;;;UCjCc;;EAEf,UAAU;;EAEV,OAAO;;EAEP,KAAK;;EAEL,MAAM,cAAc,kBAAkB;;;EAGtC,OAAO,cAAc,iBAAiB;;;;;;;;;iBAUxB,qBAAqB;;;;iBA0CrB,2BAA2B;;iBAmC3B,oBAAoB;EAClC,SAAS;EACT;EACA;;EAEA;IACE;;;UC/Ga;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;;WAEN,WAAW;;WAEX;WACA;WACA;;WAEA;aACE;aACA;;;UAII;WACN,SAAS;WACT,uBAAuB;;WAEvB,qBAAqB;;;;;;;;;;iBAWhB,yCAAyC;EACvD,SAAS;EACT,UAAU;IACR;;;UCrDa;WACN;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;UAGM;WACN;WACA;;UAGM;;WAEN;WACA,WAAW;WACX;WACA;;WAEA;;WAEA;WACA,0BAA0B;WAC1B;;UAGM;WACN,SAAS;WACT;WACA;WACA,qBAAqB;;;;;;;;;;iBAWhB,oCAAoC;EAClD,SAAS;EACT,UAAU;IACR;;;KCuDQ;;;UCtGK;EACf;EACA;EACA;EACA;EACA;EACA;EACA,SAAS;;UAGM;EACf;EACA;EACA;EACA;EACA;EACA;EACA,OAAO;;UAGQ,uBAAuB,kBAAkB,UAAU;EAClE,WAAW;EACX,WAAW,WAAW,WAAW;EACjC;EACA,SAAS,YAAY,WAAW;EAChC;EACA;EACA;;EAEA;EACA;;EAEA;EACA,UAAU;;EAEV,aAAa;;EAEb,WAAW,SAAS;;;;;;iBAON,gBAAgB,kBAAkB,UAAU,WAC1D,MAAM,uBAAuB,WAAW,aACvC;;;UCvBc,mBAAmB,kBAAkB,UAAU;EAC9D,WAAW;EACX,UAAU,WAAW,WAAW;;EAEhC,SAAS;;;;;;EAMT;EACA,SAAS,YAAY,WAAW;;EAEhC;;;EAGA;;;EAGA;;;;;;EAMA,cAAc;IAAS,UAAU;IAAW;;;;;;EAK5C;;;;;;EAMA;;;;EAIA,eAAe;EACf;EACA;;EAEA;;;EAGA,aAAa;;EAEb;;EAEA,WAAW,SAAS;;EAEpB;;;;;;;;EAQA;;;;;;;;;;EAUA;;;;;EAKA;;;;EAIA;;;EAGA;;;;EAIA;;;;;;;;;;;EAWA;;EAEA,YAAY;;EAEZ,oBAAoB,gBAAgB,gBAAgB;;;;;;EAMpD,UAAU;;;;;;;;;;;;EAYV,iBAAiB;IACf,UAAU;IACV;IACA;;;;;;;UAQa,2BAA2B;EAC1C;EACA;EACA;EACA;IACE;IACA;IACA;MACE;MACA;MACA;;;EAGJ,MAAM,mBAAmB;EACzB,MAAM;;;;;iBAMc,YAAY,kBAAkB,UAAU,WAC5D,MAAM,mBAAmB,WAAW,aACnC,QAAQ,eAAe,WAAW;;;;KCtJzB,6BAA6B,kBAAkB,UAAU,aAAa,KAChF,mBAAmB,WAAW;;UAKf;;EAEf;;EAEA,gBAAgB;EAChB;EACA;;UAGe;EACf;;EAEA;EACA;EACA;EACA;EACA;;EAEA;;UAGe;EACf;EACA;;UAGe;EACf;EACA;;UAGe;;EAEf;;EAEA;;EAEA;EACA;;EAEA;EACA;EACA;;UAGe;;EAEf,QAAQ;;EAER,SAAS;;EAET,UAAU;;EAEV,SAAS;;EAET;;EAEA;EACA;;EAEA;EACA;EACA;EACA;EACA,aAAa;;EAEb,eAAe;;EAEf,0BAA0B;;EAE1B,kBAAkB;;;UAIH,wBAAwB,kBAAkB,UAAU;;WAE1D,iBAAiB;;WAEjB,yBAAyB;;WAEzB,6BAA6B;;WAE7B,sBACP,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;;WAEJ,iBAAiB,YAAY,WAAW;;WAExC;WACA;;WAEA,YAAY,SAAS,6BAA6B,WAAW;;WAE7D,YAAY;;UAGN;;EAEf,eAAe;;EAEf,MAAM;;EAEN;;EAEA,aAAa;;;UAIE,mBAAmB,kBAAkB,WAAW,UAAU;;EAEzE;EACA,WACE,OAAO,wBAAwB,WAAW,eACvC,QAAQ;;UAGE;EACf;;EAEA;;EAEA;;EAEA;;;EAGA;IAAU;IAAa;;;EAEvB,kBAAkB;;EAElB;;EAEA,aAAa;;EAEb,gBAAgB;IACd;IACA;IACA;IACA;;EAEF,eAAe;;EAEf;;UAGe;;EAEf;EACA;;EAEA;EACA;EACA;;EAEA;;UAGe;;EAEf,QAAQ;EACR,MAAM;;EAEN,UAAU;EACV;;EAEA,kBAAkB;;EAElB,UAAU;;EAEV,WAAW;;EAEX;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,kCAAkC,kBAAkB,UAAU,mBACrE,KAAK,mBAAmB,WAAW;EAC3C,SAAS,mBAAmB,WAAW;EACvC,iBAAiB;;EAEjB,gBAAgB;;EAEhB,oBAAoB;;EAEpB,eAAe;;EAEf,sBACE,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;EACb,QAAQ,YAAY,WAAW;;;EAG/B;;EAEA,yBAAyB,6BAA6B,WAAW;;EAEjE;;;EAGA;;EAEA;;;;;iBAMoB,2BAA2B,kBAAkB,UAAU,WAC3E,MAAM,kCAAkC,WAAW,aAClD,QAAQ;;iBAkwBK,sBAAsB,SAAS,oBAAoB;;iBAYnD,kCACd,SAAS,mBACT,mBAAmB,gBAClB;;iBA0Ba,uBACd,SAAS;EAAgB;EAAe,MAAM;KAC7C;;;KCzhCS;KAEA;UAEK;EACf,MAAM;EACN,UAAU;EACV;;;EAGA,UAAU;;UAGK;EACf,QAAQ;;EAER,QAAQ,OAAO;;EAEf,aAAa;;UAGE;EACf,MAAM;EACN;EACA;EACA;;UAGe;;;;;;;;;;EAUf;IACE;IACA;;IAEA;;;;;;;;;;;;;EAcF;IACE;IACA;IACA;IACA;;;;;;;;;;EAWF;IACE,WAAW,KAAK;IAChB;IACA;IACA;IACA;;;;;;;;iBASY,YAAY,MAAM,aAAa,OAAM,gBAAqB;;;KClG9D;UAUK;EACf,UAAU;;EAEV;;;;;EAKA;;EAEA;;EAEA;;UAGe,oBAAoB;EACnC,SAAS;;UAGM;EACf;EACA,UAAU;EACV;EACA;EACA;;UAGe;EACf,UAAU;EACV,oBAAoB,OAAO;EAC3B;;;cA0DW,yBAAyB;iBA6FtB,eAAe,aAAY,gBAAqB;;;;;iBAkBhD,mBACd,gBACA,qBACA,QAAQ,cACP;;iBA4Fa,cAAc,UAAU,mBAAmB;;;UCzT1C,eAAe,kBAAkB,UAAU,mBAClD,KAAK,mBAAmB,WAAW;EAC3C;;;;;iBAMoB,QAAQ,kBAAkB,UAAU,WACxD,MAAM,eAAe,WAAW,aAC/B,QAAQ,eAAe,WAAW;;;UC0BpB,gCAAgC,WAAW,kBAAkB;;EAE5E;;EAEA,UAAU,eAAe,WAAW;;UAGrB,2BAA2B,kBAAkB,UAAU,mBAC9D,KAAK,mBAAmB,WAAW;;EAE3C,iBAAiB;;;;;;;;;EASjB,sBAAsB,gCAAgC,WAAW;;EAEjE,sBACE,SAAS,gBACT,UAAU,WACV,KAAK,WAAW,mBAAmB,WAAW,+BAC3C,QAAQ;;EAEb,UAAU,gBAAgB;EAC1B;EACA;;;EAGA;;;EAGA;;EAEA,WAAW,cAAc;;;;;;;;;;;EAWzB,qBAAqB;IACnB;IACA;IACA,YAAY;MACV;MACA,UAAU,eAAe,WAAW;MACpC;;IAEF,SAAS;;IAET,aAAa;IACb;QACI,QAAQ,cAAc;;;;;;;;;;;;;;;;;;EAkB5B,oBAAoB,UAAU,eAAe,WAAW;;KAG9C,uBACV,kBAAkB,UAClB,aACE,2BAA2B,WAAW;UAEzB,sBAAsB,WAAW,kBAAkB;EAClE,aAAa;IACX,QAAQ;IACR,UAAU;MACR;MACA,SAAS;MACT,UAAU,eAAe,WAAW;;;;EAIxC,iBAAiB;EACjB,eAAe;EACf;;;;EAIA;;;;EAIA;EACA,kBAAkB,eAAe,WAAW;;EAE5C,MAAM;;;;;;EAMN,gBAAgB;;;;;iBAMI,gBAAgB,kBAAkB,UAAU,WAChE,MAAM,uBAAuB,WAAW,aACvC,QAAQ,sBAAsB,WAAW;;;KCpJhC,0BACV,kBAAkB,UAClB,aACE,uBAAuB,WAAW;;;EAGpC,kBAAkB;;;;;;;;;EASlB;;;;EAIA,MAAM,KAAK,WAAW;;;;;EAKtB;;EAEA;EACA;;;;;;;;EAQA,cAAc,eAAe,gBAAgB,iBAAiB,mBAAmB;;UAGlE,yBAAyB,WAAW,kBAAkB,kBAC7D,sBAAsB,WAAW;EACzC,mBAAmB,eAAe,WAAW;EAC7C,iBAAiB,eAAe,WAAW;EAC3C,uBAAuB,eAAe,WAAW;EACjD,qBAAqB;EACrB,YAAY,QAAQ,WAAW,KAAK,WAAW;;;;EAI/C;;;;;EAKA;EACA,WAAW,kBAAkB;;;;;iBAMT,mBAAmB,kBAAkB,UAAU,WACnE,MAAM,0BAA0B,WAAW,aAC1C,QAAQ,yBAAyB,WAAW;;;UC1B9B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,UAAU,YAAY;;EAEtB;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA;EACA;EACA;EACA;;UAGe;EACf;IACE;IACA;;EAEF;IACE;IACA;IACA;IACA;MACE;MACA;MACA;MACA;;;EAGJ;;UAGe;EACf;EACA,MAAM;EACN;EACA,aAAa;;;;;;;UAQE;EACf;;EAEA;EACA;EACA;EACA;;EAEA;EACA;;EAEA;EACA;;EAEA;;EAEA,YAAY;;EAEZ,qBAAqB;;EAErB,UAAU;;EAEV;;EAEA;IACE,UAAU;IACV;IACA;IACA,mBAAmB;;;;;;EAMrB;;EAEA;;EAEA;;;EAGA;;EAEA,SAAS;EACT;EACA;;UAGe,wBAAwB,WAAW,kBAAkB;EACpE;EACA;EACA;EACA,iBAAiB;EACjB,eAAe;EACf;EACA;;EAEA,wBAAwB,eAAe,WAAW;;EAElD,aAAa;IACX;IACA,YAAY;IACZ;;;IAGA,UAAU;MACR;MACA,SAAS;MACT,UAAU,eAAe,WAAW;;;EAGxC,MAAM;;;;EAIN;EACA,mBAAmB,eAAe,WAAW;EAC7C,iBAAiB,eAAe,WAAW;EAC3C,qBAAqB;EACrB,uBAAuB,eAAe,WAAW;;EAEjD,cAAc,cAAc;EAC5B;EACA;EACA,qBAAqB;;UAGN,6BAA6B,WAAW,kBAAkB;EACzE;EACA;EACA;EACA,iBAAiB;EACjB,QAAQ,yBAAyB,WAAW;EAC5C,cAAc,cAAc;EAC5B;EACA;;;iBAIc,6BAA6B,WAAW,kBAAkB,UACxE,OAAO,6BAA6B,WAAW,aAC9C,wBAAwB,WAAW;;iBA4CtB,0BAA0B,WAAW,kBAAkB,UACrE,MAAM,wBAAwB,WAAW,aACxC;;iBAoRa,0BAA0B,WAAW,kBAAkB,UACrE,UAAU,eAAe,WAAW;;iBAqCtB,2BAA2B,QAAQ,uBAAuB;;iBAa1D,gBAAgB;;;;;;;;;;;;iBA2KhB,oBACd,QAAQ,sBACR;EAAQ;IACP;;iBAiJa,qBAAqB;;;;iBAMrB,oBAAoB;UAInB;EACf,QAAQ;EACR,OAAO;;EAEP;EACA;;UAGe,uBAAuB,WAAW,kBAAkB,kBAC3D,wBAAwB,WAAW;;EAE3C,SAAS;;;EAGT,eAAe;;;;;;;;;;;;iBA+EK,mBAAmB,WAAW,kBAAkB,UACpE,MAAM,uBAAuB,WAAW,aACvC,QAAQ"}
1
+ {"version":3,"file":"provenance-BOtMtoiZ.d.ts","names":[],"sources":["../src/llm-judge.ts","../src/campaign/auto-pr.ts","../src/campaign/storage.ts","../src/campaign/external-optimizer-observations.ts","../src/campaign/gepa-candidate-population.ts","../src/campaign/cell-cache.ts","../src/campaign/plan-campaign-run.ts","../src/campaign/run-campaign.ts","../src/campaign/presets/compare-optimization-methods.ts","../src/canary.ts","../src/red-team.ts","../src/campaign/presets/run-eval.ts","../src/campaign/presets/run-optimization.ts","../src/campaign/presets/run-improvement-loop.ts","../src/campaign/provenance.ts"],"mappings":";;;;;;;;;;;;KA6CY,6BAA6B;UAExB,gBAAgB,WAAW,kBAAkB,WAAW;;;;EAIvE,MAAM;;;EAGN,aAAa;;EAEb;;EAEA;EACA;EACA;;;EAGA,UAAU;;;;;EAKV;;EAEA,aAAa,UAAU;;;EAGvB,cAAc;IAAS,UAAU;IAAW,UAAU;;;EAEtD,aAAa;EACb;IAAmB;IAAc,QAAQ,EAAE;;;;;;;;;;;;;iBAoB7B,SAAS,qBAAqB,kBAAkB,WAAW,UACzE,cACA,gBACA,MAAM,gBAAgB,WAAW,aAChC,YAAY,WAAW;;;UClFT,kBAAkB,WAAW,kBAAkB;;EAE9D,QAAQ,eAAe,WAAW;;;EAGlC,MAAM;;;EAGN;;EAEA;EACA;;EAEA;;EAEA;;;EAGA;;EAEA,UAAU;IAAqB;IAAgB;IAAgB;;;UAGhD;EACf;EACA;EACA;EACA;;;;;iBAMc,WAAW,WAAW,kBAAkB,UACtD,SAAS,kBAAkB,WAAW,aACrC;;;;;;;;;;;;;;;;;;UCjCc;;EAEf,UAAU;;EAEV,OAAO;;EAEP,KAAK;;EAEL,MAAM,cAAc,kBAAkB;;;EAGtC,OAAO,cAAc,iBAAiB;;;;;;;;;iBAUxB,qBAAqB;;;;iBA0CrB,2BAA2B;;iBAmC3B,oBAAoB;EAClC,SAAS;EACT;EACA;;EAEA;IACE;;;UC/Ga;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;;WAEN,WAAW;;WAEX;WACA;WACA;;WAEA;aACE;aACA;;;UAII;WACN,SAAS;WACT,uBAAuB;;WAEvB,qBAAqB;;;;;;;;;;iBAWhB,yCAAyC;EACvD,SAAS;EACT,UAAU;IACR;;;UCrDa;WACN;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;UAGM;WACN;WACA;;UAGM;;WAEN;WACA,WAAW;WACX;WACA;;WAEA;;WAEA;WACA,0BAA0B;WAC1B;;UAGM;WACN,SAAS;WACT;WACA;WACA,qBAAqB;;;;;;;;;;iBAWhB,oCAAoC;EAClD,SAAS;EACT,UAAU;IACR;;;KCuDQ;;;UCtGK;EACf;EACA;EACA;EACA;EACA;EACA;EACA,SAAS;;UAGM;EACf;EACA;EACA;EACA;EACA;EACA;EACA,OAAO;;UAGQ,uBAAuB,kBAAkB,UAAU;EAClE,WAAW;EACX,WAAW,WAAW,WAAW;EACjC;EACA,SAAS,YAAY,WAAW;EAChC;EACA;EACA;;EAEA;EACA;;EAEA;EACA,UAAU;;EAEV,aAAa;;EAEb,WAAW,SAAS;;;;;;iBAON,gBAAgB,kBAAkB,UAAU,WAC1D,MAAM,uBAAuB,WAAW,aACvC;;;UCvBc,mBAAmB,kBAAkB,UAAU;EAC9D,WAAW;EACX,UAAU,WAAW,WAAW;;EAEhC,SAAS;;;;;;EAMT;EACA,SAAS,YAAY,WAAW;;EAEhC;;;EAGA;;;EAGA;;;;;;EAMA,cAAc;IAAS,UAAU;IAAW;;;;;;EAK5C;;;;;;EAMA;;;;EAIA,eAAe;EACf;EACA;;EAEA;;;EAGA,aAAa;;EAEb;;EAEA,WAAW,SAAS;;EAEpB;;;;;;;;EAQA;;;;;;;;;;EAUA;;;;;EAKA;;;;EAIA;;;EAGA;;;;EAIA;;;;;;;;;;;EAWA;;EAEA,YAAY;;EAEZ,oBAAoB,gBAAgB,gBAAgB;;;;;;EAMpD,UAAU;;;;;;;;;;;;EAYV,iBAAiB;IACf,UAAU;IACV;IACA;;;;;;;UAQa,2BAA2B;EAC1C;EACA;EACA;EACA;IACE;IACA;IACA;MACE;MACA;MACA;;;EAGJ,MAAM,mBAAmB;EACzB,MAAM;;;;;iBAMc,YAAY,kBAAkB,UAAU,WAC5D,MAAM,mBAAmB,WAAW,aACnC,QAAQ,eAAe,WAAW;;;;KCtJzB,6BAA6B,kBAAkB,UAAU,aAAa,KAChF,mBAAmB,WAAW;;UAKf;;EAEf;;EAEA,gBAAgB;EAChB;EACA;;UAGe;EACf;;EAEA;EACA;EACA;EACA;EACA;;EAEA;;UAGe;EACf;EACA;;UAGe;EACf;EACA;;UAGe;;EAEf;;EAEA;;EAEA;EACA;;EAEA;EACA;EACA;;UAGe;;EAEf,QAAQ;;EAER,SAAS;;EAET,UAAU;;EAEV,SAAS;;EAET;;EAEA;EACA;;EAEA;EACA;EACA;EACA;EACA,aAAa;;EAEb,eAAe;;EAEf,0BAA0B;;EAE1B,kBAAkB;;;UAIH,wBAAwB,kBAAkB,UAAU;;WAE1D,iBAAiB;;WAEjB,yBAAyB;;WAEzB,6BAA6B;;WAE7B,sBACP,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;;WAEJ,iBAAiB,YAAY,WAAW;;WAExC;WACA;;WAEA,YAAY,SAAS,6BAA6B,WAAW;;WAE7D,YAAY;;UAGN;;EAEf,eAAe;;EAEf,MAAM;;EAEN;;EAEA,aAAa;;;UAIE,mBAAmB,kBAAkB,WAAW,UAAU;;EAEzE;EACA,WACE,OAAO,wBAAwB,WAAW,eACvC,QAAQ;;UAGE;EACf;;EAEA;;EAEA;;EAEA;;;EAGA;IAAU;IAAa;;;EAEvB,kBAAkB;;EAElB;;EAEA,aAAa;;EAEb,gBAAgB;IACd;IACA;IACA;IACA;;EAEF,eAAe;;EAEf;;UAGe;;EAEf;EACA;;EAEA;EACA;EACA;;EAEA;;UAGe;;EAEf,QAAQ;EACR,MAAM;;EAEN,UAAU;EACV;;EAEA,kBAAkB;;EAElB,UAAU;;EAEV,WAAW;;EAEX;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,kCAAkC,kBAAkB,UAAU,mBACrE,KAAK,mBAAmB,WAAW;EAC3C,SAAS,mBAAmB,WAAW;EACvC,iBAAiB;;EAEjB,gBAAgB;;EAEhB,oBAAoB;;EAEpB,eAAe;;EAEf,sBACE,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;EACb,QAAQ,YAAY,WAAW;;;EAG/B;;EAEA,yBAAyB,6BAA6B,WAAW;;EAEjE;;;EAGA;;EAEA;;;;;iBAMoB,2BAA2B,kBAAkB,UAAU,WAC3E,MAAM,kCAAkC,WAAW,aAClD,QAAQ;;iBAkwBK,sBAAsB,SAAS,oBAAoB;;iBAYnD,kCACd,SAAS,mBACT,mBAAmB,gBAClB;;iBA0Ba,uBACd,SAAS;EAAgB;EAAe,MAAM;KAC7C;;;KCzhCS;KAEA;UAEK;EACf,MAAM;EACN,UAAU;EACV;;;EAGA,UAAU;;UAGK;EACf,QAAQ;;EAER,QAAQ,OAAO;;EAEf,aAAa;;UAGE;EACf,MAAM;EACN;EACA;EACA;;UAGe;;;;;;;;;;EAUf;IACE;IACA;;IAEA;;;;;;;;;;;;;EAcF;IACE;IACA;IACA;IACA;;;;;;;;;;EAWF;IACE,WAAW,KAAK;IAChB;IACA;IACA;IACA;;;;;;;;iBASY,YAAY,MAAM,aAAa,OAAM,gBAAqB;;;KClG9D;UAUK;EACf,UAAU;;EAEV;;;;;EAKA;;EAEA;;EAEA;;UAGe,oBAAoB;EACnC,SAAS;;UAGM;EACf;EACA,UAAU;EACV;EACA;EACA;;UAGe;EACf,UAAU;EACV,oBAAoB,OAAO;EAC3B;;;cA0DW,yBAAyB;iBA6FtB,eAAe,aAAY,gBAAqB;;;;;iBAkBhD,mBACd,gBACA,qBACA,QAAQ,cACP;;iBA4Fa,cAAc,UAAU,mBAAmB;;;UCzT1C,eAAe,kBAAkB,UAAU,mBAClD,KAAK,mBAAmB,WAAW;EAC3C;;;;;iBAMoB,QAAQ,kBAAkB,UAAU,WACxD,MAAM,eAAe,WAAW,aAC/B,QAAQ,eAAe,WAAW;;;UC0BpB,gCAAgC,WAAW,kBAAkB;;EAE5E;;EAEA,UAAU,eAAe,WAAW;;UAGrB,2BAA2B,kBAAkB,UAAU,mBAC9D,KAAK,mBAAmB,WAAW;;EAE3C,iBAAiB;;;;;;;;;EASjB,sBAAsB,gCAAgC,WAAW;;EAEjE,sBACE,SAAS,gBACT,UAAU,WACV,KAAK,WAAW,mBAAmB,WAAW,+BAC3C,QAAQ;;EAEb,UAAU,gBAAgB;EAC1B;EACA;;;EAGA;;;EAGA;;EAEA,WAAW,cAAc;;;;;;;;;;;EAWzB,qBAAqB;IACnB;IACA;IACA,YAAY;MACV;MACA,UAAU,eAAe,WAAW;MACpC;;IAEF,SAAS;;IAET,aAAa;IACb;QACI,QAAQ,cAAc;;;;;;;;;;;;;;;;;;EAkB5B,oBAAoB,UAAU,eAAe,WAAW;;KAG9C,uBACV,kBAAkB,UAClB,aACE,2BAA2B,WAAW;UAEzB,sBAAsB,WAAW,kBAAkB;EAClE,aAAa;IACX,QAAQ;IACR,UAAU;MACR;MACA,SAAS;MACT,UAAU,eAAe,WAAW;;;;EAIxC,iBAAiB;EACjB,eAAe;EACf;;;;EAIA;;;;EAIA;EACA,kBAAkB,eAAe,WAAW;;EAE5C,MAAM;;;;;;EAMN,gBAAgB;;;;;iBAMI,gBAAgB,kBAAkB,UAAU,WAChE,MAAM,uBAAuB,WAAW,aACvC,QAAQ,sBAAsB,WAAW;;;KCpJhC,0BACV,kBAAkB,UAClB,aACE,uBAAuB,WAAW;;;EAGpC,kBAAkB;;;;;;;;;EASlB;;;;EAIA,MAAM,KAAK,WAAW;;;;;EAKtB;;EAEA;EACA;;;;;;;;EAQA,cAAc,eAAe,gBAAgB,iBAAiB,mBAAmB;;UAGlE,yBAAyB,WAAW,kBAAkB,kBAC7D,sBAAsB,WAAW;EACzC,mBAAmB,eAAe,WAAW;EAC7C,iBAAiB,eAAe,WAAW;EAC3C,uBAAuB,eAAe,WAAW;EACjD,qBAAqB;EACrB,YAAY,QAAQ,WAAW,KAAK,WAAW;;;;EAI/C;;;;;EAKA;EACA,WAAW,kBAAkB;;;;;iBAMT,mBAAmB,kBAAkB,UAAU,WACnE,MAAM,0BAA0B,WAAW,aAC1C,QAAQ,yBAAyB,WAAW;;;UC1B9B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,UAAU,YAAY;;EAEtB;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA;EACA;EACA;EACA;;UAGe;EACf;IACE;IACA;;EAEF;IACE;IACA;IACA;IACA;MACE;MACA;MACA;MACA;;;EAGJ;;UAGe;EACf;EACA,MAAM;EACN;EACA,aAAa;;;;;;;UAQE;EACf;;EAEA;EACA;EACA;EACA;;EAEA;EACA;;EAEA;EACA;;EAEA;;EAEA,YAAY;;EAEZ,qBAAqB;;EAErB,UAAU;;EAEV;;EAEA;IACE,UAAU;IACV;IACA;IACA,mBAAmB;;;;;;EAMrB;;EAEA;;EAEA;;;EAGA;;EAEA,SAAS;EACT;EACA;;UAGe,wBAAwB,WAAW,kBAAkB;EACpE;EACA;EACA;EACA,iBAAiB;EACjB,eAAe;EACf;EACA;;EAEA,wBAAwB,eAAe,WAAW;;EAElD,aAAa;IACX;IACA,YAAY;IACZ;;;IAGA,UAAU;MACR;MACA,SAAS;MACT,UAAU,eAAe,WAAW;;;EAGxC,MAAM;;;;EAIN;EACA,mBAAmB,eAAe,WAAW;EAC7C,iBAAiB,eAAe,WAAW;EAC3C,qBAAqB;EACrB,uBAAuB,eAAe,WAAW;;EAEjD,cAAc,cAAc;EAC5B;EACA;EACA,qBAAqB;;UAGN,6BAA6B,WAAW,kBAAkB;EACzE;EACA;EACA;EACA,iBAAiB;EACjB,QAAQ,yBAAyB,WAAW;EAC5C,cAAc,cAAc;EAC5B;EACA;;;iBAIc,6BAA6B,WAAW,kBAAkB,UACxE,OAAO,6BAA6B,WAAW,aAC9C,wBAAwB,WAAW;;iBA4CtB,0BAA0B,WAAW,kBAAkB,UACrE,MAAM,wBAAwB,WAAW,aACxC;;iBAoRa,0BAA0B,WAAW,kBAAkB,UACrE,UAAU,eAAe,WAAW;;iBAqCtB,2BAA2B,QAAQ,uBAAuB;;iBAa1D,gBAAgB;;;;;;;;;;;;iBA2KhB,oBACd,QAAQ,sBACR;EAAQ;IACP;;iBAiJa,qBAAqB;;;;iBAMrB,oBAAoB;UAInB;EACf,QAAQ;EACR,OAAO;;EAEP;EACA;;UAGe,uBAAuB,WAAW,kBAAkB,kBAC3D,wBAAwB,WAAW;;EAE3C,SAAS;;;EAGT,eAAe;;;;;;;;;;;;iBA+EK,mBAAmB,WAAW,kBAAkB,UACpE,MAAM,uBAAuB,WAAW,aACvC,QAAQ"}
@@ -1,7 +1,7 @@
1
1
  import { c as CostLedgerHandle } from "./cost-ledger-DbQdN3nO.js";
2
2
  import { p as ChatClient } from "./types-Cx3YUh2r.js";
3
- import { c as AnalystRunInputs, i as AnalystFinding, l as AnalystRunResult, n as AnalystContext, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary } from "./types-DLQx4mKU.js";
4
- import { i as ExactAnalystRunEvent, o as ExactAnalystRunResult, u as ExactExecutionComponentIdentity } from "./exact-types-BH1twmAJ.js";
3
+ import { c as AnalystRunInputs, i as AnalystFinding, l as AnalystRunResult, n as AnalystContext, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary } from "./types-vXyshMwx.js";
4
+ import { i as ExactAnalystRunEvent, o as ExactAnalystRunResult, u as ExactExecutionComponentIdentity } from "./exact-types-CzbhVDr2.js";
5
5
  //#region src/analyst/registry.d.ts
6
6
  interface AnalystHooks {
7
7
  /** Legacy runs may mutate ctx; exact runs provide a frozen observational context. */
@@ -175,4 +175,4 @@ declare class AnalystRegistry {
175
175
  declare function assertExactRegistryRunOpts(value: unknown): asserts value is ExactRegistryRunOpts;
176
176
  //#endregion
177
177
  export { ExactAnalystBudgetPolicy as a, RegistryRunOpts as c, BudgetPolicy as i, assertExactRegistryRunOpts as l, AnalystRegistry as n, ExactAnalystRunExecutionError as o, AnalystRegistryOptions as r, ExactRegistryRunOpts as s, AnalystHooks as t };
178
- //# sourceMappingURL=registry-oJeeI4-a.d.ts.map
178
+ //# sourceMappingURL=registry-B_1Frl8a.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"registry-oJeeI4-a.d.ts","names":[],"sources":["../src/analyst/registry.ts"],"mappings":";;;;;UAqDiB;;EAEf,iBAAiB;IACf,SAAS;IACT,KAAK;IACL;aACS;;EAEX,gBAAgB;IACd,SAAS;IACT,SAAS;IACT,UAAU;IACV;aACS;;;;;;EAMX,SAAS;IACP,SAAS;IACT,OAAO;IACP;MACE,+BAA+B,QAAQ;;EAE3C,YAAY;IAAQ,QAAQ;aAA4B;;UAGzC;;EAEf;;EAEA,UAAU;;;;;;;EAOV,YAAY;IACV,SAAS;IACT;IACA;IACA;;;UAIa;;EAEf,OAAO;;EAEP,OAAO,aAAa,SAAS;;EAE7B,QAAQ;;EAER,gBAAgB;;EAEhB,gBAAgB;;EAEhB,eAAe;;UAGA;;EAEf;;EAEA;;EAEA,SAAS;;EAET;;EAEA,SAAS;;EAET,aAAa;;EAEb;;EAEA,OAAO;;;;;;;;;EASP,gBAAgB,cAAc,kBAAkB,eAAe,cAAc;;;;;;EAM7E;;;KAIU;WAEG;WACA;;WAGA;WACA;;WAEA,SAAS,SAAS;;;;;;;;;;UAWhB;WACN;WACA,QAAQ;WACR;WACA,QAAQ;WACR,YAAY;WACZ,oBAAoB;WACpB;WACA,MAAM,SAAS;WACf,eACL,cAAc,kBACd,SAAS,eAAe,cAAc;WAEjC;WACA;WACA;WACA;;;cAqCE,sCAAsC;WACxC;WACA,QAAQ;EAEjB,YAAY,iBAAiB,QAAQ,uBAAuB,UAAU;;cAU3D;mBACM;mBACA;EAEjB,YAAY,UAAS;EAIrB,SAAS,SAAS;EAsBlB,QAAQ;IACN;IACA;IACA;IACA,MAAM;;EAUF,IACJ,eACA,QAAQ,kBACR,UAAS,kBACR,QAAQ;;EAUL,SACJ,eACA,QAAQ,kBACR,SAAS,uBACR,QAAQ;;EAQJ,eACL,eACA,QAAQ,kBACR,SAAS,uBACR,eAAe;;;;;;;;;;;;EAmBX,UACL,eACA,QAAQ,kBACR,UAAS,kBACR,eAAe;UAIV;UA8BA;UAwFO;UA2aP;UAaA;UAUA;;;iBAiGM,2BAA2B,yBAAyB,SAAS"}
1
+ {"version":3,"file":"registry-B_1Frl8a.d.ts","names":[],"sources":["../src/analyst/registry.ts"],"mappings":";;;;;UAqDiB;;EAEf,iBAAiB;IACf,SAAS;IACT,KAAK;IACL;aACS;;EAEX,gBAAgB;IACd,SAAS;IACT,SAAS;IACT,UAAU;IACV;aACS;;;;;;EAMX,SAAS;IACP,SAAS;IACT,OAAO;IACP;MACE,+BAA+B,QAAQ;;EAE3C,YAAY;IAAQ,QAAQ;aAA4B;;UAGzC;;EAEf;;EAEA,UAAU;;;;;;;EAOV,YAAY;IACV,SAAS;IACT;IACA;IACA;;;UAIa;;EAEf,OAAO;;EAEP,OAAO,aAAa,SAAS;;EAE7B,QAAQ;;EAER,gBAAgB;;EAEhB,gBAAgB;;EAEhB,eAAe;;UAGA;;EAEf;;EAEA;;EAEA,SAAS;;EAET;;EAEA,SAAS;;EAET,aAAa;;EAEb;;EAEA,OAAO;;;;;;;;;EASP,gBAAgB,cAAc,kBAAkB,eAAe,cAAc;;;;;;EAM7E;;;KAIU;WAEG;WACA;;WAGA;WACA;;WAEA,SAAS,SAAS;;;;;;;;;;UAWhB;WACN;WACA,QAAQ;WACR;WACA,QAAQ;WACR,YAAY;WACZ,oBAAoB;WACpB;WACA,MAAM,SAAS;WACf,eACL,cAAc,kBACd,SAAS,eAAe,cAAc;WAEjC;WACA;WACA;WACA;;;cAqCE,sCAAsC;WACxC;WACA,QAAQ;EAEjB,YAAY,iBAAiB,QAAQ,uBAAuB,UAAU;;cAU3D;mBACM;mBACA;EAEjB,YAAY,UAAS;EAIrB,SAAS,SAAS;EAsBlB,QAAQ;IACN;IACA;IACA;IACA,MAAM;;EAUF,IACJ,eACA,QAAQ,kBACR,UAAS,kBACR,QAAQ;;EAUL,SACJ,eACA,QAAQ,kBACR,SAAS,uBACR,QAAQ;;EAQJ,eACL,eACA,QAAQ,kBACR,SAAS,uBACR,eAAe;;;;;;;;;;;;EAmBX,UACL,eACA,QAAQ,kBACR,UAAS,kBACR,eAAe;UAIV;UA8BA;UAwFO;UA2aP;UAaA;UAUA;;;iBAiGM,2BAA2B,yBAAyB,SAAS"}