@tangle-network/agent-eval 0.161.0 → 0.163.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/CHANGELOG.md +51 -0
  2. package/dist/{active-curriculum-CD5TU2yW.js → active-curriculum-OjIWgrUJ.js} +2 -2
  3. package/dist/{active-curriculum-CD5TU2yW.js.map → active-curriculum-OjIWgrUJ.js.map} +1 -1
  4. package/dist/analyst/index.d.ts +2 -2
  5. package/dist/analyst/index.d.ts.map +1 -1
  6. package/dist/analyst/index.js +3 -3
  7. package/dist/{baseline-BhPRQBVn.js → baseline-BC-eBZ7U.js} +2 -2
  8. package/dist/{baseline-BhPRQBVn.js.map → baseline-BC-eBZ7U.js.map} +1 -1
  9. package/dist/{benchmark-command-BVtaq_ve.js → benchmark-command-CF-4GEWZ.js} +9 -8
  10. package/dist/benchmark-command-CF-4GEWZ.js.map +1 -0
  11. package/dist/benchmarks/index.d.ts +1 -1
  12. package/dist/benchmarks/index.js +3 -3
  13. package/dist/builder-eval/index.d.ts.map +1 -1
  14. package/dist/builder-eval/index.js +22 -8
  15. package/dist/builder-eval/index.js.map +1 -1
  16. package/dist/campaign/index.js +8 -8
  17. package/dist/{campaign-BSmOwskD.js → campaign-DQZmc2Dq.js} +12 -12
  18. package/dist/{campaign-BSmOwskD.js.map → campaign-DQZmc2Dq.js.map} +1 -1
  19. package/dist/{chat-json-call-6g5sJobJ.js → chat-json-call-5Jxna-aV.js} +2 -2
  20. package/dist/{chat-json-call-6g5sJobJ.js.map → chat-json-call-5Jxna-aV.js.map} +1 -1
  21. package/dist/cli.js +3 -3
  22. package/dist/contract/index.js +7 -7
  23. package/dist/{counterfactual-D_VWavVm.js → counterfactual-Bjq1mlUu.js} +2 -2
  24. package/dist/{counterfactual-D_VWavVm.js.map → counterfactual-Bjq1mlUu.js.map} +1 -1
  25. package/dist/{define-agent-eval-h-s-sI-v.js → define-agent-eval-D08pWIJb.js} +13 -13
  26. package/dist/{define-agent-eval-h-s-sI-v.js.map → define-agent-eval-D08pWIJb.js.map} +1 -1
  27. package/dist/{descriptive-jDOuI6mz.js → descriptive-1V17A-qa.js} +2 -2
  28. package/dist/{descriptive-jDOuI6mz.js.map → descriptive-1V17A-qa.js.map} +1 -1
  29. package/dist/{dspy-rlm-engine-DptEII26.js → dspy-rlm-engine-DhA9qKIm.js} +2 -2
  30. package/dist/{dspy-rlm-engine-DptEII26.js.map → dspy-rlm-engine-DhA9qKIm.js.map} +1 -1
  31. package/dist/emitter-D_jYSGRd.d.ts.map +1 -1
  32. package/dist/{emitter-BpYFQPj4.js → emitter-DeQHiDMm.js} +13 -6
  33. package/dist/emitter-DeQHiDMm.js.map +1 -0
  34. package/dist/{eval-campaign-BsXWL2-2.js → eval-campaign-BfohKmzx.js} +3 -3
  35. package/dist/{eval-campaign-BsXWL2-2.js.map → eval-campaign-BfohKmzx.js.map} +1 -1
  36. package/dist/experiment/index.js +8 -8
  37. package/dist/{experiment-tracker-Ym6rEQT1.js → experiment-tracker-BKEumQug.js} +2 -2
  38. package/dist/{experiment-tracker-Ym6rEQT1.js.map → experiment-tracker-BKEumQug.js.map} +1 -1
  39. package/dist/{external-optimizer-process-WosTBChy.js → external-optimizer-process-BFmh36vW.js} +2 -2
  40. package/dist/{external-optimizer-process-WosTBChy.js.map → external-optimizer-process-BFmh36vW.js.map} +1 -1
  41. package/dist/{external-optimizer-subprocess-BIWbHpgD.js → external-optimizer-subprocess-CqLMW3nh.js} +2 -2
  42. package/dist/{external-optimizer-subprocess-BIWbHpgD.js.map → external-optimizer-subprocess-CqLMW3nh.js.map} +1 -1
  43. package/dist/fuzz.d.ts +1 -2
  44. package/dist/fuzz.d.ts.map +1 -1
  45. package/dist/fuzz.js +4 -10
  46. package/dist/fuzz.js.map +1 -1
  47. package/dist/index.d.ts +3 -3
  48. package/dist/index.js +26 -26
  49. package/dist/index.js.map +1 -1
  50. package/dist/{internal-BDHPCnjk.js → internal-BMFSR8Ns.js} +3 -19
  51. package/dist/internal-BMFSR8Ns.js.map +1 -0
  52. package/dist/{judge-calibration-zZjLz8hr.js → judge-calibration-BnpVKtnb.js} +3 -13
  53. package/dist/judge-calibration-BnpVKtnb.js.map +1 -0
  54. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -1
  55. package/dist/{llm-client-hgDieDNN.js → llm-client-BFMRpmqb.js} +8 -11
  56. package/dist/llm-client-BFMRpmqb.js.map +1 -0
  57. package/dist/{llm-judge-BhasIPFT.js → llm-judge-Du7WQPh7.js} +7 -7
  58. package/dist/{llm-judge-BhasIPFT.js.map → llm-judge-Du7WQPh7.js.map} +1 -1
  59. package/dist/meta-eval/index.d.ts +2 -1
  60. package/dist/meta-eval/index.d.ts.map +1 -1
  61. package/dist/meta-eval/index.js +14 -22
  62. package/dist/meta-eval/index.js.map +1 -1
  63. package/dist/openapi.json +1 -1
  64. package/dist/{paired-arms-D-XRF_fy.js → paired-arms-D4aeIHUy.js} +3 -3
  65. package/dist/{paired-arms-D-XRF_fy.js.map → paired-arms-D4aeIHUy.js.map} +1 -1
  66. package/dist/{paired-tests-BHIhYVdu.js → paired-tests-C8iCsioC.js} +3 -3
  67. package/dist/{paired-tests-BHIhYVdu.js.map → paired-tests-C8iCsioC.js.map} +1 -1
  68. package/dist/pipelines/index.d.ts +3 -2
  69. package/dist/pipelines/index.d.ts.map +1 -1
  70. package/dist/pipelines/index.js +4 -19
  71. package/dist/pipelines/index.js.map +1 -1
  72. package/dist/{power-and-mde-CHIrXJll.js → power-and-mde-B8F2RdcD.js} +3 -3
  73. package/dist/{power-and-mde-CHIrXJll.js.map → power-and-mde-B8F2RdcD.js.map} +1 -1
  74. package/dist/{power-preflight-DEw-uC7q.js → power-preflight-CFXm0Vjo.js} +3 -3
  75. package/dist/{power-preflight-DEw-uC7q.js.map → power-preflight-CFXm0Vjo.js.map} +1 -1
  76. package/dist/{produced-state-DZ89riy5.js → produced-state-Be0BK3RN.js} +3 -3
  77. package/dist/{produced-state-DZ89riy5.js.map → produced-state-Be0BK3RN.js.map} +1 -1
  78. package/dist/{promotion-policy-xzA40Evo.js → promotion-policy-LY9mVQ7W.js} +3 -3
  79. package/dist/{promotion-policy-xzA40Evo.js.map → promotion-policy-LY9mVQ7W.js.map} +1 -1
  80. package/dist/{query-DxPYqpmT.d.ts → query-CwnHlu5p.d.ts} +19 -2
  81. package/dist/query-CwnHlu5p.d.ts.map +1 -0
  82. package/dist/{query-CHmMP42p.js → query-_5g6re3_.js} +44 -2
  83. package/dist/query-_5g6re3_.js.map +1 -0
  84. package/dist/random-Dn5fPWkt.js +21 -0
  85. package/dist/random-Dn5fPWkt.js.map +1 -0
  86. package/dist/record-id-DUgsK5qp.js +17 -0
  87. package/dist/record-id-DUgsK5qp.js.map +1 -0
  88. package/dist/{release-confidence-DKfD2RYU.js → release-confidence-nGDJiiwc.js} +2 -2
  89. package/dist/{release-confidence-DKfD2RYU.js.map → release-confidence-nGDJiiwc.js.map} +1 -1
  90. package/dist/reporting.js +6 -6
  91. package/dist/{reward-hacking-t4lB1yt8.js → reward-hacking-O5zKDANP.js} +2 -2
  92. package/dist/{reward-hacking-t4lB1yt8.js.map → reward-hacking-O5zKDANP.js.map} +1 -1
  93. package/dist/rl.d.ts.map +1 -1
  94. package/dist/rl.js +9 -15
  95. package/dist/rl.js.map +1 -1
  96. package/dist/{rubric-predictive-validity-CK8SCOg-.js → rubric-predictive-validity-2D5Gw9z9.js} +3 -3
  97. package/dist/{rubric-predictive-validity-CK8SCOg-.js.map → rubric-predictive-validity-2D5Gw9z9.js.map} +1 -1
  98. package/dist/{semantic-concept-judge-BSkKKHeq.js → semantic-concept-judge-BsDMOwJr.js} +2 -2
  99. package/dist/{semantic-concept-judge-BSkKKHeq.js.map → semantic-concept-judge-BsDMOwJr.js.map} +1 -1
  100. package/dist/{sequential-rYW-Ophm.js → sequential-BLMbdrD7.js} +3 -3
  101. package/dist/{sequential-rYW-Ophm.js.map → sequential-BLMbdrD7.js.map} +1 -1
  102. package/dist/{server-BtFd4uzB.js → server-BjYiJHoJ.js} +2 -2
  103. package/dist/{server-BtFd4uzB.js.map → server-BjYiJHoJ.js.map} +1 -1
  104. package/dist/{skillopt-optimization-method-DbaekMcn.js → skillopt-optimization-method-UArRo-nr.js} +6 -6
  105. package/dist/{skillopt-optimization-method-DbaekMcn.js.map → skillopt-optimization-method-UArRo-nr.js.map} +1 -1
  106. package/dist/{store-tool-spans-Dlh9vkFK.js → store-tool-spans-BVga3c37.js} +21 -17
  107. package/dist/store-tool-spans-BVga3c37.js.map +1 -0
  108. package/dist/store-tool-spans-DPUG7UUY.d.ts.map +1 -1
  109. package/dist/{student-t-CvBq2mve.js → student-t-BA-Uy51p.js} +2 -2
  110. package/dist/{student-t-CvBq2mve.js.map → student-t-BA-Uy51p.js.map} +1 -1
  111. package/dist/{summary-report-BI5hUtvK.js → summary-report-BXeQ5Ues.js} +6 -6
  112. package/dist/{summary-report-BI5hUtvK.js.map → summary-report-BXeQ5Ues.js.map} +1 -1
  113. package/dist/supervisor-run/index.js +1 -1
  114. package/dist/{tool-waste-BqzmVdJk.js → tool-waste-8BQiUc8K.js} +4 -4
  115. package/dist/{tool-waste-BqzmVdJk.js.map → tool-waste-8BQiUc8K.js.map} +1 -1
  116. package/dist/{tool-waste-Dro0gJi3.d.ts → tool-waste-CKc7bYIg.d.ts} +2 -2
  117. package/dist/{tool-waste-Dro0gJi3.d.ts.map → tool-waste-CKc7bYIg.d.ts.map} +1 -1
  118. package/dist/trace-repair/index.js +1 -1
  119. package/dist/traces.d.ts +2 -2
  120. package/dist/traces.d.ts.map +1 -1
  121. package/dist/traces.js +8 -17
  122. package/dist/traces.js.map +1 -1
  123. package/dist/trajectory-replay/index.d.ts.map +1 -1
  124. package/dist/trajectory-replay/index.js +3 -13
  125. package/dist/trajectory-replay/index.js.map +1 -1
  126. package/dist/{types-BPb2Kf_C2.d.ts → types-BPb2Kf_C.d.ts} +1 -1
  127. package/dist/types-BPb2Kf_C.d.ts.map +1 -0
  128. package/dist/types-Bfk0uxRj.d.ts.map +1 -1
  129. package/dist/wire/index.js +1 -1
  130. package/package.json +2 -2
  131. package/dist/benchmark-command-BVtaq_ve.js.map +0 -1
  132. package/dist/emitter-BpYFQPj4.js.map +0 -1
  133. package/dist/internal-BDHPCnjk.js.map +0 -1
  134. package/dist/judge-calibration-zZjLz8hr.js.map +0 -1
  135. package/dist/llm-client-hgDieDNN.js.map +0 -1
  136. package/dist/query-CHmMP42p.js.map +0 -1
  137. package/dist/query-DxPYqpmT.d.ts.map +0 -1
  138. package/dist/store-tool-spans-Dlh9vkFK.js.map +0 -1
  139. package/dist/types-BPb2Kf_C2.d.ts.map +0 -1
@@ -1 +1 @@
1
- {"version":3,"file":"produced-state-DZ89riy5.js","names":[],"sources":["../src/agent-profile.ts","../src/completion-verifier.ts","../src/produced-state.ts"],"sourcesContent":["import { createHash } from 'node:crypto'\nimport type { AgentProfile } from '@tangle-network/agent-interface'\nimport { type HarnessType, harnessSupportsModel } from '@tangle-network/agent-interface'\nimport { ValidationError } from './errors'\nimport { canonicalString } from './ledger-core/canonical'\n\nexport type { AgentProfile, HarnessType } from '@tangle-network/agent-interface'\n\n/**\n * The agentic coding harnesses an eval sweeps by default — the ones we care about\n * ranking. This is the SINGLE source of that list; consumers import it instead of\n * re-declaring their own (a re-declared list is how the fleet drifts). Pass an\n * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known\n * harness) to widen beyond these.\n */\nexport const CODING_HARNESSES: readonly HarnessType[] = [\n 'opencode',\n 'claude-code',\n 'codex',\n 'kimi-code',\n]\n\nexport interface ProfileAxisSpec {\n /** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the\n * harness and model vary. `model.default` is the fallback model. */\n base: AgentProfile\n /** Harnesses to cross. Default: {@link CODING_HARNESSES}. */\n harnesses?: readonly HarnessType[]\n /** Models to cross. Default: `[base.model.default]` — one model, i.e. today's\n * single-model behaviour, so omitting this never changes an existing run. */\n models?: readonly string[]\n /** Force every (harness, model) pair verbatim, even ones the harness can't run —\n * for deliberately testing failure modes. Default (false): SNAP instead — a\n * vendor-locked harness runs only the swept models in its family, or its native\n * default when it supports none, so no harness is dropped and none gets a\n * guaranteed-failing foreign-model cell. */\n keepIncompatible?: boolean\n}\n\n/** Model sentinel for a vendor-locked harness that supports none of the swept models:\n * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness\n * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi\n * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship\n * table that would rot as router catalogs change. */\nexport const HARNESS_NATIVE_MODEL = 'default'\n\n/**\n * Expand a base profile across the harness × model matrix into the `AgentProfile[]`\n * that `runProfileMatrix` / `selfImprove` score — the ONE place \"which harnesses ×\n * which models do we evaluate\" lives, so no product hand-rolls its own harness list\n * or column→profile mapping (the pattern that let those copies drift and silently\n * break the harness pivot).\n *\n * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,\n * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both\n * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row\n * and results join back by harness/model via {@link harnessAxisOf} with no\n * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or\n * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every\n * requested harness runs; `keepIncompatible` forces every pair verbatim.\n *\n * Omit `harnesses`/`models` to sweep the full default set — the \"turn it on for\n * everything we care about\" switch, identical in shape whether one harness or all.\n */\nexport function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[] {\n const harnesses = spec.harnesses ?? CODING_HARNESSES\n if (harnesses.length === 0) throw new ValidationError('expandProfileAxes: no harnesses to sweep')\n const baseModel = spec.base.model?.default\n const models = spec.models ?? (baseModel ? [baseModel] : [])\n if (models.length === 0) {\n throw new ValidationError(\n 'expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied',\n )\n }\n const out: AgentProfile[] = []\n const seen = new Set<string>()\n for (const harness of harnesses) {\n // A universal (router-backed) harness — opencode/pi/claudish — runs every swept\n // model. A vendor-locked harness — codex/claude-code/kimi-code — runs only the\n // swept models in its own family; when it supports NONE of them it snaps to its\n // native default (the `HARNESS_NATIVE_MODEL` sentinel it resolves at runtime)\n // rather than being dropped, so every requested harness still appears in the\n // sweep on a model it can actually run — e.g. sweeping `deepseek/x` puts opencode\n // on deepseek and kimi-code on its own Kimi model, a real head-to-head.\n // `keepIncompatible` forces every (harness, model) pair verbatim (failure-mode runs).\n const supported = spec.keepIncompatible\n ? models\n : models.filter((model) => harnessSupportsModel(harness, model))\n const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL]\n for (const model of effective) {\n const profile: AgentProfile = {\n ...spec.base,\n harness,\n name: `${spec.base.name ?? 'agent'}/${harness}/${model}`,\n model: { ...spec.base.model, default: model },\n metadata: { ...(spec.base.metadata ?? {}), harness, harnessModel: model },\n }\n const id = agentProfileId(profile)\n if (seen.has(id)) continue\n seen.add(id)\n out.push(profile)\n }\n }\n if (out.length === 0) {\n // Unreachable in normal use — snapping guarantees ≥1 cell per harness — but keep a\n // fail-closed guard so a future refactor can't silently produce an empty sweep.\n throw new ValidationError(\n `expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(', ')}], models=[${models.join(', ')}]).`,\n )\n }\n return out\n}\n\n/**\n * Read the (harness, model) a matrix cell ran under, off a profile or a result row's\n * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile\n * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by\n * this instead of recomputing an id (recomputing the wrong key is what broke the pivot\n * in the hand-rolled copies).\n */\nexport function harnessAxisOf(\n profile: Pick<AgentProfile, 'metadata'>,\n): { harness: HarnessType; model: string } | undefined {\n const m = profile.metadata as Record<string, unknown> | undefined\n const harness = m?.harness\n const model = m?.harnessModel\n if (typeof harness === 'string' && typeof model === 'string') {\n return { harness: harness as HarnessType, model }\n }\n return undefined\n}\n\n/**\n * Collision-resistant, path-safe, human-readable profile id for eval artifacts.\n * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix\n * keys, and directory names where two profiles must not collapse onto one row.\n * The suffix is the first 64 bits of the behaviour hash, enough for ordinary\n * eval matrices while keeping filenames readable.\n */\nexport function agentProfileId(profile: AgentProfile): string {\n const label = pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? 'profile'\n return `${label}-${agentProfileHash(profile).slice(0, 16)}`\n}\n\n/**\n * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete\n * model id because run records reject bare/missing model aliases.\n */\nexport function agentProfileModelId(profile: AgentProfile): string {\n const model = profile.model?.default?.trim()\n if (!model) {\n const label = agentProfileDisplayLabel(profile) ?? 'unnamed profile'\n throw new ValidationError(\n `AgentProfile \"${label}\" has no model.default — cannot record eval run`,\n )\n }\n return model\n}\n\nfunction agentProfileDisplayLabel(profile: AgentProfile): string | undefined {\n return profile.name?.trim() || profile.version?.trim() || undefined\n}\n\nfunction pathSafeProfileLabel(label: string | undefined): string | undefined {\n const safe = label\n ?.trim()\n .replace(/[^A-Za-z0-9._-]+/g, '-')\n .replace(/-+/g, '-')\n .replace(/^-|-$/g, '')\n return safe || undefined\n}\n\nfunction compact<T extends Record<string, unknown>>(input: T): Partial<T> {\n const out: Record<string, unknown> = {}\n for (const [key, value] of Object.entries(input)) {\n if (value !== undefined) out[key] = value\n }\n return out as Partial<T>\n}\n\n/**\n * Deterministic behaviour identity for the canonical\n * `@tangle-network/agent-interface` AgentProfile.\n *\n * `name` and `description` are labels and do not affect the hash. Profile\n * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,\n * and extensions do affect the hash. Resource array order is hash-bearing\n * because mount order can change agent behaviour. Undefined fields are treated\n * as absent; explicit `null` fields remain hash-bearing.\n */\nexport function agentProfileHash(profile: AgentProfile): string {\n const model = agentProfileModelId(profile)\n const behaviour = compact({\n ...profile,\n name: undefined,\n description: undefined,\n tags: profile.tags ? [...profile.tags].sort() : undefined,\n model: compact({ ...profile.model, default: model }),\n })\n return createHash('sha256').update(canonicalString(behaviour)).digest('hex')\n}\n","/**\n * Completion verifier — the task-completion oracle.\n *\n * Answers the only eval question that is not a proxy: did the agent actually\n * COMPLETE the task — produce every required deliverable, persisted and\n * correct — rather than describe what should be done. A fluent transcript\n * that never produces the artifact scores zero here.\n *\n * Per requirement, a two-stage check:\n * 1. Structural — a produced item (vault artifact / approved proposal /\n * tool call) of the right kind is matched against the requirement and\n * carries non-empty content. Deterministic; no LLM.\n * 2. Correctness — only if structurally present AND the matched item\n * carries content, one targeted check decides whether that item\n * actually fulfils the requirement. A hallucinated artifact fails here;\n * an absent one already failed stage 1.\n *\n * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —\n * checker failures — are excluded from the denominator, never scored as\n * zeros). Quality dimensions are meaningless on an incomplete task — callers\n * gate on `fullyComplete` / `completionRate` before scoring quality.\n */\n\nimport { randomUUID } from 'node:crypto'\nimport type { ChatClient, ChatRequest } from './analyst/chat-client'\nimport type { Artifact } from './artifact-validator'\nimport { CostLedger, type CostLedgerHandle } from './cost-ledger'\nimport { assertServedModel, ModelSubstitutionError } from './integrity/served-model'\nimport { recoverTruncatedJson } from './json-recovery'\nimport { JudgeParseError } from './judges'\nimport {\n costReceiptFromLlm,\n costReceiptFromLlmError,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport { packageVersion } from './package-version'\nimport type { RawProviderEvent, RawProviderSink } from './trace/raw-provider-sink'\nimport {\n certificationEvidenceDigest,\n type DefaultVerdict,\n type VerdictCertification,\n} from './verdict'\nimport type { CheckerIdentity, VerificationStrategySource } from './verification-strategy'\n\n/** What kind of produced state can satisfy a requirement structurally. */\nexport type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any'\n\nexport interface CompletionRequirement {\n /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */\n reqId: string\n /** Human-readable description of the required deliverable. */\n title: string\n /** Optional kind/category hint, matched against a produced item's kind. */\n category?: string\n /** What produced state satisfies this requirement. Defaults to 'any'. */\n satisfiedBy?: SatisfiedBy\n}\n\nexport interface TaskGold {\n taskId: string\n requirements: CompletionRequirement[]\n}\n\nexport interface ProducedProposal {\n id: string\n title: string\n status: 'pending' | 'approved' | 'rejected'\n /** Optional persisted body — when present, enables a correctness check. */\n content?: string\n}\n\n/** Everything observable about what a run actually produced. */\nexport interface ProducedState {\n /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */\n artifacts: Artifact[]\n /** Proposals / filings the agent created. */\n proposals: ProducedProposal[]\n /** Names of tools the agent invoked. */\n toolCalls: string[]\n}\n\nexport interface RequirementCheck {\n reqId: string\n title: string\n /** A produced item of the right kind matched the requirement, non-empty. */\n structurallyPresent: boolean\n /**\n * Whether the matched item actually fulfils the requirement. `null` when\n * not structurally present, when the matched item carries no content\n * to assess, or when the correctness check itself failed (`unmeasured`).\n */\n correct: boolean | null\n /** structurallyPresent && !unmeasured && correct !== false. */\n satisfied: boolean\n /**\n * Set when the correctness check itself errored (LLM call failure or an\n * unparseable response after retry). The requirement's fulfilment is\n * UNKNOWN — `correct` stays null, `satisfied` is false, and\n * `completionVerdict` excludes the row from `completionRate`'s\n * denominator. Never folded into a zero: a synthetic zero is\n * indistinguishable from a real failure (see `JudgeParseError`).\n */\n unmeasured?: true\n /** Why the correctness check could not be measured (present iff `unmeasured`). */\n unmeasuredReason?: string\n /** Human-readable evidence for the verdict. */\n evidence: string[]\n}\n\n/** Extends the substrate verdict spine: `valid` = `fullyComplete` and\n * `score` = `completionRate` — derived in `completionVerdict()`, the one\n * place those equalities hold by construction. */\nexport interface CompletionVerdict extends DefaultVerdict {\n taskId: string\n requirements: RequirementCheck[]\n /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */\n completionRate: number\n /** Every measurable requirement satisfied (false when anything is unmeasured). */\n fullyComplete: boolean\n /** Requirements whose correctness check errored — reported, never scored as zero. */\n unmeasuredCount: number\n}\n\n/**\n * Construct a `CompletionVerdict` from the per-requirement checks, deriving\n * `completionRate` / `fullyComplete` and the spine fields (`valid` =\n * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero\n * requirements — a verdict over nothing is a misconfiguration, mirroring\n * `verifyCompletion`'s gold-spec guard.\n */\nexport function completionVerdict(input: {\n taskId: string\n requirements: RequirementCheck[]\n /** What certified the correctness stage, when anything did. Omitted =\n * an uncertified verdict — the honest default for a bare checker. */\n certification?: VerdictCertification\n}): CompletionVerdict {\n if (input.requirements.length === 0) {\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`,\n )\n }\n const measurable = input.requirements.filter((r) => !r.unmeasured)\n const unmeasuredCount = input.requirements.length - measurable.length\n if (measurable.length === 0) {\n // Every check errored: this is an infrastructure failure, not a scored\n // run. A 0-rate verdict here would be a fabricated measurement.\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? 'unknown reason'})`,\n )\n }\n const satisfiedCount = measurable.filter((r) => r.satisfied).length\n const completionRate = satisfiedCount / measurable.length\n // A run with unmeasured rows can still report a rate over what WAS\n // measured, but must not claim full completion.\n const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length\n return {\n taskId: input.taskId,\n requirements: input.requirements,\n completionRate,\n fullyComplete,\n unmeasuredCount,\n valid: fullyComplete,\n score: completionRate,\n ...(input.certification === undefined ? {} : { certification: input.certification }),\n }\n}\n\n/**\n * What a correctness checker declares about itself so the completion\n * verdict can carry a certification: which strategy member it discharges,\n * its exact identity, and the steps its answers rest on unverified.\n */\nexport interface CorrectnessCheckerAttestation {\n strategy: VerificationStrategySource\n checker: CheckerIdentity\n assumptions: string[]\n}\n\n/**\n * Decides whether a produced item's content actually fulfils a requirement.\n * Injected so the structural verifier stays pure and unit-testable; the\n * production implementation is `createLlmCorrectnessChecker`.\n *\n * `attestation` is optional metadata on the function value: a checker that\n * carries one yields certified completion verdicts; a bare function yields\n * the same verdict uncertified. A plain arrow function remains a valid\n * checker.\n */\nexport interface CorrectnessChecker {\n (\n requirement: CompletionRequirement,\n content: string,\n ): Promise<{ correct: boolean; reason: string }>\n attestation?: CorrectnessCheckerAttestation\n}\n\nconst STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'of',\n 'for',\n 'and',\n 'or',\n 'to',\n 'in',\n 'on',\n 'with',\n 'by',\n])\n\n// Deliverable-FORM vocabulary — words that name the SHAPE of an output, not its\n// domain content. A correct \"swap-comparison view persisted as a ui/** artifact\"\n// is an OpenUI JSON whose body says nothing about \"artifact\" / \"persisted\" /\n// \"view\"; the discriminative tokens are the domain nouns (swap, comparison).\n// Stripped from the REQUIREMENT side of structural recall so a deliverable is\n// matched on what it IS about, not on the boilerplate describing its form. The\n// correctness checker strips the same class via TITLE_STOPWORDS. Anti-game holds:\n// the distinctive domain tokens remain, so an off-topic item still fails.\nconst REQUIREMENT_FORM_STOPWORDS = new Set([\n 'generated',\n 'generate',\n 'view',\n 'render',\n 'rendered',\n 'persisted',\n 'persist',\n 'artifact',\n 'file',\n 'document',\n 'note',\n 'proposal',\n 'deliverable',\n 'output',\n 'created',\n 'create',\n 'produce',\n 'produced',\n 'flag',\n])\n\nconst MATCH_THRESHOLD = 0.5\nconst MIN_CONTENT_CHARS = 50\n\nfunction tokens(s: string, extraStop?: Set<string>): Set<string> {\n return new Set(\n s\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)),\n )\n}\n\n/**\n * Recall of the requirement's tokens within a candidate's identifying text.\n * Recall, not Jaccard — a candidate's path/id legitimately carries extra\n * tokens the requirement does not name. The requirement side drops\n * deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.\n */\nfunction tokenRecall(requirementText: string, candidateText: string): number {\n const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS)\n if (req.size === 0) return 0\n const cand = tokens(candidateText)\n let hit = 0\n for (const t of req) if (cand.has(t)) hit++\n return hit / req.size\n}\n\ninterface Candidate {\n reqIndex: number\n /** Unique key for a produced item — each item satisfies at most one requirement. */\n itemKey: string\n score: number\n evidence: string\n /** Content to correctness-check, or null when the matched item has none. */\n content: string | null\n}\n\nfunction artifactCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n artifacts: Artifact[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n artifacts.forEach((a, i) => {\n if ((a.content ?? '').trim().length < MIN_CONTENT_CHARS) return\n // Match against the artifact CONTENT too, not just its path + kind — a\n // generated view / note whose path is generic still satisfies a requirement\n // when its body covers it (e.g. an OpenUI comparison grounded in the on-file\n // figures). Bounded slice keeps the recall text cheap; MATCH_THRESHOLD holds.\n let score = tokenRecall(\n reqText,\n `${a.path ?? ''} ${a.kind} ${(a.content ?? '').slice(0, 4000)}`,\n )\n if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) {\n score = Math.max(score, 1)\n }\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `artifact:${i}`,\n score,\n evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,\n content: a.content ?? null,\n })\n })\n return out\n}\n\nfunction proposalCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n proposals: ProducedProposal[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n for (const p of proposals) {\n // Pending or rejected work is not a completed deliverable.\n if (p.status !== 'approved') continue\n // A proposal needs an assessable BODY to be a deliverable. A bare title is\n // not completion: correctness cannot be judged on it, so a title-only match\n // would auto-pass the oracle (structurallyPresent && correct===null →\n // satisfied) with no verifiable content. Tool calls are the only\n // legitimately content-less deliverable (`toolCallCandidates`).\n const body = (p.content ?? '').trim()\n if (body.length < MIN_CONTENT_CHARS) continue\n // Match against the body as well as the (often short) title — a refusal /\n // flag / analysis proposal whose title is a label still satisfies a\n // descriptively-worded requirement when its content covers it. MATCH_THRESHOLD\n // + the requirement's distinctive tokens keep an off-topic proposal out;\n // correctness (a SEMANTIC checker, NOT this lexical pass) then judges\n // polarity/fulfilment, so a negation that merely contains the tokens fails.\n // Structural and correctness must use different evidence or the two-stage\n // check collapses to one lexical gate.\n const score = tokenRecall(reqText, `${p.title} ${body}`)\n if (score < MATCH_THRESHOLD) continue\n out.push({\n reqIndex,\n itemKey: `proposal:${p.id}`,\n score,\n evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,\n content: body,\n })\n }\n return out\n}\n\nfunction toolCallCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n toolCalls: string[],\n): Candidate[] {\n const out: Candidate[] = []\n toolCalls.forEach((name, i) => {\n const score = tokenRecall(req.title, name)\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `tool:${i}`,\n score,\n evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,\n content: null,\n })\n })\n return out\n}\n\n/**\n * Verify whether a run completed the task. `checkCorrectness` is injected —\n * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.\n *\n * Throws on a gold spec with no requirements: an eval task that requires\n * nothing is a misconfiguration, not a vacuously-complete task.\n */\nexport async function verifyCompletion(\n gold: TaskGold,\n state: ProducedState,\n checkCorrectness: CorrectnessChecker,\n): Promise<CompletionVerdict> {\n if (gold.requirements.length === 0) {\n throw new Error(\n `verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`,\n )\n }\n\n // Collect every above-threshold (requirement, produced-item) candidate, then\n // assign greedily by descending score: each requirement and each produced\n // item is used at most once. One deliverable fulfils one requirement.\n const candidates: Candidate[] = []\n gold.requirements.forEach((req, i) => {\n const by = req.satisfiedBy ?? 'any'\n if (by === 'artifact' || by === 'any') {\n candidates.push(...artifactCandidates(req, i, state.artifacts))\n }\n if (by === 'proposal' || by === 'any') {\n candidates.push(...proposalCandidates(req, i, state.proposals))\n }\n if (by === 'tool-call' || by === 'any') {\n candidates.push(...toolCallCandidates(req, i, state.toolCalls))\n }\n })\n candidates.sort((a, b) => b.score - a.score)\n\n const assigned = new Map<number, Candidate>()\n const itemTaken = new Set<string>()\n for (const c of candidates) {\n if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue\n assigned.set(c.reqIndex, c)\n itemTaken.add(c.itemKey)\n }\n\n const requirements: RequirementCheck[] = []\n for (let i = 0; i < gold.requirements.length; i++) {\n const req = gold.requirements[i]!\n const match = assigned.get(i)\n const evidence: string[] = []\n let correct: boolean | null = null\n let unmeasuredReason: string | undefined\n\n if (match) {\n evidence.push(match.evidence)\n if (match.content !== null) {\n try {\n const r = await checkCorrectness(req, match.content)\n correct = r.correct\n evidence.push(`correctness: ${r.correct ? 'pass' : 'fail'} — ${r.reason}`)\n } catch (err) {\n // The CHECKER failed, not the requirement. Recording this as a\n // zero would fabricate a model failure out of an infrastructure\n // one; the requirement is unmeasured and leaves the denominator.\n unmeasuredReason =\n err instanceof JudgeParseError\n ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}`\n : `checker call failed: ${err instanceof Error ? err.message : String(err)}`\n evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`)\n }\n } else {\n evidence.push('correctness: not assessed — matched item carries no content')\n }\n } else {\n const by = req.satisfiedBy ?? 'any'\n const kind = by === 'any' ? 'artifact/proposal/tool-call' : by\n evidence.push(`no produced ${kind} matched this requirement`)\n }\n\n const structurallyPresent = match !== undefined\n const unmeasured = unmeasuredReason !== undefined\n const satisfied = structurallyPresent && !unmeasured && correct !== false\n requirements.push({\n reqId: req.reqId,\n title: req.title,\n structurallyPresent,\n correct,\n satisfied,\n ...(unmeasured ? { unmeasured: true as const, unmeasuredReason } : {}),\n evidence,\n })\n }\n\n // The verdict blends the deterministic structural stage with the injected\n // correctness stage, so the certification names the checker's own member\n // and carries the structural stage's lexical nature as an assumption. A\n // checker without an attestation yields an uncertified verdict.\n const attestation = checkCorrectness.attestation\n const certification: VerdictCertification | undefined = attestation\n ? {\n strategy: attestation.strategy,\n checker: attestation.checker,\n assumptions: [\n 'structural matching is lexical token recall over produced items — the correctness stage only sees items it matched',\n ...attestation.assumptions,\n ],\n evidenceDigest: certificationEvidenceDigest({ taskId: gold.taskId, requirements }),\n }\n : undefined\n\n return completionVerdict({\n taskId: gold.taskId,\n requirements,\n ...(certification === undefined ? {} : { certification }),\n })\n}\n\nexport interface LlmCorrectnessCheckerOpts {\n model?: string\n /** Optional ledger for direct use. */\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /** Max chars of artifact content sent to the checker. */\n maxContentChars?: number\n /**\n * Checker LLM calls per requirement before giving up (parse failures and\n * call errors both consume attempts). The failure then surfaces as an\n * `unmeasured` requirement, never a zero.\n */\n maxAttempts?: number\n /**\n * Forensic capture of every checker request/response/error — without it a\n * checker failure is unauditable (the agent-turn raws never contain the\n * checker's own calls). Same sink contract as `LlmClient`.\n */\n rawSink?: RawProviderSink\n}\n\n/**\n * Parse the correctness checker's model response. Tolerates a response\n * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the\n * verdict boolean usually lands in the first few tokens, so a recovered\n * prefix with a boolean `correct` is a real measurement, not a guess.\n * Fails loud (JudgeParseError) when no boolean verdict is recoverable.\n */\nexport function parseCorrectnessResponse(raw: string): { correct: boolean; reason: string } {\n const readVerdict = (candidate: unknown): { correct: boolean; reason: string } | null => {\n if (candidate === null || typeof candidate !== 'object') return null\n const { correct, reason } = candidate as { correct?: unknown; reason?: unknown }\n if (typeof correct !== 'boolean') return null\n return { correct, reason: typeof reason === 'string' ? reason : '' }\n }\n\n const match = raw.match(/\\{[\\s\\S]*\\}/)\n if (match) {\n try {\n const strict = readVerdict(JSON.parse(match[0]))\n if (strict) return strict\n } catch {\n // fall through to truncation recovery\n }\n }\n // The strict path needs a closing `}`; a cap-hit response has none. Take\n // everything from the first `{` and auto-close it.\n const start = raw.indexOf('{')\n if (start !== -1) {\n const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)))\n if (recovered) return recovered\n }\n throw new JudgeParseError('correctness-checker', raw)\n}\n\n/**\n * Production `CorrectnessChecker` — one LLM call per matched artifact,\n * deterministic (temperature 0), structured JSON out. Judges fulfilment\n * only: a plan, a gesture, or a description of what should be done does not\n * fulfil a requirement — the artifact must BE the deliverable.\n */\nexport function createLlmCorrectnessChecker(\n chat: ChatClient,\n opts: LlmCorrectnessCheckerOpts = {},\n): CorrectnessChecker {\n const model = opts.model ?? 'claude-sonnet-4-6'\n const maxContentChars = opts.maxContentChars ?? 8000\n const maxAttempts = opts.maxAttempts ?? 2\n const costLedger = opts.costLedger ?? new CostLedger()\n const sink = opts.rawSink\n const record = async (event: RawProviderEvent): Promise<void> => {\n // Forensic capture is best-effort; the verdict is the system of record.\n try {\n await sink?.record(event)\n } catch {\n // Intentionally swallowed.\n }\n }\n const checker: CorrectnessChecker = async (requirement, content) => {\n const request = {\n model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}.',\n },\n {\n role: 'user' as const,\n content: `Requirement: ${requirement.title}\\n${\n requirement.category ? `Category: ${requirement.category}\\n` : ''\n }\\nProduced artifact:\\n${content.slice(0, maxContentChars)}`,\n },\n ],\n temperature: 0,\n maxTokens: 200,\n } satisfies ChatRequest\n let lastErr: unknown\n for (let attempt = 0; attempt < maxAttempts; attempt++) {\n const started = Date.now()\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'request',\n timestamp: started,\n requestBody: request,\n redactedFields: [],\n })\n try {\n const paid = await costLedger.runPaidCall({\n channel: 'verifier',\n phase: opts.costPhase ?? 'completion.correctness',\n actor: 'correctness-checker',\n model,\n maximumCharge:\n chat.maximumAttempts === undefined\n ? undefined\n : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),\n tags: {\n ...opts.costTags,\n requirementId: requirement.reqId,\n attempt: String(attempt),\n },\n signal: opts.signal,\n execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),\n receipt: costReceiptFromLlm,\n receiptFromError: costReceiptFromLlmError,\n })\n if (!paid.succeeded) throw paid.error\n const resp = paid.value\n // Hold the transport to its own word: a verdict produced by a model\n // other than the one requested is not this model's verdict. A\n // transport that echoes no id cannot be made to prove identity here —\n // callers needing that proof enable it at the client\n // (`LlmClientOptions.assertServedModel`) or gate on `assertModelsServed`.\n assertServedModel(model, resp.servedModel, {\n allowUnreported: true,\n context: `correctness checker for requirement ${requirement.reqId}`,\n })\n const raw = resp.content\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'response',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n responseBody: resp,\n redactedFields: [],\n })\n return parseCorrectnessResponse(raw)\n } catch (err) {\n lastErr = err\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n // Substitution is a verdict about identity, not a transient fault:\n // another attempt reaches the same wrong model and spends more tokens.\n if (err instanceof ModelSubstitutionError) throw err\n }\n }\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n }\n checker.attestation = {\n strategy: 'judge',\n checker: { name: 'llm-correctness-checker', version: model },\n assumptions: [\n 'served-model identity is accepted when the transport reports none (assertServedModel allowUnreported)',\n ],\n }\n return checker\n}\n\n/** Stopwords for requirement-title tokenization — drops the imperative verbs\n * ('review', 'update', …) common to deliverable titles so recall keys on the\n * substantive nouns, not the boilerplate ask. */\nconst TITLE_STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'and',\n 'or',\n 'for',\n 'to',\n 'of',\n 'in',\n 'on',\n 'with',\n 'review',\n 'update',\n 'new',\n 'proposed',\n])\n\n/**\n * Deterministic `CorrectnessChecker` — the no-LLM counterpart to\n * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its\n * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`\n * of the requirement title's significant tokens. No network.\n *\n * Polarity-blind: token recall credits a negation that contains the\n * requirement's tokens (\"I will NOT produce the comparison\" recalls every token\n * of \"produce the comparison\"). The structural match stage is ALSO lexical, so\n * pairing the two collapses to a single gameable gate. Use this only as an\n * opt-in structural pre-filter or for tasks whose requirements have no polarity\n * to invert; for produced-state grading the correctness checker MUST be semantic\n * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.\n */\nexport function createTokenRecallChecker(\n opts: { minRecall?: number; minContentLength?: number } = {},\n): CorrectnessChecker {\n const minRecall = opts.minRecall ?? 0.5\n const minLen = opts.minContentLength ?? 120\n const checker: CorrectnessChecker = async (requirement, content) => {\n const body = content.trim()\n if (body.length < minLen)\n return {\n correct: false,\n reason: `content too thin (${body.length} chars) to be the deliverable`,\n }\n const titleTokens = requirement.title\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t))\n if (titleTokens.length === 0)\n return {\n correct: true,\n reason: 'requirement title has no significant tokens — structural match accepted',\n }\n const lower = body.toLowerCase()\n const hits = titleTokens.filter((t) => lower.includes(t)).length\n const recall = hits / titleTokens.length\n return recall >= minRecall\n ? {\n correct: true,\n reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`,\n }\n : {\n correct: false,\n reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`,\n }\n }\n // 'schema': token recall checks shape, not meaning — the member whose\n // documented failure mode (\"a well-formed wrong answer passes\") is\n // exactly this checker's polarity blindness.\n checker.attestation = {\n strategy: 'schema',\n checker: { name: 'token-recall-checker', version: packageVersion() },\n assumptions: ['polarity-blind: a negation that recalls the requirement tokens passes'],\n }\n return checker\n}\n","/**\n * Produced-state extraction — normalize a run's runtime event stream into the\n * typed `ProducedState` the completion oracle consumes.\n *\n * `ProducedState` answers \"what did the agent actually produce\" — vault\n * artifacts, proposals, tool calls. The runtime emits these as a stream of\n * events; this module is the single normalization point from that stream to\n * the shape `verifyCompletion` expects.\n *\n * Input is structurally typed (`RuntimeEventLike`) so this module does not\n * depend on agent-runtime — agent-runtime's `RuntimeStreamEvent` satisfies it\n * structurally. The `content` on `ArtifactEventLike` and the whole\n * `proposal_created` variant are the runtime-side enrichments this contract\n * requires; the runtime emits them, this module consumes them.\n */\n\nimport type { Artifact } from './artifact-validator'\nimport type { ProducedProposal, ProducedState } from './completion-verifier'\n\n/** A tool the agent invoked. */\nexport interface ToolCallEventLike {\n type: 'tool_call'\n toolName: string\n}\n\n/**\n * An artifact the agent produced. `content` is the enriched field — the\n * runtime's base `artifact` event carries only metadata; the completion\n * oracle needs the body to verify the deliverable, so the runtime emits it.\n */\nexport interface ArtifactEventLike {\n type: 'artifact'\n artifactId: string\n name?: string\n mimeType?: string\n uri?: string\n content?: string\n}\n\n/** A proposal / filing the agent created. */\nexport interface ProposalEventLike {\n type: 'proposal_created'\n proposalId: string\n title: string\n status?: 'pending' | 'approved' | 'rejected'\n // body of the proposal (e.g. a submit_proposal `description`). When present,\n // the completion oracle correctness-checks it like artifact content; absent,\n // the proposal is graded presence-only.\n content?: string\n}\n\n/**\n * The subset of runtime stream events `extractProducedState` consumes.\n * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;\n * the `{ type: string }` catch-all keeps the input permissive so callers can\n * pass the whole unfiltered telemetry stream — unrecognized events are skipped.\n */\nexport type RuntimeEventLike =\n | ToolCallEventLike\n | ArtifactEventLike\n | ProposalEventLike\n | { type: string }\n\nfunction artifactKind(mimeType: string | undefined): string {\n if (!mimeType) return 'file'\n if (mimeType.includes('json')) return 'json'\n if (mimeType.startsWith('text/')) return 'text'\n return 'file'\n}\n\n/**\n * Normalize a run's runtime event stream into `ProducedState`.\n *\n * Pure and total — unrecognized event types are skipped. `toolCalls` is\n * deduplicated by name in first-seen order (completion cares about a tool's\n * presence, not its call count). An artifact with neither a name nor a uri\n * still yields an entry keyed by its `artifactId` so it is never silently\n * dropped; an artifact with no `content` yields empty content, which the\n * completion oracle's structural check then rejects on its own.\n */\nexport function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState {\n const artifacts: Artifact[] = []\n const proposals: ProducedProposal[] = []\n const toolCalls: string[] = []\n const seenTools = new Set<string>()\n\n for (const ev of events) {\n if (ev.type === 'tool_call') {\n const name = (ev as ToolCallEventLike).toolName\n if (name && !seenTools.has(name)) {\n seenTools.add(name)\n toolCalls.push(name)\n }\n } else if (ev.type === 'artifact') {\n const a = ev as ArtifactEventLike\n artifacts.push({\n kind: artifactKind(a.mimeType),\n path: a.name ?? a.uri ?? a.artifactId,\n content: a.content ?? '',\n })\n } else if (ev.type === 'proposal_created') {\n const p = ev as ProposalEventLike\n proposals.push({\n id: p.proposalId,\n title: p.title,\n status: p.status ?? 'pending',\n ...(p.content !== undefined ? { content: p.content } : {}),\n })\n }\n }\n\n return { artifacts, proposals, toolCalls }\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAa,mBAA2C;CACtD;CACA;CACA;CACA;AACF;;;;;;AAwBA,MAAa,uBAAuB;;;;;;;;;;;;;;;;;;;AAoBpC,SAAgB,kBAAkB,MAAuC;CACvE,MAAM,YAAY,KAAK,aAAa;CACpC,IAAI,UAAU,WAAW,GAAG,MAAM,IAAI,gBAAgB,0CAA0C;CAChG,MAAM,YAAY,KAAK,KAAK,OAAO;CACnC,MAAM,SAAS,KAAK,WAAW,YAAY,CAAC,SAAS,IAAI,CAAC;CAC1D,IAAI,OAAO,WAAW,GACpB,MAAM,IAAI,gBACR,kGACF;CAEF,MAAM,MAAsB,CAAC;CAC7B,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,WAAW,WAAW;EAS/B,MAAM,YAAY,KAAK,mBACnB,SACA,OAAO,QAAQ,UAAU,qBAAqB,SAAS,KAAK,CAAC;EACjE,MAAM,YAAY,UAAU,SAAS,IAAI,YAAY,CAAC,oBAAoB;EAC1E,KAAK,MAAM,SAAS,WAAW;GAC7B,MAAM,UAAwB;IAC5B,GAAG,KAAK;IACR;IACA,MAAM,GAAG,KAAK,KAAK,QAAQ,QAAQ,GAAG,QAAQ,GAAG;IACjD,OAAO;KAAE,GAAG,KAAK,KAAK;KAAO,SAAS;IAAM;IAC5C,UAAU;KAAE,GAAI,KAAK,KAAK,YAAY,CAAC;KAAI;KAAS,cAAc;IAAM;GAC1E;GACA,MAAM,KAAK,eAAe,OAAO;GACjC,IAAI,KAAK,IAAI,EAAE,GAAG;GAClB,KAAK,IAAI,EAAE;GACX,IAAI,KAAK,OAAO;EAClB;CACF;CACA,IAAI,IAAI,WAAW,GAGjB,MAAM,IAAI,gBACR,uDAAuD,UAAU,KAAK,IAAI,EAAE,aAAa,OAAO,KAAK,IAAI,EAAE,IAC7G;CAEF,OAAO;AACT;;;;;;;;AASA,SAAgB,cACd,SACqD;CACrD,MAAM,IAAI,QAAQ;CAClB,MAAM,UAAU,GAAG;CACnB,MAAM,QAAQ,GAAG;CACjB,IAAI,OAAO,YAAY,YAAY,OAAO,UAAU,UAClD,OAAO;EAAW;EAAwB;CAAM;AAGpD;;;;;;;;AASA,SAAgB,eAAe,SAA+B;CAE5D,OAAO,GADO,qBAAqB,yBAAyB,OAAO,CAAC,KAAK,UACzD,GAAG,iBAAiB,OAAO,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1D;;;;;AAMA,SAAgB,oBAAoB,SAA+B;CACjE,MAAM,QAAQ,QAAQ,OAAO,SAAS,KAAK;CAC3C,IAAI,CAAC,OAEH,MAAM,IAAI,gBACR,iBAFY,yBAAyB,OAAO,KAAK,kBAE1B,gDACzB;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,SAA2C;CAC3E,OAAO,QAAQ,MAAM,KAAK,KAAK,QAAQ,SAAS,KAAK,KAAK,KAAA;AAC5D;AAEA,SAAS,qBAAqB,OAA+C;CAM3E,OALa,OACT,KAAK,CAAC,CACP,QAAQ,qBAAqB,GAAG,CAAC,CACjC,QAAQ,OAAO,GAAG,CAAC,CACnB,QAAQ,UAAU,EAAE,KACR,KAAA;AACjB;AAEA,SAAS,QAA2C,OAAsB;CACxE,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,KAAK,GAC7C,IAAI,UAAU,KAAA,GAAW,IAAI,OAAO;CAEtC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,iBAAiB,SAA+B;CAC9D,MAAM,QAAQ,oBAAoB,OAAO;CACzC,MAAM,YAAY,QAAQ;EACxB,GAAG;EACH,MAAM,KAAA;EACN,aAAa,KAAA;EACb,MAAM,QAAQ,OAAO,CAAC,GAAG,QAAQ,IAAI,CAAC,CAAC,KAAK,IAAI,KAAA;EAChD,OAAO,QAAQ;GAAE,GAAG,QAAQ;GAAO,SAAS;EAAM,CAAC;CACrD,CAAC;CACD,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,gBAAgB,SAAS,CAAC,CAAC,CAAC,OAAO,KAAK;AAC7E;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACtEA,SAAgB,kBAAkB,OAMZ;CACpB,IAAI,MAAM,aAAa,WAAW,GAChC,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,+DAC3C;CAEF,MAAM,aAAa,MAAM,aAAa,QAAQ,MAAM,CAAC,EAAE,UAAU;CACjE,MAAM,kBAAkB,MAAM,aAAa,SAAS,WAAW;CAC/D,IAAI,WAAW,WAAW,GAGxB,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,yCAAyC,MAAM,aAAa,OAAO,8BAA8B,MAAM,aAAa,EAAE,EAAE,oBAAoB,iBAAiB,EACxM;CAEF,MAAM,iBAAiB,WAAW,QAAQ,MAAM,EAAE,SAAS,CAAC,CAAC;CAC7D,MAAM,iBAAiB,iBAAiB,WAAW;CAGnD,MAAM,gBAAgB,oBAAoB,KAAK,mBAAmB,WAAW;CAC7E,OAAO;EACL,QAAQ,MAAM;EACd,cAAc,MAAM;EACpB;EACA;EACA;EACA,OAAO;EACP,OAAO;EACP,GAAI,MAAM,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,eAAe,MAAM,cAAc;CACpF;AACF;AA+BA,MAAM,4BAAY,IAAI,IAAI;CACxB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAUD,MAAM,6CAA6B,IAAI,IAAI;CACzC;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAM,kBAAkB;AACxB,MAAM,oBAAoB;AAE1B,SAAS,OAAO,GAAW,WAAsC;CAC/D,OAAO,IAAI,IACT,EACG,YAAY,CAAC,CACb,MAAM,YAAY,CAAC,CACnB,QAAQ,MAAM,EAAE,SAAS,KAAK,CAAC,UAAU,IAAI,CAAC,KAAK,CAAC,WAAW,IAAI,CAAC,CAAC,CAC1E;AACF;;;;;;;AAQA,SAAS,YAAY,iBAAyB,eAA+B;CAC3E,MAAM,MAAM,OAAO,iBAAiB,0BAA0B;CAC9D,IAAI,IAAI,SAAS,GAAG,OAAO;CAC3B,MAAM,OAAO,OAAO,aAAa;CACjC,IAAI,MAAM;CACV,KAAK,MAAM,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC,GAAG;CACtC,OAAO,MAAM,IAAI;AACnB;AAYA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,GAAG,MAAM;EAC1B,KAAK,EAAE,WAAW,GAAA,CAAI,KAAK,CAAC,CAAC,SAAS,mBAAmB;EAKzD,IAAI,QAAQ,YACV,SACA,GAAG,EAAE,QAAQ,GAAG,GAAG,EAAE,KAAK,IAAI,EAAE,WAAW,GAAA,CAAI,MAAM,GAAG,GAAI,GAC9D;EACA,IAAI,IAAI,YAAY,EAAE,QAAQ,IAAI,SAAS,YAAY,MAAM,EAAE,KAAK,YAAY,GAC9E,QAAQ,KAAK,IAAI,OAAO,CAAC;EAE3B,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY;GACrB;GACA,UAAU,aAAa,EAAE,QAAQ,EAAE,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,CAAC;CACD,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,KAAK,MAAM,KAAK,WAAW;EAEzB,IAAI,EAAE,WAAW,YAAY;EAM7B,MAAM,QAAQ,EAAE,WAAW,GAAA,CAAI,KAAK;EACpC,IAAI,KAAK,SAAS,mBAAmB;EASrC,MAAM,QAAQ,YAAY,SAAS,GAAG,EAAE,MAAM,GAAG,MAAM;EACvD,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY,EAAE;GACvB;GACA,UAAU,sBAAsB,EAAE,MAAM,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS;EACX,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,MAAM,MAAM;EAC7B,MAAM,QAAQ,YAAY,IAAI,OAAO,IAAI;EACzC,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,QAAQ;GACjB;GACA,UAAU,cAAc,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACxE,SAAS;EACX,CAAC;CACH,CAAC;CACD,OAAO;AACT;;;;;;;;AASA,eAAsB,iBACpB,MACA,OACA,kBAC4B;CAC5B,IAAI,KAAK,aAAa,WAAW,GAC/B,MAAM,IAAI,MACR,2BAA2B,KAAK,OAAO,4CACzC;CAMF,MAAM,aAA0B,CAAC;CACjC,KAAK,aAAa,SAAS,KAAK,MAAM;EACpC,MAAM,KAAK,IAAI,eAAe;EAC9B,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,eAAe,OAAO,OAC/B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;CAElE,CAAC;CACD,WAAW,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;CAE3C,MAAM,2BAAW,IAAI,IAAuB;CAC5C,MAAM,4BAAY,IAAI,IAAY;CAClC,KAAK,MAAM,KAAK,YAAY;EAC1B,IAAI,SAAS,IAAI,EAAE,QAAQ,KAAK,UAAU,IAAI,EAAE,OAAO,GAAG;EAC1D,SAAS,IAAI,EAAE,UAAU,CAAC;EAC1B,UAAU,IAAI,EAAE,OAAO;CACzB;CAEA,MAAM,eAAmC,CAAC;CAC1C,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,aAAa,QAAQ,KAAK;EACjD,MAAM,MAAM,KAAK,aAAa;EAC9B,MAAM,QAAQ,SAAS,IAAI,CAAC;EAC5B,MAAM,WAAqB,CAAC;EAC5B,IAAI,UAA0B;EAC9B,IAAI;EAEJ,IAAI,OAAO;GACT,SAAS,KAAK,MAAM,QAAQ;GAC5B,IAAI,MAAM,YAAY,MACpB,IAAI;IACF,MAAM,IAAI,MAAM,iBAAiB,KAAK,MAAM,OAAO;IACnD,UAAU,EAAE;IACZ,SAAS,KAAK,gBAAgB,EAAE,UAAU,SAAS,OAAO,KAAK,EAAE,QAAQ;GAC3E,SAAS,KAAK;IAIZ,mBACE,eAAe,kBACX,6CAA6C,IAAI,IAAI,MAAM,GAAG,GAAG,MACjE,wBAAwB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAC7E,SAAS,KAAK,6BAA6B,kBAAkB;GAC/D;QAEA,SAAS,KAAK,6DAA6D;EAE/E,OAAO;GACL,MAAM,KAAK,IAAI,eAAe;GAC9B,MAAM,OAAO,OAAO,QAAQ,gCAAgC;GAC5D,SAAS,KAAK,eAAe,KAAK,0BAA0B;EAC9D;EAEA,MAAM,sBAAsB,UAAU,KAAA;EACtC,MAAM,aAAa,qBAAqB,KAAA;EACxC,MAAM,YAAY,uBAAuB,CAAC,cAAc,YAAY;EACpE,aAAa,KAAK;GAChB,OAAO,IAAI;GACX,OAAO,IAAI;GACX;GACA;GACA;GACA,GAAI,aAAa;IAAE,YAAY;IAAe;GAAiB,IAAI,CAAC;GACpE;EACF,CAAC;CACH;CAMA,MAAM,cAAc,iBAAiB;CACrC,MAAM,gBAAkD,cACpD;EACE,UAAU,YAAY;EACtB,SAAS,YAAY;EACrB,aAAa,CACX,sHACA,GAAG,YAAY,WACjB;EACA,gBAAgB,4BAA4B;GAAE,QAAQ,KAAK;GAAQ;EAAa,CAAC;CACnF,IACA,KAAA;CAEJ,OAAO,kBAAkB;EACvB,QAAQ,KAAK;EACb;EACA,GAAI,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,cAAc;CACzD,CAAC;AACH;;;;;;;;AAgCA,SAAgB,yBAAyB,KAAmD;CAC1F,MAAM,eAAe,cAAoE;EACvF,IAAI,cAAc,QAAQ,OAAO,cAAc,UAAU,OAAO;EAChE,MAAM,EAAE,SAAS,WAAW;EAC5B,IAAI,OAAO,YAAY,WAAW,OAAO;EACzC,OAAO;GAAE;GAAS,QAAQ,OAAO,WAAW,WAAW,SAAS;EAAG;CACrE;CAEA,MAAM,QAAQ,IAAI,MAAM,aAAa;CACrC,IAAI,OACF,IAAI;EACF,MAAM,SAAS,YAAY,KAAK,MAAM,MAAM,EAAE,CAAC;EAC/C,IAAI,QAAQ,OAAO;CACrB,QAAQ,CAER;CAIF,MAAM,QAAQ,IAAI,QAAQ,GAAG;CAC7B,IAAI,UAAU,IAAI;EAChB,MAAM,YAAY,YAAY,qBAAqB,IAAI,MAAM,KAAK,CAAC,CAAC;EACpE,IAAI,WAAW,OAAO;CACxB;CACA,MAAM,IAAI,gBAAgB,uBAAuB,GAAG;AACtD;;;;;;;AAQA,SAAgB,4BACd,MACA,OAAkC,CAAC,GACf;CACpB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,cAAc,KAAK,eAAe;CACxC,MAAM,aAAa,KAAK,cAAc,IAAI,WAAW;CACrD,MAAM,OAAO,KAAK;CAClB,MAAM,SAAS,OAAO,UAA2C;EAE/D,IAAI;GACF,MAAM,MAAM,OAAO,KAAK;EAC1B,QAAQ,CAER;CACF;CACA,MAAM,UAA8B,OAAO,aAAa,YAAY;EAClE,MAAM,UAAU;GACd;GACA,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IACE,MAAM;IACN,SAAS,gBAAgB,YAAY,MAAM,IACzC,YAAY,WAAW,aAAa,YAAY,SAAS,MAAM,GAChE,wBAAwB,QAAQ,MAAM,GAAG,eAAe;GAC3D,CACF;GACA,aAAa;GACb,WAAW;EACb;EACA,IAAI;EACJ,KAAK,IAAI,UAAU,GAAG,UAAU,aAAa,WAAW;GACtD,MAAM,UAAU,KAAK,IAAI;GACzB,MAAM,OAAO;IACX,SAAS,WAAW;IACpB,UAAU,KAAK;IACf;IACA,UAAU;IACV,SAAS;IACT,cAAc;IACd,WAAW;IACX,WAAW;IACX,aAAa;IACb,gBAAgB,CAAC;GACnB,CAAC;GACD,IAAI;IACF,MAAM,OAAO,MAAM,WAAW,YAAY;KACxC,SAAS;KACT,OAAO,KAAK,aAAa;KACzB,OAAO;KACP;KACA,eACE,KAAK,oBAAoB,KAAA,IACrB,KAAA,IACA,2BAA2B,SAAS,EAAE,iBAAiB,KAAK,gBAAgB,CAAC;KACnF,MAAM;MACJ,GAAG,KAAK;MACR,eAAe,YAAY;MAC3B,SAAS,OAAO,OAAO;KACzB;KACA,QAAQ,KAAK;KACb,UAAU,QAAQ,WAAW,KAAK,KAAK,SAAS;MAAE;MAAQ,gBAAgB;KAAO,CAAC;KAClF,SAAS;KACT,kBAAkB;IACpB,CAAC;IACD,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;IAChC,MAAM,OAAO,KAAK;IAMlB,kBAAkB,OAAO,KAAK,aAAa;KACzC,iBAAiB;KACjB,SAAS,uCAAuC,YAAY;IAC9D,CAAC;IACD,MAAM,MAAM,KAAK;IACjB,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc;KACd,gBAAgB,CAAC;IACnB,CAAC;IACD,OAAO,yBAAyB,GAAG;GACrC,SAAS,KAAK;IACZ,UAAU;IACV,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;KAC7D,gBAAgB,CAAC;IACnB,CAAC;IAGD,IAAI,eAAe,wBAAwB,MAAM;GACnD;EACF;EACA,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;CACtE;CACA,QAAQ,cAAc;EACpB,UAAU;EACV,SAAS;GAAE,MAAM;GAA2B,SAAS;EAAM;EAC3D,aAAa,CACX,uGACF;CACF;CACA,OAAO;AACT;;;ACpmBA,SAAS,aAAa,UAAsC;CAC1D,IAAI,CAAC,UAAU,OAAO;CACtB,IAAI,SAAS,SAAS,MAAM,GAAG,OAAO;CACtC,IAAI,SAAS,WAAW,OAAO,GAAG,OAAO;CACzC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,qBAAqB,QAAoD;CACvF,MAAM,YAAwB,CAAC;CAC/B,MAAM,YAAgC,CAAC;CACvC,MAAM,YAAsB,CAAC;CAC7B,MAAM,4BAAY,IAAI,IAAY;CAElC,KAAK,MAAM,MAAM,QACf,IAAI,GAAG,SAAS,aAAa;EAC3B,MAAM,OAAQ,GAAyB;EACvC,IAAI,QAAQ,CAAC,UAAU,IAAI,IAAI,GAAG;GAChC,UAAU,IAAI,IAAI;GAClB,UAAU,KAAK,IAAI;EACrB;CACF,OAAO,IAAI,GAAG,SAAS,YAAY;EACjC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,MAAM,aAAa,EAAE,QAAQ;GAC7B,MAAM,EAAE,QAAQ,EAAE,OAAO,EAAE;GAC3B,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,OAAO,IAAI,GAAG,SAAS,oBAAoB;EACzC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,IAAI,EAAE;GACN,OAAO,EAAE;GACT,QAAQ,EAAE,UAAU;GACpB,GAAI,EAAE,YAAY,KAAA,IAAY,EAAE,SAAS,EAAE,QAAQ,IAAI,CAAC;EAC1D,CAAC;CACH;CAGF,OAAO;EAAE;EAAW;EAAW;CAAU;AAC3C"}
1
+ {"version":3,"file":"produced-state-Be0BK3RN.js","names":[],"sources":["../src/agent-profile.ts","../src/completion-verifier.ts","../src/produced-state.ts"],"sourcesContent":["import { createHash } from 'node:crypto'\nimport type { AgentProfile } from '@tangle-network/agent-interface'\nimport { type HarnessType, harnessSupportsModel } from '@tangle-network/agent-interface'\nimport { ValidationError } from './errors'\nimport { canonicalString } from './ledger-core/canonical'\n\nexport type { AgentProfile, HarnessType } from '@tangle-network/agent-interface'\n\n/**\n * The agentic coding harnesses an eval sweeps by default — the ones we care about\n * ranking. This is the SINGLE source of that list; consumers import it instead of\n * re-declaring their own (a re-declared list is how the fleet drifts). Pass an\n * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known\n * harness) to widen beyond these.\n */\nexport const CODING_HARNESSES: readonly HarnessType[] = [\n 'opencode',\n 'claude-code',\n 'codex',\n 'kimi-code',\n]\n\nexport interface ProfileAxisSpec {\n /** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the\n * harness and model vary. `model.default` is the fallback model. */\n base: AgentProfile\n /** Harnesses to cross. Default: {@link CODING_HARNESSES}. */\n harnesses?: readonly HarnessType[]\n /** Models to cross. Default: `[base.model.default]` — one model, i.e. today's\n * single-model behaviour, so omitting this never changes an existing run. */\n models?: readonly string[]\n /** Force every (harness, model) pair verbatim, even ones the harness can't run —\n * for deliberately testing failure modes. Default (false): SNAP instead — a\n * vendor-locked harness runs only the swept models in its family, or its native\n * default when it supports none, so no harness is dropped and none gets a\n * guaranteed-failing foreign-model cell. */\n keepIncompatible?: boolean\n}\n\n/** Model sentinel for a vendor-locked harness that supports none of the swept models:\n * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness\n * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi\n * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship\n * table that would rot as router catalogs change. */\nexport const HARNESS_NATIVE_MODEL = 'default'\n\n/**\n * Expand a base profile across the harness × model matrix into the `AgentProfile[]`\n * that `runProfileMatrix` / `selfImprove` score — the ONE place \"which harnesses ×\n * which models do we evaluate\" lives, so no product hand-rolls its own harness list\n * or column→profile mapping (the pattern that let those copies drift and silently\n * break the harness pivot).\n *\n * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,\n * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both\n * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row\n * and results join back by harness/model via {@link harnessAxisOf} with no\n * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or\n * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every\n * requested harness runs; `keepIncompatible` forces every pair verbatim.\n *\n * Omit `harnesses`/`models` to sweep the full default set — the \"turn it on for\n * everything we care about\" switch, identical in shape whether one harness or all.\n */\nexport function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[] {\n const harnesses = spec.harnesses ?? CODING_HARNESSES\n if (harnesses.length === 0) throw new ValidationError('expandProfileAxes: no harnesses to sweep')\n const baseModel = spec.base.model?.default\n const models = spec.models ?? (baseModel ? [baseModel] : [])\n if (models.length === 0) {\n throw new ValidationError(\n 'expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied',\n )\n }\n const out: AgentProfile[] = []\n const seen = new Set<string>()\n for (const harness of harnesses) {\n // A universal (router-backed) harness — opencode/pi/claudish — runs every swept\n // model. A vendor-locked harness — codex/claude-code/kimi-code — runs only the\n // swept models in its own family; when it supports NONE of them it snaps to its\n // native default (the `HARNESS_NATIVE_MODEL` sentinel it resolves at runtime)\n // rather than being dropped, so every requested harness still appears in the\n // sweep on a model it can actually run — e.g. sweeping `deepseek/x` puts opencode\n // on deepseek and kimi-code on its own Kimi model, a real head-to-head.\n // `keepIncompatible` forces every (harness, model) pair verbatim (failure-mode runs).\n const supported = spec.keepIncompatible\n ? models\n : models.filter((model) => harnessSupportsModel(harness, model))\n const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL]\n for (const model of effective) {\n const profile: AgentProfile = {\n ...spec.base,\n harness,\n name: `${spec.base.name ?? 'agent'}/${harness}/${model}`,\n model: { ...spec.base.model, default: model },\n metadata: { ...(spec.base.metadata ?? {}), harness, harnessModel: model },\n }\n const id = agentProfileId(profile)\n if (seen.has(id)) continue\n seen.add(id)\n out.push(profile)\n }\n }\n if (out.length === 0) {\n // Unreachable in normal use — snapping guarantees ≥1 cell per harness — but keep a\n // fail-closed guard so a future refactor can't silently produce an empty sweep.\n throw new ValidationError(\n `expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(', ')}], models=[${models.join(', ')}]).`,\n )\n }\n return out\n}\n\n/**\n * Read the (harness, model) a matrix cell ran under, off a profile or a result row's\n * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile\n * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by\n * this instead of recomputing an id (recomputing the wrong key is what broke the pivot\n * in the hand-rolled copies).\n */\nexport function harnessAxisOf(\n profile: Pick<AgentProfile, 'metadata'>,\n): { harness: HarnessType; model: string } | undefined {\n const m = profile.metadata as Record<string, unknown> | undefined\n const harness = m?.harness\n const model = m?.harnessModel\n if (typeof harness === 'string' && typeof model === 'string') {\n return { harness: harness as HarnessType, model }\n }\n return undefined\n}\n\n/**\n * Collision-resistant, path-safe, human-readable profile id for eval artifacts.\n * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix\n * keys, and directory names where two profiles must not collapse onto one row.\n * The suffix is the first 64 bits of the behaviour hash, enough for ordinary\n * eval matrices while keeping filenames readable.\n */\nexport function agentProfileId(profile: AgentProfile): string {\n const label = pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? 'profile'\n return `${label}-${agentProfileHash(profile).slice(0, 16)}`\n}\n\n/**\n * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete\n * model id because run records reject bare/missing model aliases.\n */\nexport function agentProfileModelId(profile: AgentProfile): string {\n const model = profile.model?.default?.trim()\n if (!model) {\n const label = agentProfileDisplayLabel(profile) ?? 'unnamed profile'\n throw new ValidationError(\n `AgentProfile \"${label}\" has no model.default — cannot record eval run`,\n )\n }\n return model\n}\n\nfunction agentProfileDisplayLabel(profile: AgentProfile): string | undefined {\n return profile.name?.trim() || profile.version?.trim() || undefined\n}\n\nfunction pathSafeProfileLabel(label: string | undefined): string | undefined {\n const safe = label\n ?.trim()\n .replace(/[^A-Za-z0-9._-]+/g, '-')\n .replace(/-+/g, '-')\n .replace(/^-|-$/g, '')\n return safe || undefined\n}\n\nfunction compact<T extends Record<string, unknown>>(input: T): Partial<T> {\n const out: Record<string, unknown> = {}\n for (const [key, value] of Object.entries(input)) {\n if (value !== undefined) out[key] = value\n }\n return out as Partial<T>\n}\n\n/**\n * Deterministic behaviour identity for the canonical\n * `@tangle-network/agent-interface` AgentProfile.\n *\n * `name` and `description` are labels and do not affect the hash. Profile\n * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,\n * and extensions do affect the hash. Resource array order is hash-bearing\n * because mount order can change agent behaviour. Undefined fields are treated\n * as absent; explicit `null` fields remain hash-bearing.\n */\nexport function agentProfileHash(profile: AgentProfile): string {\n const model = agentProfileModelId(profile)\n const behaviour = compact({\n ...profile,\n name: undefined,\n description: undefined,\n tags: profile.tags ? [...profile.tags].sort() : undefined,\n model: compact({ ...profile.model, default: model }),\n })\n return createHash('sha256').update(canonicalString(behaviour)).digest('hex')\n}\n","/**\n * Completion verifier — the task-completion oracle.\n *\n * Answers the only eval question that is not a proxy: did the agent actually\n * COMPLETE the task — produce every required deliverable, persisted and\n * correct — rather than describe what should be done. A fluent transcript\n * that never produces the artifact scores zero here.\n *\n * Per requirement, a two-stage check:\n * 1. Structural — a produced item (vault artifact / approved proposal /\n * tool call) of the right kind is matched against the requirement and\n * carries non-empty content. Deterministic; no LLM.\n * 2. Correctness — only if structurally present AND the matched item\n * carries content, one targeted check decides whether that item\n * actually fulfils the requirement. A hallucinated artifact fails here;\n * an absent one already failed stage 1.\n *\n * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —\n * checker failures — are excluded from the denominator, never scored as\n * zeros). Quality dimensions are meaningless on an incomplete task — callers\n * gate on `fullyComplete` / `completionRate` before scoring quality.\n */\n\nimport { randomUUID } from 'node:crypto'\nimport type { ChatClient, ChatRequest } from './analyst/chat-client'\nimport type { Artifact } from './artifact-validator'\nimport { CostLedger, type CostLedgerHandle } from './cost-ledger'\nimport { assertServedModel, ModelSubstitutionError } from './integrity/served-model'\nimport { recoverTruncatedJson } from './json-recovery'\nimport { JudgeParseError } from './judges'\nimport {\n costReceiptFromLlm,\n costReceiptFromLlmError,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport { packageVersion } from './package-version'\nimport type { RawProviderEvent, RawProviderSink } from './trace/raw-provider-sink'\nimport {\n certificationEvidenceDigest,\n type DefaultVerdict,\n type VerdictCertification,\n} from './verdict'\nimport type { CheckerIdentity, VerificationStrategySource } from './verification-strategy'\n\n/** What kind of produced state can satisfy a requirement structurally. */\nexport type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any'\n\nexport interface CompletionRequirement {\n /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */\n reqId: string\n /** Human-readable description of the required deliverable. */\n title: string\n /** Optional kind/category hint, matched against a produced item's kind. */\n category?: string\n /** What produced state satisfies this requirement. Defaults to 'any'. */\n satisfiedBy?: SatisfiedBy\n}\n\nexport interface TaskGold {\n taskId: string\n requirements: CompletionRequirement[]\n}\n\nexport interface ProducedProposal {\n id: string\n title: string\n status: 'pending' | 'approved' | 'rejected'\n /** Optional persisted body — when present, enables a correctness check. */\n content?: string\n}\n\n/** Everything observable about what a run actually produced. */\nexport interface ProducedState {\n /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */\n artifacts: Artifact[]\n /** Proposals / filings the agent created. */\n proposals: ProducedProposal[]\n /** Names of tools the agent invoked. */\n toolCalls: string[]\n}\n\nexport interface RequirementCheck {\n reqId: string\n title: string\n /** A produced item of the right kind matched the requirement, non-empty. */\n structurallyPresent: boolean\n /**\n * Whether the matched item actually fulfils the requirement. `null` when\n * not structurally present, when the matched item carries no content\n * to assess, or when the correctness check itself failed (`unmeasured`).\n */\n correct: boolean | null\n /** structurallyPresent && !unmeasured && correct !== false. */\n satisfied: boolean\n /**\n * Set when the correctness check itself errored (LLM call failure or an\n * unparseable response after retry). The requirement's fulfilment is\n * UNKNOWN — `correct` stays null, `satisfied` is false, and\n * `completionVerdict` excludes the row from `completionRate`'s\n * denominator. Never folded into a zero: a synthetic zero is\n * indistinguishable from a real failure (see `JudgeParseError`).\n */\n unmeasured?: true\n /** Why the correctness check could not be measured (present iff `unmeasured`). */\n unmeasuredReason?: string\n /** Human-readable evidence for the verdict. */\n evidence: string[]\n}\n\n/** Extends the substrate verdict spine: `valid` = `fullyComplete` and\n * `score` = `completionRate` — derived in `completionVerdict()`, the one\n * place those equalities hold by construction. */\nexport interface CompletionVerdict extends DefaultVerdict {\n taskId: string\n requirements: RequirementCheck[]\n /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */\n completionRate: number\n /** Every measurable requirement satisfied (false when anything is unmeasured). */\n fullyComplete: boolean\n /** Requirements whose correctness check errored — reported, never scored as zero. */\n unmeasuredCount: number\n}\n\n/**\n * Construct a `CompletionVerdict` from the per-requirement checks, deriving\n * `completionRate` / `fullyComplete` and the spine fields (`valid` =\n * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero\n * requirements — a verdict over nothing is a misconfiguration, mirroring\n * `verifyCompletion`'s gold-spec guard.\n */\nexport function completionVerdict(input: {\n taskId: string\n requirements: RequirementCheck[]\n /** What certified the correctness stage, when anything did. Omitted =\n * an uncertified verdict — the honest default for a bare checker. */\n certification?: VerdictCertification\n}): CompletionVerdict {\n if (input.requirements.length === 0) {\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`,\n )\n }\n const measurable = input.requirements.filter((r) => !r.unmeasured)\n const unmeasuredCount = input.requirements.length - measurable.length\n if (measurable.length === 0) {\n // Every check errored: this is an infrastructure failure, not a scored\n // run. A 0-rate verdict here would be a fabricated measurement.\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? 'unknown reason'})`,\n )\n }\n const satisfiedCount = measurable.filter((r) => r.satisfied).length\n const completionRate = satisfiedCount / measurable.length\n // A run with unmeasured rows can still report a rate over what WAS\n // measured, but must not claim full completion.\n const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length\n return {\n taskId: input.taskId,\n requirements: input.requirements,\n completionRate,\n fullyComplete,\n unmeasuredCount,\n valid: fullyComplete,\n score: completionRate,\n ...(input.certification === undefined ? {} : { certification: input.certification }),\n }\n}\n\n/**\n * What a correctness checker declares about itself so the completion\n * verdict can carry a certification: which strategy member it discharges,\n * its exact identity, and the steps its answers rest on unverified.\n */\nexport interface CorrectnessCheckerAttestation {\n strategy: VerificationStrategySource\n checker: CheckerIdentity\n assumptions: string[]\n}\n\n/**\n * Decides whether a produced item's content actually fulfils a requirement.\n * Injected so the structural verifier stays pure and unit-testable; the\n * production implementation is `createLlmCorrectnessChecker`.\n *\n * `attestation` is optional metadata on the function value: a checker that\n * carries one yields certified completion verdicts; a bare function yields\n * the same verdict uncertified. A plain arrow function remains a valid\n * checker.\n */\nexport interface CorrectnessChecker {\n (\n requirement: CompletionRequirement,\n content: string,\n ): Promise<{ correct: boolean; reason: string }>\n attestation?: CorrectnessCheckerAttestation\n}\n\nconst STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'of',\n 'for',\n 'and',\n 'or',\n 'to',\n 'in',\n 'on',\n 'with',\n 'by',\n])\n\n// Deliverable-FORM vocabulary — words that name the SHAPE of an output, not its\n// domain content. A correct \"swap-comparison view persisted as a ui/** artifact\"\n// is an OpenUI JSON whose body says nothing about \"artifact\" / \"persisted\" /\n// \"view\"; the discriminative tokens are the domain nouns (swap, comparison).\n// Stripped from the REQUIREMENT side of structural recall so a deliverable is\n// matched on what it IS about, not on the boilerplate describing its form. The\n// correctness checker strips the same class via TITLE_STOPWORDS. Anti-game holds:\n// the distinctive domain tokens remain, so an off-topic item still fails.\nconst REQUIREMENT_FORM_STOPWORDS = new Set([\n 'generated',\n 'generate',\n 'view',\n 'render',\n 'rendered',\n 'persisted',\n 'persist',\n 'artifact',\n 'file',\n 'document',\n 'note',\n 'proposal',\n 'deliverable',\n 'output',\n 'created',\n 'create',\n 'produce',\n 'produced',\n 'flag',\n])\n\nconst MATCH_THRESHOLD = 0.5\nconst MIN_CONTENT_CHARS = 50\n\nfunction tokens(s: string, extraStop?: Set<string>): Set<string> {\n return new Set(\n s\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)),\n )\n}\n\n/**\n * Recall of the requirement's tokens within a candidate's identifying text.\n * Recall, not Jaccard — a candidate's path/id legitimately carries extra\n * tokens the requirement does not name. The requirement side drops\n * deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.\n */\nfunction tokenRecall(requirementText: string, candidateText: string): number {\n const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS)\n if (req.size === 0) return 0\n const cand = tokens(candidateText)\n let hit = 0\n for (const t of req) if (cand.has(t)) hit++\n return hit / req.size\n}\n\ninterface Candidate {\n reqIndex: number\n /** Unique key for a produced item — each item satisfies at most one requirement. */\n itemKey: string\n score: number\n evidence: string\n /** Content to correctness-check, or null when the matched item has none. */\n content: string | null\n}\n\nfunction artifactCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n artifacts: Artifact[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n artifacts.forEach((a, i) => {\n if ((a.content ?? '').trim().length < MIN_CONTENT_CHARS) return\n // Match against the artifact CONTENT too, not just its path + kind — a\n // generated view / note whose path is generic still satisfies a requirement\n // when its body covers it (e.g. an OpenUI comparison grounded in the on-file\n // figures). Bounded slice keeps the recall text cheap; MATCH_THRESHOLD holds.\n let score = tokenRecall(\n reqText,\n `${a.path ?? ''} ${a.kind} ${(a.content ?? '').slice(0, 4000)}`,\n )\n if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) {\n score = Math.max(score, 1)\n }\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `artifact:${i}`,\n score,\n evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,\n content: a.content ?? null,\n })\n })\n return out\n}\n\nfunction proposalCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n proposals: ProducedProposal[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n for (const p of proposals) {\n // Pending or rejected work is not a completed deliverable.\n if (p.status !== 'approved') continue\n // A proposal needs an assessable BODY to be a deliverable. A bare title is\n // not completion: correctness cannot be judged on it, so a title-only match\n // would auto-pass the oracle (structurallyPresent && correct===null →\n // satisfied) with no verifiable content. Tool calls are the only\n // legitimately content-less deliverable (`toolCallCandidates`).\n const body = (p.content ?? '').trim()\n if (body.length < MIN_CONTENT_CHARS) continue\n // Match against the body as well as the (often short) title — a refusal /\n // flag / analysis proposal whose title is a label still satisfies a\n // descriptively-worded requirement when its content covers it. MATCH_THRESHOLD\n // + the requirement's distinctive tokens keep an off-topic proposal out;\n // correctness (a SEMANTIC checker, NOT this lexical pass) then judges\n // polarity/fulfilment, so a negation that merely contains the tokens fails.\n // Structural and correctness must use different evidence or the two-stage\n // check collapses to one lexical gate.\n const score = tokenRecall(reqText, `${p.title} ${body}`)\n if (score < MATCH_THRESHOLD) continue\n out.push({\n reqIndex,\n itemKey: `proposal:${p.id}`,\n score,\n evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,\n content: body,\n })\n }\n return out\n}\n\nfunction toolCallCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n toolCalls: string[],\n): Candidate[] {\n const out: Candidate[] = []\n toolCalls.forEach((name, i) => {\n const score = tokenRecall(req.title, name)\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `tool:${i}`,\n score,\n evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,\n content: null,\n })\n })\n return out\n}\n\n/**\n * Verify whether a run completed the task. `checkCorrectness` is injected —\n * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.\n *\n * Throws on a gold spec with no requirements: an eval task that requires\n * nothing is a misconfiguration, not a vacuously-complete task.\n */\nexport async function verifyCompletion(\n gold: TaskGold,\n state: ProducedState,\n checkCorrectness: CorrectnessChecker,\n): Promise<CompletionVerdict> {\n if (gold.requirements.length === 0) {\n throw new Error(\n `verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`,\n )\n }\n\n // Collect every above-threshold (requirement, produced-item) candidate, then\n // assign greedily by descending score: each requirement and each produced\n // item is used at most once. One deliverable fulfils one requirement.\n const candidates: Candidate[] = []\n gold.requirements.forEach((req, i) => {\n const by = req.satisfiedBy ?? 'any'\n if (by === 'artifact' || by === 'any') {\n candidates.push(...artifactCandidates(req, i, state.artifacts))\n }\n if (by === 'proposal' || by === 'any') {\n candidates.push(...proposalCandidates(req, i, state.proposals))\n }\n if (by === 'tool-call' || by === 'any') {\n candidates.push(...toolCallCandidates(req, i, state.toolCalls))\n }\n })\n candidates.sort((a, b) => b.score - a.score)\n\n const assigned = new Map<number, Candidate>()\n const itemTaken = new Set<string>()\n for (const c of candidates) {\n if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue\n assigned.set(c.reqIndex, c)\n itemTaken.add(c.itemKey)\n }\n\n const requirements: RequirementCheck[] = []\n for (let i = 0; i < gold.requirements.length; i++) {\n const req = gold.requirements[i]!\n const match = assigned.get(i)\n const evidence: string[] = []\n let correct: boolean | null = null\n let unmeasuredReason: string | undefined\n\n if (match) {\n evidence.push(match.evidence)\n if (match.content !== null) {\n try {\n const r = await checkCorrectness(req, match.content)\n correct = r.correct\n evidence.push(`correctness: ${r.correct ? 'pass' : 'fail'} — ${r.reason}`)\n } catch (err) {\n // The CHECKER failed, not the requirement. Recording this as a\n // zero would fabricate a model failure out of an infrastructure\n // one; the requirement is unmeasured and leaves the denominator.\n unmeasuredReason =\n err instanceof JudgeParseError\n ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}`\n : `checker call failed: ${err instanceof Error ? err.message : String(err)}`\n evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`)\n }\n } else {\n evidence.push('correctness: not assessed — matched item carries no content')\n }\n } else {\n const by = req.satisfiedBy ?? 'any'\n const kind = by === 'any' ? 'artifact/proposal/tool-call' : by\n evidence.push(`no produced ${kind} matched this requirement`)\n }\n\n const structurallyPresent = match !== undefined\n const unmeasured = unmeasuredReason !== undefined\n const satisfied = structurallyPresent && !unmeasured && correct !== false\n requirements.push({\n reqId: req.reqId,\n title: req.title,\n structurallyPresent,\n correct,\n satisfied,\n ...(unmeasured ? { unmeasured: true as const, unmeasuredReason } : {}),\n evidence,\n })\n }\n\n // The verdict blends the deterministic structural stage with the injected\n // correctness stage, so the certification names the checker's own member\n // and carries the structural stage's lexical nature as an assumption. A\n // checker without an attestation yields an uncertified verdict.\n const attestation = checkCorrectness.attestation\n const certification: VerdictCertification | undefined = attestation\n ? {\n strategy: attestation.strategy,\n checker: attestation.checker,\n assumptions: [\n 'structural matching is lexical token recall over produced items — the correctness stage only sees items it matched',\n ...attestation.assumptions,\n ],\n evidenceDigest: certificationEvidenceDigest({ taskId: gold.taskId, requirements }),\n }\n : undefined\n\n return completionVerdict({\n taskId: gold.taskId,\n requirements,\n ...(certification === undefined ? {} : { certification }),\n })\n}\n\nexport interface LlmCorrectnessCheckerOpts {\n model?: string\n /** Optional ledger for direct use. */\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /** Max chars of artifact content sent to the checker. */\n maxContentChars?: number\n /**\n * Checker LLM calls per requirement before giving up (parse failures and\n * call errors both consume attempts). The failure then surfaces as an\n * `unmeasured` requirement, never a zero.\n */\n maxAttempts?: number\n /**\n * Forensic capture of every checker request/response/error — without it a\n * checker failure is unauditable (the agent-turn raws never contain the\n * checker's own calls). Same sink contract as `LlmClient`.\n */\n rawSink?: RawProviderSink\n}\n\n/**\n * Parse the correctness checker's model response. Tolerates a response\n * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the\n * verdict boolean usually lands in the first few tokens, so a recovered\n * prefix with a boolean `correct` is a real measurement, not a guess.\n * Fails loud (JudgeParseError) when no boolean verdict is recoverable.\n */\nexport function parseCorrectnessResponse(raw: string): { correct: boolean; reason: string } {\n const readVerdict = (candidate: unknown): { correct: boolean; reason: string } | null => {\n if (candidate === null || typeof candidate !== 'object') return null\n const { correct, reason } = candidate as { correct?: unknown; reason?: unknown }\n if (typeof correct !== 'boolean') return null\n return { correct, reason: typeof reason === 'string' ? reason : '' }\n }\n\n const match = raw.match(/\\{[\\s\\S]*\\}/)\n if (match) {\n try {\n const strict = readVerdict(JSON.parse(match[0]))\n if (strict) return strict\n } catch {\n // fall through to truncation recovery\n }\n }\n // The strict path needs a closing `}`; a cap-hit response has none. Take\n // everything from the first `{` and auto-close it.\n const start = raw.indexOf('{')\n if (start !== -1) {\n const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)))\n if (recovered) return recovered\n }\n throw new JudgeParseError('correctness-checker', raw)\n}\n\n/**\n * Production `CorrectnessChecker` — one LLM call per matched artifact,\n * deterministic (temperature 0), structured JSON out. Judges fulfilment\n * only: a plan, a gesture, or a description of what should be done does not\n * fulfil a requirement — the artifact must BE the deliverable.\n */\nexport function createLlmCorrectnessChecker(\n chat: ChatClient,\n opts: LlmCorrectnessCheckerOpts = {},\n): CorrectnessChecker {\n const model = opts.model ?? 'claude-sonnet-4-6'\n const maxContentChars = opts.maxContentChars ?? 8000\n const maxAttempts = opts.maxAttempts ?? 2\n const costLedger = opts.costLedger ?? new CostLedger()\n const sink = opts.rawSink\n const record = async (event: RawProviderEvent): Promise<void> => {\n // Forensic capture is best-effort; the verdict is the system of record.\n try {\n await sink?.record(event)\n } catch {\n // Intentionally swallowed.\n }\n }\n const checker: CorrectnessChecker = async (requirement, content) => {\n const request = {\n model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}.',\n },\n {\n role: 'user' as const,\n content: `Requirement: ${requirement.title}\\n${\n requirement.category ? `Category: ${requirement.category}\\n` : ''\n }\\nProduced artifact:\\n${content.slice(0, maxContentChars)}`,\n },\n ],\n temperature: 0,\n maxTokens: 200,\n } satisfies ChatRequest\n let lastErr: unknown\n for (let attempt = 0; attempt < maxAttempts; attempt++) {\n const started = Date.now()\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'request',\n timestamp: started,\n requestBody: request,\n redactedFields: [],\n })\n try {\n const paid = await costLedger.runPaidCall({\n channel: 'verifier',\n phase: opts.costPhase ?? 'completion.correctness',\n actor: 'correctness-checker',\n model,\n maximumCharge:\n chat.maximumAttempts === undefined\n ? undefined\n : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),\n tags: {\n ...opts.costTags,\n requirementId: requirement.reqId,\n attempt: String(attempt),\n },\n signal: opts.signal,\n execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),\n receipt: costReceiptFromLlm,\n receiptFromError: costReceiptFromLlmError,\n })\n if (!paid.succeeded) throw paid.error\n const resp = paid.value\n // Hold the transport to its own word: a verdict produced by a model\n // other than the one requested is not this model's verdict. A\n // transport that echoes no id cannot be made to prove identity here —\n // callers needing that proof enable it at the client\n // (`LlmClientOptions.assertServedModel`) or gate on `assertModelsServed`.\n assertServedModel(model, resp.servedModel, {\n allowUnreported: true,\n context: `correctness checker for requirement ${requirement.reqId}`,\n })\n const raw = resp.content\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'response',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n responseBody: resp,\n redactedFields: [],\n })\n return parseCorrectnessResponse(raw)\n } catch (err) {\n lastErr = err\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n // Substitution is a verdict about identity, not a transient fault:\n // another attempt reaches the same wrong model and spends more tokens.\n if (err instanceof ModelSubstitutionError) throw err\n }\n }\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n }\n checker.attestation = {\n strategy: 'judge',\n checker: { name: 'llm-correctness-checker', version: model },\n assumptions: [\n 'served-model identity is accepted when the transport reports none (assertServedModel allowUnreported)',\n ],\n }\n return checker\n}\n\n/** Stopwords for requirement-title tokenization — drops the imperative verbs\n * ('review', 'update', …) common to deliverable titles so recall keys on the\n * substantive nouns, not the boilerplate ask. */\nconst TITLE_STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'and',\n 'or',\n 'for',\n 'to',\n 'of',\n 'in',\n 'on',\n 'with',\n 'review',\n 'update',\n 'new',\n 'proposed',\n])\n\n/**\n * Deterministic `CorrectnessChecker` — the no-LLM counterpart to\n * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its\n * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`\n * of the requirement title's significant tokens. No network.\n *\n * Polarity-blind: token recall credits a negation that contains the\n * requirement's tokens (\"I will NOT produce the comparison\" recalls every token\n * of \"produce the comparison\"). The structural match stage is ALSO lexical, so\n * pairing the two collapses to a single gameable gate. Use this only as an\n * opt-in structural pre-filter or for tasks whose requirements have no polarity\n * to invert; for produced-state grading the correctness checker MUST be semantic\n * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.\n */\nexport function createTokenRecallChecker(\n opts: { minRecall?: number; minContentLength?: number } = {},\n): CorrectnessChecker {\n const minRecall = opts.minRecall ?? 0.5\n const minLen = opts.minContentLength ?? 120\n const checker: CorrectnessChecker = async (requirement, content) => {\n const body = content.trim()\n if (body.length < minLen)\n return {\n correct: false,\n reason: `content too thin (${body.length} chars) to be the deliverable`,\n }\n const titleTokens = requirement.title\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t))\n if (titleTokens.length === 0)\n return {\n correct: true,\n reason: 'requirement title has no significant tokens — structural match accepted',\n }\n const lower = body.toLowerCase()\n const hits = titleTokens.filter((t) => lower.includes(t)).length\n const recall = hits / titleTokens.length\n return recall >= minRecall\n ? {\n correct: true,\n reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`,\n }\n : {\n correct: false,\n reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`,\n }\n }\n // 'schema': token recall checks shape, not meaning — the member whose\n // documented failure mode (\"a well-formed wrong answer passes\") is\n // exactly this checker's polarity blindness.\n checker.attestation = {\n strategy: 'schema',\n checker: { name: 'token-recall-checker', version: packageVersion() },\n assumptions: ['polarity-blind: a negation that recalls the requirement tokens passes'],\n }\n return checker\n}\n","/**\n * Produced-state extraction — normalize a run's runtime event stream into the\n * typed `ProducedState` the completion oracle consumes.\n *\n * `ProducedState` answers \"what did the agent actually produce\" — vault\n * artifacts, proposals, tool calls. The runtime emits these as a stream of\n * events; this module is the single normalization point from that stream to\n * the shape `verifyCompletion` expects.\n *\n * Input is structurally typed (`RuntimeEventLike`) so this module does not\n * depend on agent-runtime — agent-runtime's `RuntimeStreamEvent` satisfies it\n * structurally. The `content` on `ArtifactEventLike` and the whole\n * `proposal_created` variant are the runtime-side enrichments this contract\n * requires; the runtime emits them, this module consumes them.\n */\n\nimport type { Artifact } from './artifact-validator'\nimport type { ProducedProposal, ProducedState } from './completion-verifier'\n\n/** A tool the agent invoked. */\nexport interface ToolCallEventLike {\n type: 'tool_call'\n toolName: string\n}\n\n/**\n * An artifact the agent produced. `content` is the enriched field — the\n * runtime's base `artifact` event carries only metadata; the completion\n * oracle needs the body to verify the deliverable, so the runtime emits it.\n */\nexport interface ArtifactEventLike {\n type: 'artifact'\n artifactId: string\n name?: string\n mimeType?: string\n uri?: string\n content?: string\n}\n\n/** A proposal / filing the agent created. */\nexport interface ProposalEventLike {\n type: 'proposal_created'\n proposalId: string\n title: string\n status?: 'pending' | 'approved' | 'rejected'\n // body of the proposal (e.g. a submit_proposal `description`). When present,\n // the completion oracle correctness-checks it like artifact content; absent,\n // the proposal is graded presence-only.\n content?: string\n}\n\n/**\n * The subset of runtime stream events `extractProducedState` consumes.\n * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;\n * the `{ type: string }` catch-all keeps the input permissive so callers can\n * pass the whole unfiltered telemetry stream — unrecognized events are skipped.\n */\nexport type RuntimeEventLike =\n | ToolCallEventLike\n | ArtifactEventLike\n | ProposalEventLike\n | { type: string }\n\nfunction artifactKind(mimeType: string | undefined): string {\n if (!mimeType) return 'file'\n if (mimeType.includes('json')) return 'json'\n if (mimeType.startsWith('text/')) return 'text'\n return 'file'\n}\n\n/**\n * Normalize a run's runtime event stream into `ProducedState`.\n *\n * Pure and total — unrecognized event types are skipped. `toolCalls` is\n * deduplicated by name in first-seen order (completion cares about a tool's\n * presence, not its call count). An artifact with neither a name nor a uri\n * still yields an entry keyed by its `artifactId` so it is never silently\n * dropped; an artifact with no `content` yields empty content, which the\n * completion oracle's structural check then rejects on its own.\n */\nexport function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState {\n const artifacts: Artifact[] = []\n const proposals: ProducedProposal[] = []\n const toolCalls: string[] = []\n const seenTools = new Set<string>()\n\n for (const ev of events) {\n if (ev.type === 'tool_call') {\n const name = (ev as ToolCallEventLike).toolName\n if (name && !seenTools.has(name)) {\n seenTools.add(name)\n toolCalls.push(name)\n }\n } else if (ev.type === 'artifact') {\n const a = ev as ArtifactEventLike\n artifacts.push({\n kind: artifactKind(a.mimeType),\n path: a.name ?? a.uri ?? a.artifactId,\n content: a.content ?? '',\n })\n } else if (ev.type === 'proposal_created') {\n const p = ev as ProposalEventLike\n proposals.push({\n id: p.proposalId,\n title: p.title,\n status: p.status ?? 'pending',\n ...(p.content !== undefined ? { content: p.content } : {}),\n })\n }\n }\n\n return { artifacts, proposals, toolCalls }\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAa,mBAA2C;CACtD;CACA;CACA;CACA;AACF;;;;;;AAwBA,MAAa,uBAAuB;;;;;;;;;;;;;;;;;;;AAoBpC,SAAgB,kBAAkB,MAAuC;CACvE,MAAM,YAAY,KAAK,aAAa;CACpC,IAAI,UAAU,WAAW,GAAG,MAAM,IAAI,gBAAgB,0CAA0C;CAChG,MAAM,YAAY,KAAK,KAAK,OAAO;CACnC,MAAM,SAAS,KAAK,WAAW,YAAY,CAAC,SAAS,IAAI,CAAC;CAC1D,IAAI,OAAO,WAAW,GACpB,MAAM,IAAI,gBACR,kGACF;CAEF,MAAM,MAAsB,CAAC;CAC7B,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,WAAW,WAAW;EAS/B,MAAM,YAAY,KAAK,mBACnB,SACA,OAAO,QAAQ,UAAU,qBAAqB,SAAS,KAAK,CAAC;EACjE,MAAM,YAAY,UAAU,SAAS,IAAI,YAAY,CAAC,oBAAoB;EAC1E,KAAK,MAAM,SAAS,WAAW;GAC7B,MAAM,UAAwB;IAC5B,GAAG,KAAK;IACR;IACA,MAAM,GAAG,KAAK,KAAK,QAAQ,QAAQ,GAAG,QAAQ,GAAG;IACjD,OAAO;KAAE,GAAG,KAAK,KAAK;KAAO,SAAS;IAAM;IAC5C,UAAU;KAAE,GAAI,KAAK,KAAK,YAAY,CAAC;KAAI;KAAS,cAAc;IAAM;GAC1E;GACA,MAAM,KAAK,eAAe,OAAO;GACjC,IAAI,KAAK,IAAI,EAAE,GAAG;GAClB,KAAK,IAAI,EAAE;GACX,IAAI,KAAK,OAAO;EAClB;CACF;CACA,IAAI,IAAI,WAAW,GAGjB,MAAM,IAAI,gBACR,uDAAuD,UAAU,KAAK,IAAI,EAAE,aAAa,OAAO,KAAK,IAAI,EAAE,IAC7G;CAEF,OAAO;AACT;;;;;;;;AASA,SAAgB,cACd,SACqD;CACrD,MAAM,IAAI,QAAQ;CAClB,MAAM,UAAU,GAAG;CACnB,MAAM,QAAQ,GAAG;CACjB,IAAI,OAAO,YAAY,YAAY,OAAO,UAAU,UAClD,OAAO;EAAW;EAAwB;CAAM;AAGpD;;;;;;;;AASA,SAAgB,eAAe,SAA+B;CAE5D,OAAO,GADO,qBAAqB,yBAAyB,OAAO,CAAC,KAAK,UACzD,GAAG,iBAAiB,OAAO,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1D;;;;;AAMA,SAAgB,oBAAoB,SAA+B;CACjE,MAAM,QAAQ,QAAQ,OAAO,SAAS,KAAK;CAC3C,IAAI,CAAC,OAEH,MAAM,IAAI,gBACR,iBAFY,yBAAyB,OAAO,KAAK,kBAE1B,gDACzB;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,SAA2C;CAC3E,OAAO,QAAQ,MAAM,KAAK,KAAK,QAAQ,SAAS,KAAK,KAAK,KAAA;AAC5D;AAEA,SAAS,qBAAqB,OAA+C;CAM3E,OALa,OACT,KAAK,CAAC,CACP,QAAQ,qBAAqB,GAAG,CAAC,CACjC,QAAQ,OAAO,GAAG,CAAC,CACnB,QAAQ,UAAU,EAAE,KACR,KAAA;AACjB;AAEA,SAAS,QAA2C,OAAsB;CACxE,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,KAAK,GAC7C,IAAI,UAAU,KAAA,GAAW,IAAI,OAAO;CAEtC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,iBAAiB,SAA+B;CAC9D,MAAM,QAAQ,oBAAoB,OAAO;CACzC,MAAM,YAAY,QAAQ;EACxB,GAAG;EACH,MAAM,KAAA;EACN,aAAa,KAAA;EACb,MAAM,QAAQ,OAAO,CAAC,GAAG,QAAQ,IAAI,CAAC,CAAC,KAAK,IAAI,KAAA;EAChD,OAAO,QAAQ;GAAE,GAAG,QAAQ;GAAO,SAAS;EAAM,CAAC;CACrD,CAAC;CACD,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,gBAAgB,SAAS,CAAC,CAAC,CAAC,OAAO,KAAK;AAC7E;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACtEA,SAAgB,kBAAkB,OAMZ;CACpB,IAAI,MAAM,aAAa,WAAW,GAChC,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,+DAC3C;CAEF,MAAM,aAAa,MAAM,aAAa,QAAQ,MAAM,CAAC,EAAE,UAAU;CACjE,MAAM,kBAAkB,MAAM,aAAa,SAAS,WAAW;CAC/D,IAAI,WAAW,WAAW,GAGxB,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,yCAAyC,MAAM,aAAa,OAAO,8BAA8B,MAAM,aAAa,EAAE,EAAE,oBAAoB,iBAAiB,EACxM;CAEF,MAAM,iBAAiB,WAAW,QAAQ,MAAM,EAAE,SAAS,CAAC,CAAC;CAC7D,MAAM,iBAAiB,iBAAiB,WAAW;CAGnD,MAAM,gBAAgB,oBAAoB,KAAK,mBAAmB,WAAW;CAC7E,OAAO;EACL,QAAQ,MAAM;EACd,cAAc,MAAM;EACpB;EACA;EACA;EACA,OAAO;EACP,OAAO;EACP,GAAI,MAAM,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,eAAe,MAAM,cAAc;CACpF;AACF;AA+BA,MAAM,4BAAY,IAAI,IAAI;CACxB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAUD,MAAM,6CAA6B,IAAI,IAAI;CACzC;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAM,kBAAkB;AACxB,MAAM,oBAAoB;AAE1B,SAAS,OAAO,GAAW,WAAsC;CAC/D,OAAO,IAAI,IACT,EACG,YAAY,CAAC,CACb,MAAM,YAAY,CAAC,CACnB,QAAQ,MAAM,EAAE,SAAS,KAAK,CAAC,UAAU,IAAI,CAAC,KAAK,CAAC,WAAW,IAAI,CAAC,CAAC,CAC1E;AACF;;;;;;;AAQA,SAAS,YAAY,iBAAyB,eAA+B;CAC3E,MAAM,MAAM,OAAO,iBAAiB,0BAA0B;CAC9D,IAAI,IAAI,SAAS,GAAG,OAAO;CAC3B,MAAM,OAAO,OAAO,aAAa;CACjC,IAAI,MAAM;CACV,KAAK,MAAM,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC,GAAG;CACtC,OAAO,MAAM,IAAI;AACnB;AAYA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,GAAG,MAAM;EAC1B,KAAK,EAAE,WAAW,GAAA,CAAI,KAAK,CAAC,CAAC,SAAS,mBAAmB;EAKzD,IAAI,QAAQ,YACV,SACA,GAAG,EAAE,QAAQ,GAAG,GAAG,EAAE,KAAK,IAAI,EAAE,WAAW,GAAA,CAAI,MAAM,GAAG,GAAI,GAC9D;EACA,IAAI,IAAI,YAAY,EAAE,QAAQ,IAAI,SAAS,YAAY,MAAM,EAAE,KAAK,YAAY,GAC9E,QAAQ,KAAK,IAAI,OAAO,CAAC;EAE3B,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY;GACrB;GACA,UAAU,aAAa,EAAE,QAAQ,EAAE,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,CAAC;CACD,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,KAAK,MAAM,KAAK,WAAW;EAEzB,IAAI,EAAE,WAAW,YAAY;EAM7B,MAAM,QAAQ,EAAE,WAAW,GAAA,CAAI,KAAK;EACpC,IAAI,KAAK,SAAS,mBAAmB;EASrC,MAAM,QAAQ,YAAY,SAAS,GAAG,EAAE,MAAM,GAAG,MAAM;EACvD,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY,EAAE;GACvB;GACA,UAAU,sBAAsB,EAAE,MAAM,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS;EACX,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,MAAM,MAAM;EAC7B,MAAM,QAAQ,YAAY,IAAI,OAAO,IAAI;EACzC,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,QAAQ;GACjB;GACA,UAAU,cAAc,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACxE,SAAS;EACX,CAAC;CACH,CAAC;CACD,OAAO;AACT;;;;;;;;AASA,eAAsB,iBACpB,MACA,OACA,kBAC4B;CAC5B,IAAI,KAAK,aAAa,WAAW,GAC/B,MAAM,IAAI,MACR,2BAA2B,KAAK,OAAO,4CACzC;CAMF,MAAM,aAA0B,CAAC;CACjC,KAAK,aAAa,SAAS,KAAK,MAAM;EACpC,MAAM,KAAK,IAAI,eAAe;EAC9B,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,eAAe,OAAO,OAC/B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;CAElE,CAAC;CACD,WAAW,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;CAE3C,MAAM,2BAAW,IAAI,IAAuB;CAC5C,MAAM,4BAAY,IAAI,IAAY;CAClC,KAAK,MAAM,KAAK,YAAY;EAC1B,IAAI,SAAS,IAAI,EAAE,QAAQ,KAAK,UAAU,IAAI,EAAE,OAAO,GAAG;EAC1D,SAAS,IAAI,EAAE,UAAU,CAAC;EAC1B,UAAU,IAAI,EAAE,OAAO;CACzB;CAEA,MAAM,eAAmC,CAAC;CAC1C,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,aAAa,QAAQ,KAAK;EACjD,MAAM,MAAM,KAAK,aAAa;EAC9B,MAAM,QAAQ,SAAS,IAAI,CAAC;EAC5B,MAAM,WAAqB,CAAC;EAC5B,IAAI,UAA0B;EAC9B,IAAI;EAEJ,IAAI,OAAO;GACT,SAAS,KAAK,MAAM,QAAQ;GAC5B,IAAI,MAAM,YAAY,MACpB,IAAI;IACF,MAAM,IAAI,MAAM,iBAAiB,KAAK,MAAM,OAAO;IACnD,UAAU,EAAE;IACZ,SAAS,KAAK,gBAAgB,EAAE,UAAU,SAAS,OAAO,KAAK,EAAE,QAAQ;GAC3E,SAAS,KAAK;IAIZ,mBACE,eAAe,kBACX,6CAA6C,IAAI,IAAI,MAAM,GAAG,GAAG,MACjE,wBAAwB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAC7E,SAAS,KAAK,6BAA6B,kBAAkB;GAC/D;QAEA,SAAS,KAAK,6DAA6D;EAE/E,OAAO;GACL,MAAM,KAAK,IAAI,eAAe;GAC9B,MAAM,OAAO,OAAO,QAAQ,gCAAgC;GAC5D,SAAS,KAAK,eAAe,KAAK,0BAA0B;EAC9D;EAEA,MAAM,sBAAsB,UAAU,KAAA;EACtC,MAAM,aAAa,qBAAqB,KAAA;EACxC,MAAM,YAAY,uBAAuB,CAAC,cAAc,YAAY;EACpE,aAAa,KAAK;GAChB,OAAO,IAAI;GACX,OAAO,IAAI;GACX;GACA;GACA;GACA,GAAI,aAAa;IAAE,YAAY;IAAe;GAAiB,IAAI,CAAC;GACpE;EACF,CAAC;CACH;CAMA,MAAM,cAAc,iBAAiB;CACrC,MAAM,gBAAkD,cACpD;EACE,UAAU,YAAY;EACtB,SAAS,YAAY;EACrB,aAAa,CACX,sHACA,GAAG,YAAY,WACjB;EACA,gBAAgB,4BAA4B;GAAE,QAAQ,KAAK;GAAQ;EAAa,CAAC;CACnF,IACA,KAAA;CAEJ,OAAO,kBAAkB;EACvB,QAAQ,KAAK;EACb;EACA,GAAI,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,cAAc;CACzD,CAAC;AACH;;;;;;;;AAgCA,SAAgB,yBAAyB,KAAmD;CAC1F,MAAM,eAAe,cAAoE;EACvF,IAAI,cAAc,QAAQ,OAAO,cAAc,UAAU,OAAO;EAChE,MAAM,EAAE,SAAS,WAAW;EAC5B,IAAI,OAAO,YAAY,WAAW,OAAO;EACzC,OAAO;GAAE;GAAS,QAAQ,OAAO,WAAW,WAAW,SAAS;EAAG;CACrE;CAEA,MAAM,QAAQ,IAAI,MAAM,aAAa;CACrC,IAAI,OACF,IAAI;EACF,MAAM,SAAS,YAAY,KAAK,MAAM,MAAM,EAAE,CAAC;EAC/C,IAAI,QAAQ,OAAO;CACrB,QAAQ,CAER;CAIF,MAAM,QAAQ,IAAI,QAAQ,GAAG;CAC7B,IAAI,UAAU,IAAI;EAChB,MAAM,YAAY,YAAY,qBAAqB,IAAI,MAAM,KAAK,CAAC,CAAC;EACpE,IAAI,WAAW,OAAO;CACxB;CACA,MAAM,IAAI,gBAAgB,uBAAuB,GAAG;AACtD;;;;;;;AAQA,SAAgB,4BACd,MACA,OAAkC,CAAC,GACf;CACpB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,cAAc,KAAK,eAAe;CACxC,MAAM,aAAa,KAAK,cAAc,IAAI,WAAW;CACrD,MAAM,OAAO,KAAK;CAClB,MAAM,SAAS,OAAO,UAA2C;EAE/D,IAAI;GACF,MAAM,MAAM,OAAO,KAAK;EAC1B,QAAQ,CAER;CACF;CACA,MAAM,UAA8B,OAAO,aAAa,YAAY;EAClE,MAAM,UAAU;GACd;GACA,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IACE,MAAM;IACN,SAAS,gBAAgB,YAAY,MAAM,IACzC,YAAY,WAAW,aAAa,YAAY,SAAS,MAAM,GAChE,wBAAwB,QAAQ,MAAM,GAAG,eAAe;GAC3D,CACF;GACA,aAAa;GACb,WAAW;EACb;EACA,IAAI;EACJ,KAAK,IAAI,UAAU,GAAG,UAAU,aAAa,WAAW;GACtD,MAAM,UAAU,KAAK,IAAI;GACzB,MAAM,OAAO;IACX,SAAS,WAAW;IACpB,UAAU,KAAK;IACf;IACA,UAAU;IACV,SAAS;IACT,cAAc;IACd,WAAW;IACX,WAAW;IACX,aAAa;IACb,gBAAgB,CAAC;GACnB,CAAC;GACD,IAAI;IACF,MAAM,OAAO,MAAM,WAAW,YAAY;KACxC,SAAS;KACT,OAAO,KAAK,aAAa;KACzB,OAAO;KACP;KACA,eACE,KAAK,oBAAoB,KAAA,IACrB,KAAA,IACA,2BAA2B,SAAS,EAAE,iBAAiB,KAAK,gBAAgB,CAAC;KACnF,MAAM;MACJ,GAAG,KAAK;MACR,eAAe,YAAY;MAC3B,SAAS,OAAO,OAAO;KACzB;KACA,QAAQ,KAAK;KACb,UAAU,QAAQ,WAAW,KAAK,KAAK,SAAS;MAAE;MAAQ,gBAAgB;KAAO,CAAC;KAClF,SAAS;KACT,kBAAkB;IACpB,CAAC;IACD,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;IAChC,MAAM,OAAO,KAAK;IAMlB,kBAAkB,OAAO,KAAK,aAAa;KACzC,iBAAiB;KACjB,SAAS,uCAAuC,YAAY;IAC9D,CAAC;IACD,MAAM,MAAM,KAAK;IACjB,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc;KACd,gBAAgB,CAAC;IACnB,CAAC;IACD,OAAO,yBAAyB,GAAG;GACrC,SAAS,KAAK;IACZ,UAAU;IACV,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;KAC7D,gBAAgB,CAAC;IACnB,CAAC;IAGD,IAAI,eAAe,wBAAwB,MAAM;GACnD;EACF;EACA,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;CACtE;CACA,QAAQ,cAAc;EACpB,UAAU;EACV,SAAS;GAAE,MAAM;GAA2B,SAAS;EAAM;EAC3D,aAAa,CACX,uGACF;CACF;CACA,OAAO;AACT;;;ACpmBA,SAAS,aAAa,UAAsC;CAC1D,IAAI,CAAC,UAAU,OAAO;CACtB,IAAI,SAAS,SAAS,MAAM,GAAG,OAAO;CACtC,IAAI,SAAS,WAAW,OAAO,GAAG,OAAO;CACzC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,qBAAqB,QAAoD;CACvF,MAAM,YAAwB,CAAC;CAC/B,MAAM,YAAgC,CAAC;CACvC,MAAM,YAAsB,CAAC;CAC7B,MAAM,4BAAY,IAAI,IAAY;CAElC,KAAK,MAAM,MAAM,QACf,IAAI,GAAG,SAAS,aAAa;EAC3B,MAAM,OAAQ,GAAyB;EACvC,IAAI,QAAQ,CAAC,UAAU,IAAI,IAAI,GAAG;GAChC,UAAU,IAAI,IAAI;GAClB,UAAU,KAAK,IAAI;EACrB;CACF,OAAO,IAAI,GAAG,SAAS,YAAY;EACjC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,MAAM,aAAa,EAAE,QAAQ;GAC7B,MAAM,EAAE,QAAQ,EAAE,OAAO,EAAE;GAC3B,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,OAAO,IAAI,GAAG,SAAS,oBAAoB;EACzC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,IAAI,EAAE;GACN,OAAO,EAAE;GACT,QAAQ,EAAE,UAAU;GACpB,GAAI,EAAE,YAAY,KAAA,IAAY,EAAE,SAAS,EAAE,QAAQ,IAAI,CAAC;EAC1D,CAAC;CACH;CAGF,OAAO;EAAE;EAAW;EAAW;CAAU;AAC3C"}
@@ -1,5 +1,5 @@
1
- import { r as pairedBootstrap } from "./paired-tests-BHIhYVdu.js";
2
- import { o as pairHoldout, r as detectScale, s as decidePairedPromotion } from "./power-preflight-DEw-uC7q.js";
1
+ import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
2
+ import { o as pairHoldout, r as detectScale, s as decidePairedPromotion } from "./power-preflight-CFXm0Vjo.js";
3
3
  //#region src/campaign/gates/promotion-policy.ts
4
4
  /**
5
5
  * Promotion policy over the evidence VECTOR — the substrate's answer to "never
@@ -183,4 +183,4 @@ function paretoSignificanceGate(options) {
183
183
  //#endregion
184
184
  export { paretoPolicy as n, paretoSignificanceGate as r, buildEvidenceVector as t };
185
185
 
186
- //# sourceMappingURL=promotion-policy-xzA40Evo.js.map
186
+ //# sourceMappingURL=promotion-policy-LY9mVQ7W.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"promotion-policy-xzA40Evo.js","names":[],"sources":["../src/campaign/gates/promotion-policy.ts"],"sourcesContent":["/**\n * Promotion policy over the evidence VECTOR — the substrate's answer to \"never\n * collapse the multi-objective promotion decision into one scalar.\" A\n * `defaultProductionGate` is one opinionated composition; this module factors\n * the decision into two reusable pieces so MANY policies can compete over the\n * SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):\n *\n * buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus\n * PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy\n * paretoPolicy(ev) // the default strategy\n * paretoSignificanceGate(options): Gate // bus + policy as a Gate\n *\n * The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a\n * potential gain source AND a safety floor (unlike `defaultProductionGate`,\n * where only `composite` can win and `criticalDimensions` are pure floors). A\n * candidate ships iff it weakly DOMINATES the baseline at the confidence level —\n * no objective credibly worse (CI floor breach) AND at least one objective\n * credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work\n * (NOT folded into hold: \"gather more reps\" and \"reject\" are different actions).\n *\n * Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate\n * per-side cost, no per-cell observation vector to bootstrap. Treat them as hard\n * constraints (compose with a budget gate via `composeGate`), not faked CIs.\n */\n\nimport {\n decidePairedPromotion,\n type PairedDecisionMethod,\n type PairedDecisionStatistic,\n type PairedMcNemarEvidence,\n} from '../../paired-promotion-decision'\nimport type { Direction } from '../../pareto'\nimport {\n DECISION_PAIRED_DELTA_STATISTIC,\n type PairedBootstrapResult,\n pairedBootstrap,\n} from '../../statistics'\nimport type { Gate, GateContext, GateDecision, GateResult, JudgeScore, Scenario } from '../types'\nimport { detectScale, pairHoldout } from './statistical-heldout'\n\n/** Where an objective's per-cell scalar comes from. `composite` reads the\n * judge's composite; `dimension` reads a named per-dimension score. */\nexport type ObjectiveSource = { kind: 'composite' } | { kind: 'dimension'; dimension: string }\n\nexport interface PromotionObjective {\n /** Stable label used in reports + `contributingGates`. */\n name: string\n source: ObjectiveSource\n /** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients\n * the paired delta so a positive bootstrap always means \"candidate better\". */\n direction: Direction\n /** The good-direction paired-delta CI lower bound must EXCEED this to count\n * as a significant gain on this axis. Interpreted in the judge's native\n * scale. Default 0 (⇒ \"confidently better\"). */\n gainThreshold?: number\n /** A floor breach (regression) is declared when the good-direction CI lower\n * bound is below −floorTolerance, or when the exact small-sample test proves\n * a drop past it. When omitted it auto-scales off observed magnitudes\n * (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */\n floorTolerance?: number\n}\n\n/** Per-axis verdict from the good-direction paired bootstrap. */\nexport type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs'\n\nexport interface AxisEvidence {\n name: string\n source: ObjectiveSource\n direction: Direction\n /** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):\n * a positive value means the candidate is better on this axis.\n *\n * DIAGNOSTIC on a pass/fail axis: there the verdict is decided on Tango's\n * score interval instead, because a percentile bootstrap over a three-atom\n * delta lattice is not a valid interval at the nonzero margin `floorTolerance`\n * and `gainThreshold` create. `ci` carries the interval that decided. */\n bootstrap: PairedBootstrapResult\n /** Which paired statistic `bootstrap.low`/`.high` bracket. `'mean'` unless the\n * caller asked for the median — on a pass/fail axis the median and its whole\n * CI are pinned at 0 by tie domination and can see neither a gain nor a\n * regression. `bootstrap.median` still carries the median point estimate. */\n bootstrapStatistic: 'median' | 'mean'\n /** The interval the axis verdict was actually decided on, good-direction and\n * in the axis's native units. */\n ci: { low: number; high: number }\n /** Which estimator produced `ci`. */\n decisionStatistic: PairedDecisionStatistic\n /** McNemar's exact evidence on a pass/fail axis; null otherwise. */\n mcnemar: PairedMcNemarEvidence | null\n /** `ci` has zero width — no evidence in either direction, so the axis is\n * neither improved nor regressed however the point estimate sits. */\n indeterminate: boolean\n /** Paired observations contributing to this axis. */\n n: number\n minimumRequired: number\n decisionMethod: PairedDecisionMethod\n gainThreshold: number\n floorTolerance: number\n verdict: AxisVerdict\n}\n\nexport interface EvidenceVector {\n /** One entry per objective — NOTHING averaged across axes. */\n axes: AxisEvidence[]\n /** Smallest paired n across axes that produced observations — the binding\n * evidence-sufficiency constraint. 0 when no axis produced observations. */\n minN: number\n /** Aggregate per-side cost from the gate context (a constraint input, not a\n * CI axis — see the module header). */\n cost: { candidate: number; baseline: number }\n}\n\n/** A promotion strategy: a pure function from the evidence vector to a verdict.\n * Many policies can run over the same `EvidenceVector` and disagree — that's\n * the point (competing strategies, shared evidence). */\nexport type PromotionPolicy = (ev: EvidenceVector) => GateResult\n\nexport interface BuildEvidenceVectorOptions {\n /** Minimum paired observations before an axis can claim significance; below\n * it the axis is `few_runs`. The exact small-sample test may require more\n * observations at the selected confidence. */\n minProductiveRuns?: number\n /** Confidence level for every axis bootstrap. Default 0.95. */\n confidence?: number\n /** Bootstrap resamples. Default 2000. */\n resamples?: number\n /** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */\n seed?: number\n /** Paired statistic every axis CI is computed on. Default `'mean'` — see\n * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */\n statistic?: 'mean' | 'median'\n}\n\n/**\n * The Evidence Bus. For each objective, pair candidate vs baseline by full\n * cellId and bootstrap a CI on the good-direction paired delta. Reuses the\n * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so\n * a single source of truth governs pairing granularity + scale handling.\n */\nexport function buildEvidenceVector<TArtifact, TScenario extends Scenario>(\n ctx: GateContext<TArtifact, TScenario>,\n objectives: PromotionObjective[],\n opts: BuildEvidenceVectorOptions = {},\n): EvidenceVector {\n if (objectives.length === 0) {\n throw new Error('buildEvidenceVector: at least 1 objective required')\n }\n const confidence = opts.confidence ?? 0.95\n const resamples = opts.resamples ?? 2000\n const seed = opts.seed ?? 1337\n const baseline = ctx.baselineJudgeScores ?? ctx.judgeScores\n const scenarioIds = new Set(ctx.scenarios.map((s) => s.id))\n\n const axes: AxisEvidence[] = []\n for (const obj of objectives) {\n let select: (s: JudgeScore) => number | undefined\n if (obj.source.kind === 'composite') {\n select = (s) => s.composite\n } else {\n const dim = obj.source.dimension\n select = (s) => s.dimensions[dim]\n }\n const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select)\n // Orient to the good direction: maximize ⇒ bootstrap (candidate − baseline);\n // minimize ⇒ bootstrap (baseline − candidate) by swapping args, so a\n // positive bootstrap always reads as \"candidate better on this axis\".\n const before = obj.direction === 'maximize' ? paired.before : paired.after\n const after = obj.direction === 'maximize' ? paired.after : paired.before\n const n = paired.before.length\n const floorTolerance =\n obj.floorTolerance ?? 0.05 * detectScale([...paired.before, ...paired.after])\n const gainThreshold = obj.gainThreshold ?? 0\n // Axes are decided on the MEAN paired delta — which for a pass/fail axis is\n // exactly the change in success rate. The median is structurally blind on\n // the shapes eval data lands in: with most pairs tied its bootstrap CI\n // collapses to [0,0] and the axis reads 'flat', hiding real gains AND real\n // regressions — pass/fail axes on ANY encoding ({0,1} and the 0-100 one\n // `detectScale` exists for), and low-cardinality axes even below half ties.\n // Orthogonal to `pairedDeltaTest`'s own small-sample switch: that picks the\n // TEST (bootstrap CI at n ≥ 20, exact sign test below), this picks the\n // ESTIMATOR the test is applied to. Both are needed — an exact sign test on\n // a tie-pinned median is still blind.\n const bootstrapStatistic = opts.statistic ?? DECISION_PAIRED_DELTA_STATISTIC\n // Both burdens of proof route through the ONE shared rule\n // (`decidePairedPromotion`), so a pass/fail axis is judged on Tango's score\n // interval — the only paired-binary construction valid at the nonzero\n // margins `gainThreshold` / `floorTolerance` create — and a zero-width\n // interval cannot buy a verdict in either direction.\n const improvement = decidePairedPromotion(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: gainThreshold,\n minPairs: opts.minProductiveRuns,\n })\n const regression = decidePairedPromotion(after, before, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: floorTolerance,\n minPairs: opts.minProductiveRuns,\n })\n const bootstrap =\n improvement.bootstrap ??\n pairedBootstrap(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n })\n // A floor breach fires on EITHER burden of proof, because they cover\n // different failures and the floor is the anti-Goodhart guard:\n // - `improvement.low < -floorTolerance` — the CREDIBLE WORST CASE exceeds\n // the tolerance. This is the contract `AxisEvidence.floorTolerance` and\n // `paretoPolicy`'s own reason string state, and it is the conservative\n // posture a safety axis needs: block unless the data can rule the\n // breach out, rather than waiting for the breach to be proven. Read off\n // the DECIDING interval, so a pass/fail axis is not screened by a\n // bootstrap that is pinned wherever ties dominate.\n // - `regression.promote` — a PROVEN drop past the tolerance. Adds the\n // small-sample path, where the decision is an exact sign test because\n // the bootstrap interval is descriptive only.\n // The credible-worst-case arm stays on the BOOTSTRAP, deliberately. Reading\n // it off the score interval instead would change what the floor MEANS on a\n // pass/fail axis: with every pair concordant the score interval is\n // ±z²/(n+z²) — ±0.39 at n=6, ±0.16 at n=20 — so a completely unchanged\n // safety axis would breach a 0.05 floor at any realistic n, and the gate\n // would refuse everything. That the bootstrap arm is instead fail-OPEN on a\n // tied pass/fail axis is a real and separate weakness: the honest fix is a\n // minimum-power requirement on the floor, not a wider interval, because the\n // data genuinely cannot rule a 5pp drop out at n=20 and a gate that says so\n // by blocking every candidate is not usable. `regression.promote` — the\n // PROVEN-drop arm — does route through the shared rule, so a real pass/fail\n // regression is now caught on an interval valid at the nonzero tolerance.\n const floorBreached = bootstrap.low < -floorTolerance || regression.promote\n // Floor check precedes the gain check: a credible regression must never be\n // masked as \"improved\". With the defaults (gainThreshold 0, positive floor)\n // the regions are disjoint and order is moot, but a consumer who sets a\n // negative gainThreshold (\"accept small dips\") could otherwise have a real\n // floor breach classified as a gain — anti-Goodhart wins the tie.\n const verdict: AxisVerdict = !improvement.sufficient\n ? 'few_runs'\n : floorBreached\n ? 'regressed'\n : improvement.promote\n ? 'improved'\n : 'flat'\n axes.push({\n name: obj.name,\n source: obj.source,\n direction: obj.direction,\n bootstrap,\n bootstrapStatistic,\n ci: { low: improvement.low, high: improvement.high },\n decisionStatistic: improvement.statistic,\n mcnemar: improvement.mcnemar,\n indeterminate: improvement.indeterminate,\n n,\n minimumRequired: improvement.minimumPairs,\n decisionMethod: improvement.method,\n gainThreshold,\n floorTolerance,\n verdict,\n })\n }\n const ns = axes.map((a) => a.n).filter((n) => n > 0)\n const minN = ns.length > 0 ? Math.min(...ns) : 0\n return { axes, minN, cost: { candidate: ctx.cost.candidate, baseline: ctx.cost.baseline } }\n}\n\n/**\n * The default strategy: symmetric multi-objective Pareto significance. Ship iff\n * the candidate weakly dominates the baseline at the confidence level — no axis\n * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold\n * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →\n * need_more_work. Statistically equivalent → hold (never ship noise).\n */\nexport const paretoPolicy: PromotionPolicy = (ev) => {\n const contributingGates = ev.axes.map((ax) => ({\n name: `objective:${ax.name}`,\n status:\n ax.verdict === 'regressed'\n ? ('fail' as const)\n : ax.verdict === 'few_runs'\n ? ('not_evaluated' as const)\n : ('pass' as const),\n detail: {\n direction: ax.direction,\n source: ax.source,\n verdict: ax.verdict,\n n: ax.n,\n deltaMedian: ax.bootstrap.median,\n ciLow: ax.ci.low,\n ciHigh: ax.ci.high,\n decisionStatistic: ax.decisionStatistic,\n decisionMethod: ax.decisionMethod,\n mcnemar: ax.mcnemar,\n indeterminate: ax.indeterminate,\n bootstrapCiLow: ax.bootstrap.low,\n bootstrapCiHigh: ax.bootstrap.high,\n confidence: ax.bootstrap.confidence,\n gainThreshold: ax.gainThreshold,\n floorTolerance: ax.floorTolerance,\n },\n }))\n\n const regressed = ev.axes.filter((a) => a.verdict === 'regressed')\n const fewRuns = ev.axes.filter((a) => a.verdict === 'few_runs')\n const improved = ev.axes.filter((a) => a.verdict === 'improved')\n\n let decision: GateDecision\n const reasons: string[] = []\n if (regressed.length > 0) {\n // Floor breach dominates: a credible regression on ANY axis blocks ship even\n // if another axis improved. This makes the +gain/−safety false positive\n // structurally impossible whenever the safety dim is an objective.\n decision = 'hold'\n for (const a of regressed) {\n reasons.push(\n `objective '${a.name}' regressed: good-direction CI.low ${a.ci.low.toFixed(3)} < -${a.floorTolerance} (n=${a.n})`,\n )\n }\n } else if (fewRuns.length > 0) {\n // No credible regression on the scored axes, but ≥1 axis lacks the evidence\n // to claim a gain ⇒ gather more reps, do NOT reject.\n decision = 'need_more_work'\n for (const a of fewRuns) {\n reasons.push(\n `objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance`,\n )\n }\n } else if (improved.length > 0) {\n // Weakly dominates (no axis worse) AND strictly better on ≥1 axis ⇒ a Pareto\n // improvement at the confidence level.\n decision = 'ship'\n reasons.push(\n `Pareto improvement at the confidence level: ${improved\n .map(\n (a) =>\n `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`,\n )\n .join(', ')}; no objective regressed`,\n )\n } else {\n // Enough evidence, nothing credibly better or worse ⇒ statistically\n // equivalent. Do NOT ship a no-op.\n decision = 'hold'\n reasons.push(\n 'no Pareto improvement: candidate statistically equivalent to baseline on every objective',\n )\n }\n\n // `delta` surfaces the composite axis if present, else the first axis — a\n // single convenience scalar; the vector lives in `contributingGates`.\n const composite = ev.axes.find((a) => a.source.kind === 'composite') ?? ev.axes[0]\n return { decision, reasons, contributingGates, delta: composite?.bootstrap.median }\n}\n\nexport interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {\n /** The objective vector. Every axis is both a gain source and a safety floor. */\n objectives: PromotionObjective[]\n /** Strategy applied to the evidence vector. Default `paretoPolicy`. Override\n * to run a stricter/looser strategy over the SAME bus (competing policies). */\n policy?: PromotionPolicy\n /** Override the gate name in reports. */\n name?: string\n}\n\n/**\n * Wrap the bus + a policy as a `Gate`. Plugs into the existing\n * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default\n * loop behavior is unchanged because consumers opt in by passing this gate.\n */\nexport function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(\n options: ParetoSignificanceGateOptions,\n): Gate<TArtifact, TScenario> {\n if (options.objectives.length === 0) {\n throw new Error('paretoSignificanceGate: at least 1 objective required')\n }\n const policy = options.policy ?? paretoPolicy\n return {\n name: options.name ?? 'paretoSignificanceGate',\n async decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult> {\n const ev = buildEvidenceVector(ctx, options.objectives, options)\n return policy(ev)\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA2IA,SAAgB,oBACd,KACA,YACA,OAAmC,CAAC,GACpB;CAChB,IAAI,WAAW,WAAW,GACxB,MAAM,IAAI,MAAM,oDAAoD;CAEtE,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,WAAW,IAAI,uBAAuB,IAAI;CAChD,MAAM,cAAc,IAAI,IAAI,IAAI,UAAU,KAAK,MAAM,EAAE,EAAE,CAAC;CAE1D,MAAM,OAAuB,CAAC;CAC9B,KAAK,MAAM,OAAO,YAAY;EAC5B,IAAI;EACJ,IAAI,IAAI,OAAO,SAAS,aACtB,UAAU,MAAM,EAAE;OACb;GACL,MAAM,MAAM,IAAI,OAAO;GACvB,UAAU,MAAM,EAAE,WAAW;EAC/B;EACA,MAAM,SAAS,YAAY,IAAI,aAAa,UAAU,aAAa,MAAM;EAIzE,MAAM,SAAS,IAAI,cAAc,aAAa,OAAO,SAAS,OAAO;EACrE,MAAM,QAAQ,IAAI,cAAc,aAAa,OAAO,QAAQ,OAAO;EACnE,MAAM,IAAI,OAAO,OAAO;EACxB,MAAM,iBACJ,IAAI,kBAAkB,MAAO,YAAY,CAAC,GAAG,OAAO,QAAQ,GAAG,OAAO,KAAK,CAAC;EAC9E,MAAM,gBAAgB,IAAI,iBAAiB;EAW3C,MAAM,qBAAqB,KAAK,aAAA;EAMhC,MAAM,cAAc,sBAAsB,QAAQ,OAAO;GACvD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;EACjB,CAAC;EACD,MAAM,aAAa,sBAAsB,OAAO,QAAQ;GACtD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;EACjB,CAAC;EACD,MAAM,YACJ,YAAY,aACZ,gBAAgB,QAAQ,OAAO;GAC7B;GACA;GACA,WAAW;GACX;EACF,CAAC;EAyBH,MAAM,gBAAgB,UAAU,MAAM,CAAC,kBAAkB,WAAW;EAMpE,MAAM,UAAuB,CAAC,YAAY,aACtC,aACA,gBACE,cACA,YAAY,UACV,aACA;EACR,KAAK,KAAK;GACR,MAAM,IAAI;GACV,QAAQ,IAAI;GACZ,WAAW,IAAI;GACf;GACA;GACA,IAAI;IAAE,KAAK,YAAY;IAAK,MAAM,YAAY;GAAK;GACnD,mBAAmB,YAAY;GAC/B,SAAS,YAAY;GACrB,eAAe,YAAY;GAC3B;GACA,iBAAiB,YAAY;GAC7B,gBAAgB,YAAY;GAC5B;GACA;GACA;EACF,CAAC;CACH;CACA,MAAM,KAAK,KAAK,KAAK,MAAM,EAAE,CAAC,CAAC,CAAC,QAAQ,MAAM,IAAI,CAAC;CAEnD,OAAO;EAAE;EAAM,MADF,GAAG,SAAS,IAAI,KAAK,IAAI,GAAG,EAAE,IAAI;EAC1B,MAAM;GAAE,WAAW,IAAI,KAAK;GAAW,UAAU,IAAI,KAAK;EAAS;CAAE;AAC5F;;;;;;;;AASA,MAAa,gBAAiC,OAAO;CACnD,MAAM,oBAAoB,GAAG,KAAK,KAAK,QAAQ;EAC7C,MAAM,aAAa,GAAG;EACtB,QACE,GAAG,YAAY,cACV,SACD,GAAG,YAAY,aACZ,kBACA;EACT,QAAQ;GACN,WAAW,GAAG;GACd,QAAQ,GAAG;GACX,SAAS,GAAG;GACZ,GAAG,GAAG;GACN,aAAa,GAAG,UAAU;GAC1B,OAAO,GAAG,GAAG;GACb,QAAQ,GAAG,GAAG;GACd,mBAAmB,GAAG;GACtB,gBAAgB,GAAG;GACnB,SAAS,GAAG;GACZ,eAAe,GAAG;GAClB,gBAAgB,GAAG,UAAU;GAC7B,iBAAiB,GAAG,UAAU;GAC9B,YAAY,GAAG,UAAU;GACzB,eAAe,GAAG;GAClB,gBAAgB,GAAG;EACrB;CACF,EAAE;CAEF,MAAM,YAAY,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,WAAW;CACjE,MAAM,UAAU,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,UAAU;CAC9D,MAAM,WAAW,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,UAAU;CAE/D,IAAI;CACJ,MAAM,UAAoB,CAAC;CAC3B,IAAI,UAAU,SAAS,GAAG;EAIxB,WAAW;EACX,KAAK,MAAM,KAAK,WACd,QAAQ,KACN,cAAc,EAAE,KAAK,qCAAqC,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,MAAM,EAAE,eAAe,MAAM,EAAE,EAAE,EACjH;CAEJ,OAAO,IAAI,QAAQ,SAAS,GAAG;EAG7B,WAAW;EACX,KAAK,MAAM,KAAK,SACd,QAAQ,KACN,cAAc,EAAE,KAAK,eAAe,EAAE,EAAE,2DAC1C;CAEJ,OAAO,IAAI,SAAS,SAAS,GAAG;EAG9B,WAAW;EACX,QAAQ,KACN,+CAA+C,SAC5C,KACE,MACC,IAAI,EAAE,KAAK,KAAK,EAAE,GAAG,MAAM,IAAI,EAAE,GAAG,IAAI,QAAQ,CAAC,IAAI,EAAE,UAAU,KAAK,QAAQ,CAAC,EAAE,WAAW,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,EACpH,CAAC,CACA,KAAK,IAAI,EAAE,yBAChB;CACF,OAAO;EAGL,WAAW;EACX,QAAQ,KACN,0FACF;CACF;CAIA,MAAM,YAAY,GAAG,KAAK,MAAM,MAAM,EAAE,OAAO,SAAS,WAAW,KAAK,GAAG,KAAK;CAChF,OAAO;EAAE;EAAU;EAAS;EAAmB,OAAO,WAAW,UAAU;CAAO;AACpF;;;;;;AAiBA,SAAgB,uBACd,SAC4B;CAC5B,IAAI,QAAQ,WAAW,WAAW,GAChC,MAAM,IAAI,MAAM,uDAAuD;CAEzE,MAAM,SAAS,QAAQ,UAAU;CACjC,OAAO;EACL,MAAM,QAAQ,QAAQ;EACtB,MAAM,OAAO,KAA6D;GACxE,MAAM,KAAK,oBAAoB,KAAK,QAAQ,YAAY,OAAO;GAC/D,OAAO,OAAO,EAAE;EAClB;CACF;AACF"}
1
+ {"version":3,"file":"promotion-policy-LY9mVQ7W.js","names":[],"sources":["../src/campaign/gates/promotion-policy.ts"],"sourcesContent":["/**\n * Promotion policy over the evidence VECTOR — the substrate's answer to \"never\n * collapse the multi-objective promotion decision into one scalar.\" A\n * `defaultProductionGate` is one opinionated composition; this module factors\n * the decision into two reusable pieces so MANY policies can compete over the\n * SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):\n *\n * buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus\n * PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy\n * paretoPolicy(ev) // the default strategy\n * paretoSignificanceGate(options): Gate // bus + policy as a Gate\n *\n * The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a\n * potential gain source AND a safety floor (unlike `defaultProductionGate`,\n * where only `composite` can win and `criticalDimensions` are pure floors). A\n * candidate ships iff it weakly DOMINATES the baseline at the confidence level —\n * no objective credibly worse (CI floor breach) AND at least one objective\n * credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work\n * (NOT folded into hold: \"gather more reps\" and \"reject\" are different actions).\n *\n * Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate\n * per-side cost, no per-cell observation vector to bootstrap. Treat them as hard\n * constraints (compose with a budget gate via `composeGate`), not faked CIs.\n */\n\nimport {\n decidePairedPromotion,\n type PairedDecisionMethod,\n type PairedDecisionStatistic,\n type PairedMcNemarEvidence,\n} from '../../paired-promotion-decision'\nimport type { Direction } from '../../pareto'\nimport {\n DECISION_PAIRED_DELTA_STATISTIC,\n type PairedBootstrapResult,\n pairedBootstrap,\n} from '../../statistics'\nimport type { Gate, GateContext, GateDecision, GateResult, JudgeScore, Scenario } from '../types'\nimport { detectScale, pairHoldout } from './statistical-heldout'\n\n/** Where an objective's per-cell scalar comes from. `composite` reads the\n * judge's composite; `dimension` reads a named per-dimension score. */\nexport type ObjectiveSource = { kind: 'composite' } | { kind: 'dimension'; dimension: string }\n\nexport interface PromotionObjective {\n /** Stable label used in reports + `contributingGates`. */\n name: string\n source: ObjectiveSource\n /** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients\n * the paired delta so a positive bootstrap always means \"candidate better\". */\n direction: Direction\n /** The good-direction paired-delta CI lower bound must EXCEED this to count\n * as a significant gain on this axis. Interpreted in the judge's native\n * scale. Default 0 (⇒ \"confidently better\"). */\n gainThreshold?: number\n /** A floor breach (regression) is declared when the good-direction CI lower\n * bound is below −floorTolerance, or when the exact small-sample test proves\n * a drop past it. When omitted it auto-scales off observed magnitudes\n * (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */\n floorTolerance?: number\n}\n\n/** Per-axis verdict from the good-direction paired bootstrap. */\nexport type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs'\n\nexport interface AxisEvidence {\n name: string\n source: ObjectiveSource\n direction: Direction\n /** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):\n * a positive value means the candidate is better on this axis.\n *\n * DIAGNOSTIC on a pass/fail axis: there the verdict is decided on Tango's\n * score interval instead, because a percentile bootstrap over a three-atom\n * delta lattice is not a valid interval at the nonzero margin `floorTolerance`\n * and `gainThreshold` create. `ci` carries the interval that decided. */\n bootstrap: PairedBootstrapResult\n /** Which paired statistic `bootstrap.low`/`.high` bracket. `'mean'` unless the\n * caller asked for the median — on a pass/fail axis the median and its whole\n * CI are pinned at 0 by tie domination and can see neither a gain nor a\n * regression. `bootstrap.median` still carries the median point estimate. */\n bootstrapStatistic: 'median' | 'mean'\n /** The interval the axis verdict was actually decided on, good-direction and\n * in the axis's native units. */\n ci: { low: number; high: number }\n /** Which estimator produced `ci`. */\n decisionStatistic: PairedDecisionStatistic\n /** McNemar's exact evidence on a pass/fail axis; null otherwise. */\n mcnemar: PairedMcNemarEvidence | null\n /** `ci` has zero width — no evidence in either direction, so the axis is\n * neither improved nor regressed however the point estimate sits. */\n indeterminate: boolean\n /** Paired observations contributing to this axis. */\n n: number\n minimumRequired: number\n decisionMethod: PairedDecisionMethod\n gainThreshold: number\n floorTolerance: number\n verdict: AxisVerdict\n}\n\nexport interface EvidenceVector {\n /** One entry per objective — NOTHING averaged across axes. */\n axes: AxisEvidence[]\n /** Smallest paired n across axes that produced observations — the binding\n * evidence-sufficiency constraint. 0 when no axis produced observations. */\n minN: number\n /** Aggregate per-side cost from the gate context (a constraint input, not a\n * CI axis — see the module header). */\n cost: { candidate: number; baseline: number }\n}\n\n/** A promotion strategy: a pure function from the evidence vector to a verdict.\n * Many policies can run over the same `EvidenceVector` and disagree — that's\n * the point (competing strategies, shared evidence). */\nexport type PromotionPolicy = (ev: EvidenceVector) => GateResult\n\nexport interface BuildEvidenceVectorOptions {\n /** Minimum paired observations before an axis can claim significance; below\n * it the axis is `few_runs`. The exact small-sample test may require more\n * observations at the selected confidence. */\n minProductiveRuns?: number\n /** Confidence level for every axis bootstrap. Default 0.95. */\n confidence?: number\n /** Bootstrap resamples. Default 2000. */\n resamples?: number\n /** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */\n seed?: number\n /** Paired statistic every axis CI is computed on. Default `'mean'` — see\n * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */\n statistic?: 'mean' | 'median'\n}\n\n/**\n * The Evidence Bus. For each objective, pair candidate vs baseline by full\n * cellId and bootstrap a CI on the good-direction paired delta. Reuses the\n * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so\n * a single source of truth governs pairing granularity + scale handling.\n */\nexport function buildEvidenceVector<TArtifact, TScenario extends Scenario>(\n ctx: GateContext<TArtifact, TScenario>,\n objectives: PromotionObjective[],\n opts: BuildEvidenceVectorOptions = {},\n): EvidenceVector {\n if (objectives.length === 0) {\n throw new Error('buildEvidenceVector: at least 1 objective required')\n }\n const confidence = opts.confidence ?? 0.95\n const resamples = opts.resamples ?? 2000\n const seed = opts.seed ?? 1337\n const baseline = ctx.baselineJudgeScores ?? ctx.judgeScores\n const scenarioIds = new Set(ctx.scenarios.map((s) => s.id))\n\n const axes: AxisEvidence[] = []\n for (const obj of objectives) {\n let select: (s: JudgeScore) => number | undefined\n if (obj.source.kind === 'composite') {\n select = (s) => s.composite\n } else {\n const dim = obj.source.dimension\n select = (s) => s.dimensions[dim]\n }\n const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select)\n // Orient to the good direction: maximize ⇒ bootstrap (candidate − baseline);\n // minimize ⇒ bootstrap (baseline − candidate) by swapping args, so a\n // positive bootstrap always reads as \"candidate better on this axis\".\n const before = obj.direction === 'maximize' ? paired.before : paired.after\n const after = obj.direction === 'maximize' ? paired.after : paired.before\n const n = paired.before.length\n const floorTolerance =\n obj.floorTolerance ?? 0.05 * detectScale([...paired.before, ...paired.after])\n const gainThreshold = obj.gainThreshold ?? 0\n // Axes are decided on the MEAN paired delta — which for a pass/fail axis is\n // exactly the change in success rate. The median is structurally blind on\n // the shapes eval data lands in: with most pairs tied its bootstrap CI\n // collapses to [0,0] and the axis reads 'flat', hiding real gains AND real\n // regressions — pass/fail axes on ANY encoding ({0,1} and the 0-100 one\n // `detectScale` exists for), and low-cardinality axes even below half ties.\n // Orthogonal to `pairedDeltaTest`'s own small-sample switch: that picks the\n // TEST (bootstrap CI at n ≥ 20, exact sign test below), this picks the\n // ESTIMATOR the test is applied to. Both are needed — an exact sign test on\n // a tie-pinned median is still blind.\n const bootstrapStatistic = opts.statistic ?? DECISION_PAIRED_DELTA_STATISTIC\n // Both burdens of proof route through the ONE shared rule\n // (`decidePairedPromotion`), so a pass/fail axis is judged on Tango's score\n // interval — the only paired-binary construction valid at the nonzero\n // margins `gainThreshold` / `floorTolerance` create — and a zero-width\n // interval cannot buy a verdict in either direction.\n const improvement = decidePairedPromotion(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: gainThreshold,\n minPairs: opts.minProductiveRuns,\n })\n const regression = decidePairedPromotion(after, before, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: floorTolerance,\n minPairs: opts.minProductiveRuns,\n })\n const bootstrap =\n improvement.bootstrap ??\n pairedBootstrap(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n })\n // A floor breach fires on EITHER burden of proof, because they cover\n // different failures and the floor is the anti-Goodhart guard:\n // - `improvement.low < -floorTolerance` — the CREDIBLE WORST CASE exceeds\n // the tolerance. This is the contract `AxisEvidence.floorTolerance` and\n // `paretoPolicy`'s own reason string state, and it is the conservative\n // posture a safety axis needs: block unless the data can rule the\n // breach out, rather than waiting for the breach to be proven. Read off\n // the DECIDING interval, so a pass/fail axis is not screened by a\n // bootstrap that is pinned wherever ties dominate.\n // - `regression.promote` — a PROVEN drop past the tolerance. Adds the\n // small-sample path, where the decision is an exact sign test because\n // the bootstrap interval is descriptive only.\n // The credible-worst-case arm stays on the BOOTSTRAP, deliberately. Reading\n // it off the score interval instead would change what the floor MEANS on a\n // pass/fail axis: with every pair concordant the score interval is\n // ±z²/(n+z²) — ±0.39 at n=6, ±0.16 at n=20 — so a completely unchanged\n // safety axis would breach a 0.05 floor at any realistic n, and the gate\n // would refuse everything. That the bootstrap arm is instead fail-OPEN on a\n // tied pass/fail axis is a real and separate weakness: the honest fix is a\n // minimum-power requirement on the floor, not a wider interval, because the\n // data genuinely cannot rule a 5pp drop out at n=20 and a gate that says so\n // by blocking every candidate is not usable. `regression.promote` — the\n // PROVEN-drop arm — does route through the shared rule, so a real pass/fail\n // regression is now caught on an interval valid at the nonzero tolerance.\n const floorBreached = bootstrap.low < -floorTolerance || regression.promote\n // Floor check precedes the gain check: a credible regression must never be\n // masked as \"improved\". With the defaults (gainThreshold 0, positive floor)\n // the regions are disjoint and order is moot, but a consumer who sets a\n // negative gainThreshold (\"accept small dips\") could otherwise have a real\n // floor breach classified as a gain — anti-Goodhart wins the tie.\n const verdict: AxisVerdict = !improvement.sufficient\n ? 'few_runs'\n : floorBreached\n ? 'regressed'\n : improvement.promote\n ? 'improved'\n : 'flat'\n axes.push({\n name: obj.name,\n source: obj.source,\n direction: obj.direction,\n bootstrap,\n bootstrapStatistic,\n ci: { low: improvement.low, high: improvement.high },\n decisionStatistic: improvement.statistic,\n mcnemar: improvement.mcnemar,\n indeterminate: improvement.indeterminate,\n n,\n minimumRequired: improvement.minimumPairs,\n decisionMethod: improvement.method,\n gainThreshold,\n floorTolerance,\n verdict,\n })\n }\n const ns = axes.map((a) => a.n).filter((n) => n > 0)\n const minN = ns.length > 0 ? Math.min(...ns) : 0\n return { axes, minN, cost: { candidate: ctx.cost.candidate, baseline: ctx.cost.baseline } }\n}\n\n/**\n * The default strategy: symmetric multi-objective Pareto significance. Ship iff\n * the candidate weakly dominates the baseline at the confidence level — no axis\n * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold\n * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →\n * need_more_work. Statistically equivalent → hold (never ship noise).\n */\nexport const paretoPolicy: PromotionPolicy = (ev) => {\n const contributingGates = ev.axes.map((ax) => ({\n name: `objective:${ax.name}`,\n status:\n ax.verdict === 'regressed'\n ? ('fail' as const)\n : ax.verdict === 'few_runs'\n ? ('not_evaluated' as const)\n : ('pass' as const),\n detail: {\n direction: ax.direction,\n source: ax.source,\n verdict: ax.verdict,\n n: ax.n,\n deltaMedian: ax.bootstrap.median,\n ciLow: ax.ci.low,\n ciHigh: ax.ci.high,\n decisionStatistic: ax.decisionStatistic,\n decisionMethod: ax.decisionMethod,\n mcnemar: ax.mcnemar,\n indeterminate: ax.indeterminate,\n bootstrapCiLow: ax.bootstrap.low,\n bootstrapCiHigh: ax.bootstrap.high,\n confidence: ax.bootstrap.confidence,\n gainThreshold: ax.gainThreshold,\n floorTolerance: ax.floorTolerance,\n },\n }))\n\n const regressed = ev.axes.filter((a) => a.verdict === 'regressed')\n const fewRuns = ev.axes.filter((a) => a.verdict === 'few_runs')\n const improved = ev.axes.filter((a) => a.verdict === 'improved')\n\n let decision: GateDecision\n const reasons: string[] = []\n if (regressed.length > 0) {\n // Floor breach dominates: a credible regression on ANY axis blocks ship even\n // if another axis improved. This makes the +gain/−safety false positive\n // structurally impossible whenever the safety dim is an objective.\n decision = 'hold'\n for (const a of regressed) {\n reasons.push(\n `objective '${a.name}' regressed: good-direction CI.low ${a.ci.low.toFixed(3)} < -${a.floorTolerance} (n=${a.n})`,\n )\n }\n } else if (fewRuns.length > 0) {\n // No credible regression on the scored axes, but ≥1 axis lacks the evidence\n // to claim a gain ⇒ gather more reps, do NOT reject.\n decision = 'need_more_work'\n for (const a of fewRuns) {\n reasons.push(\n `objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance`,\n )\n }\n } else if (improved.length > 0) {\n // Weakly dominates (no axis worse) AND strictly better on ≥1 axis ⇒ a Pareto\n // improvement at the confidence level.\n decision = 'ship'\n reasons.push(\n `Pareto improvement at the confidence level: ${improved\n .map(\n (a) =>\n `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`,\n )\n .join(', ')}; no objective regressed`,\n )\n } else {\n // Enough evidence, nothing credibly better or worse ⇒ statistically\n // equivalent. Do NOT ship a no-op.\n decision = 'hold'\n reasons.push(\n 'no Pareto improvement: candidate statistically equivalent to baseline on every objective',\n )\n }\n\n // `delta` surfaces the composite axis if present, else the first axis — a\n // single convenience scalar; the vector lives in `contributingGates`.\n const composite = ev.axes.find((a) => a.source.kind === 'composite') ?? ev.axes[0]\n return { decision, reasons, contributingGates, delta: composite?.bootstrap.median }\n}\n\nexport interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {\n /** The objective vector. Every axis is both a gain source and a safety floor. */\n objectives: PromotionObjective[]\n /** Strategy applied to the evidence vector. Default `paretoPolicy`. Override\n * to run a stricter/looser strategy over the SAME bus (competing policies). */\n policy?: PromotionPolicy\n /** Override the gate name in reports. */\n name?: string\n}\n\n/**\n * Wrap the bus + a policy as a `Gate`. Plugs into the existing\n * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default\n * loop behavior is unchanged because consumers opt in by passing this gate.\n */\nexport function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(\n options: ParetoSignificanceGateOptions,\n): Gate<TArtifact, TScenario> {\n if (options.objectives.length === 0) {\n throw new Error('paretoSignificanceGate: at least 1 objective required')\n }\n const policy = options.policy ?? paretoPolicy\n return {\n name: options.name ?? 'paretoSignificanceGate',\n async decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult> {\n const ev = buildEvidenceVector(ctx, options.objectives, options)\n return policy(ev)\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA2IA,SAAgB,oBACd,KACA,YACA,OAAmC,CAAC,GACpB;CAChB,IAAI,WAAW,WAAW,GACxB,MAAM,IAAI,MAAM,oDAAoD;CAEtE,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,WAAW,IAAI,uBAAuB,IAAI;CAChD,MAAM,cAAc,IAAI,IAAI,IAAI,UAAU,KAAK,MAAM,EAAE,EAAE,CAAC;CAE1D,MAAM,OAAuB,CAAC;CAC9B,KAAK,MAAM,OAAO,YAAY;EAC5B,IAAI;EACJ,IAAI,IAAI,OAAO,SAAS,aACtB,UAAU,MAAM,EAAE;OACb;GACL,MAAM,MAAM,IAAI,OAAO;GACvB,UAAU,MAAM,EAAE,WAAW;EAC/B;EACA,MAAM,SAAS,YAAY,IAAI,aAAa,UAAU,aAAa,MAAM;EAIzE,MAAM,SAAS,IAAI,cAAc,aAAa,OAAO,SAAS,OAAO;EACrE,MAAM,QAAQ,IAAI,cAAc,aAAa,OAAO,QAAQ,OAAO;EACnE,MAAM,IAAI,OAAO,OAAO;EACxB,MAAM,iBACJ,IAAI,kBAAkB,MAAO,YAAY,CAAC,GAAG,OAAO,QAAQ,GAAG,OAAO,KAAK,CAAC;EAC9E,MAAM,gBAAgB,IAAI,iBAAiB;EAW3C,MAAM,qBAAqB,KAAK,aAAA;EAMhC,MAAM,cAAc,sBAAsB,QAAQ,OAAO;GACvD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;EACjB,CAAC;EACD,MAAM,aAAa,sBAAsB,OAAO,QAAQ;GACtD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;EACjB,CAAC;EACD,MAAM,YACJ,YAAY,aACZ,gBAAgB,QAAQ,OAAO;GAC7B;GACA;GACA,WAAW;GACX;EACF,CAAC;EAyBH,MAAM,gBAAgB,UAAU,MAAM,CAAC,kBAAkB,WAAW;EAMpE,MAAM,UAAuB,CAAC,YAAY,aACtC,aACA,gBACE,cACA,YAAY,UACV,aACA;EACR,KAAK,KAAK;GACR,MAAM,IAAI;GACV,QAAQ,IAAI;GACZ,WAAW,IAAI;GACf;GACA;GACA,IAAI;IAAE,KAAK,YAAY;IAAK,MAAM,YAAY;GAAK;GACnD,mBAAmB,YAAY;GAC/B,SAAS,YAAY;GACrB,eAAe,YAAY;GAC3B;GACA,iBAAiB,YAAY;GAC7B,gBAAgB,YAAY;GAC5B;GACA;GACA;EACF,CAAC;CACH;CACA,MAAM,KAAK,KAAK,KAAK,MAAM,EAAE,CAAC,CAAC,CAAC,QAAQ,MAAM,IAAI,CAAC;CAEnD,OAAO;EAAE;EAAM,MADF,GAAG,SAAS,IAAI,KAAK,IAAI,GAAG,EAAE,IAAI;EAC1B,MAAM;GAAE,WAAW,IAAI,KAAK;GAAW,UAAU,IAAI,KAAK;EAAS;CAAE;AAC5F;;;;;;;;AASA,MAAa,gBAAiC,OAAO;CACnD,MAAM,oBAAoB,GAAG,KAAK,KAAK,QAAQ;EAC7C,MAAM,aAAa,GAAG;EACtB,QACE,GAAG,YAAY,cACV,SACD,GAAG,YAAY,aACZ,kBACA;EACT,QAAQ;GACN,WAAW,GAAG;GACd,QAAQ,GAAG;GACX,SAAS,GAAG;GACZ,GAAG,GAAG;GACN,aAAa,GAAG,UAAU;GAC1B,OAAO,GAAG,GAAG;GACb,QAAQ,GAAG,GAAG;GACd,mBAAmB,GAAG;GACtB,gBAAgB,GAAG;GACnB,SAAS,GAAG;GACZ,eAAe,GAAG;GAClB,gBAAgB,GAAG,UAAU;GAC7B,iBAAiB,GAAG,UAAU;GAC9B,YAAY,GAAG,UAAU;GACzB,eAAe,GAAG;GAClB,gBAAgB,GAAG;EACrB;CACF,EAAE;CAEF,MAAM,YAAY,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,WAAW;CACjE,MAAM,UAAU,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,UAAU;CAC9D,MAAM,WAAW,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,UAAU;CAE/D,IAAI;CACJ,MAAM,UAAoB,CAAC;CAC3B,IAAI,UAAU,SAAS,GAAG;EAIxB,WAAW;EACX,KAAK,MAAM,KAAK,WACd,QAAQ,KACN,cAAc,EAAE,KAAK,qCAAqC,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,MAAM,EAAE,eAAe,MAAM,EAAE,EAAE,EACjH;CAEJ,OAAO,IAAI,QAAQ,SAAS,GAAG;EAG7B,WAAW;EACX,KAAK,MAAM,KAAK,SACd,QAAQ,KACN,cAAc,EAAE,KAAK,eAAe,EAAE,EAAE,2DAC1C;CAEJ,OAAO,IAAI,SAAS,SAAS,GAAG;EAG9B,WAAW;EACX,QAAQ,KACN,+CAA+C,SAC5C,KACE,MACC,IAAI,EAAE,KAAK,KAAK,EAAE,GAAG,MAAM,IAAI,EAAE,GAAG,IAAI,QAAQ,CAAC,IAAI,EAAE,UAAU,KAAK,QAAQ,CAAC,EAAE,WAAW,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,EACpH,CAAC,CACA,KAAK,IAAI,EAAE,yBAChB;CACF,OAAO;EAGL,WAAW;EACX,QAAQ,KACN,0FACF;CACF;CAIA,MAAM,YAAY,GAAG,KAAK,MAAM,MAAM,EAAE,OAAO,SAAS,WAAW,KAAK,GAAG,KAAK;CAChF,OAAO;EAAE;EAAU;EAAS;EAAmB,OAAO,WAAW,UAAU;CAAO;AACpF;;;;;;AAiBA,SAAgB,uBACd,SAC4B;CAC5B,IAAI,QAAQ,WAAW,WAAW,GAChC,MAAM,IAAI,MAAM,uDAAuD;CAEzE,MAAM,SAAS,QAAQ,UAAU;CACjC,OAAO;EACL,MAAM,QAAQ,QAAQ;EACtB,MAAM,OAAO,KAA6D;GACxE,MAAM,KAAK,oBAAoB,KAAK,QAAQ,YAAY,OAAO;GAC/D,OAAO,OAAO,EAAE;EAClB;CACF;AACF"}
@@ -29,6 +29,23 @@ declare function aggregateLlm(spans: LlmSpan[]): {
29
29
  };
30
30
  /** Pick the outcome's failure class when present, else derive 'success' from run status. */
31
31
  declare function runFailureClass(run: Run): FailureClass;
32
+ /**
33
+ * Metrics `regressionView`, `correlationStudy`, and `calibrationCurve` can
34
+ * read from a run without a caller-supplied extractor. The type derives from
35
+ * this array, so a new metric cannot be declared without an extractor arm.
36
+ */
37
+ declare const RUN_METRICS: readonly ['score', 'overallScore', 'pass', 'durationMs', 'costUsd', 'inputTokens', 'outputTokens', 'failureClass'];
38
+ type RunMetric = (typeof RUN_METRICS)[number];
39
+ declare function isRunMetric(metric: string): metric is RunMetric;
40
+ /**
41
+ * The extractor for one built-in metric. `null` means this run carries no
42
+ * value for the metric — the caller drops that run from the sample.
43
+ *
44
+ * Throws `ValidationError` on a metric name this package does not define:
45
+ * an unrecognized name would otherwise read as "every run is missing this
46
+ * value" and produce an empty study instead of a refusal.
47
+ */
48
+ declare function runMetricExtractor(metric: string): (run: Run, store: TraceStore) => Promise<number | null>;
32
49
  //#endregion
33
- export { judgeSpans as a, runsForScenario as c, hasCapturedToolArgs as i, toolSpans as l, argHash as n, llmSpans as o, groupBy as r, runFailureClass as s, aggregateLlm as t };
34
- //# sourceMappingURL=query-DxPYqpmT.d.ts.map
50
+ export { groupBy as a, judgeSpans as c, runMetricExtractor as d, runsForScenario as f, argHash as i, llmSpans as l, RunMetric as n, hasCapturedToolArgs as o, toolSpans as p, aggregateLlm as r, isRunMetric as s, RUN_METRICS as t, runFailureClass as u };
51
+ //# sourceMappingURL=query-CwnHlu5p.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"query-CwnHlu5p.d.ts","names":[],"sources":["../src/trace/query.ts"],"mappings":";;;iBAesB,gBAAgB,OAAO,YAAY,qBAAqB,QAAQ;iBAIhE,SAAS,OAAO,YAAY,iBAAiB,QAAQ;iBAKrD,UACpB,OAAO,YACP,gBACA,oBACC,QAAQ;;iBAMW,WAAW,OAAO,YAAY,iBAAiB,QAAQ;;iBAM7D,QAAQ,GAAG,2BAA2B,OAAO,KAAK,MAAM,GAAG,MAAM,IAAI,IAAI,GAAG;;;;;;;;iBAqB5E,QAAQ;;iBAMR,oBAAoB,MAAM;;iBAK1B,aAAa,OAAO;EAClC;EACA;EACA;EACA;EACA;EACA;;;iBAuBc,gBAAgB,KAAK,MAAM;;;;;;cAY9B;KAWD,oBAAoB;iBAEhB,YAAY,iBAAiB,UAAU;;;;;;;;;iBAYvC,mBACd,kBACE,KAAK,KAAK,OAAO,eAAe"}
@@ -1,3 +1,4 @@
1
+ import { s as ValidationError } from "./errors-Dngq5h35.js";
1
2
  import { r as canonicalString } from "./canonical-IL-Bu-14.js";
2
3
  import { a as isToolSpan, i as isLlmSpan, r as isJudgeSpan } from "./schema-k6ZBftVv.js";
3
4
  //#region src/trace/query.ts
@@ -86,7 +87,48 @@ function runFailureClass(run) {
86
87
  if (run.status === "aborted") return "budget_exceeded";
87
88
  return "unknown";
88
89
  }
90
+ /**
91
+ * Metrics `regressionView`, `correlationStudy`, and `calibrationCurve` can
92
+ * read from a run without a caller-supplied extractor. The type derives from
93
+ * this array, so a new metric cannot be declared without an extractor arm.
94
+ */
95
+ const RUN_METRICS = [
96
+ "score",
97
+ "overallScore",
98
+ "pass",
99
+ "durationMs",
100
+ "costUsd",
101
+ "inputTokens",
102
+ "outputTokens",
103
+ "failureClass"
104
+ ];
105
+ function isRunMetric(metric) {
106
+ return RUN_METRICS.includes(metric);
107
+ }
108
+ /**
109
+ * The extractor for one built-in metric. `null` means this run carries no
110
+ * value for the metric — the caller drops that run from the sample.
111
+ *
112
+ * Throws `ValidationError` on a metric name this package does not define:
113
+ * an unrecognized name would otherwise read as "every run is missing this
114
+ * value" and produce an empty study instead of a refusal.
115
+ */
116
+ function runMetricExtractor(metric) {
117
+ if (!isRunMetric(metric)) throw new ValidationError(`unknown run metric '${metric}' — pass an \`extract\` function or use one of: ${RUN_METRICS.join(", ")}`);
118
+ return async (run, store) => {
119
+ switch (metric) {
120
+ case "score":
121
+ case "overallScore": return run.outcome?.score ?? null;
122
+ case "pass": return run.outcome?.pass === true ? 1 : 0;
123
+ case "durationMs": return run.endedAt && run.startedAt ? run.endedAt - run.startedAt : null;
124
+ case "costUsd": return aggregateLlm(await llmSpans(store, run.runId)).costUsd;
125
+ case "inputTokens": return aggregateLlm(await llmSpans(store, run.runId)).inputTokens;
126
+ case "outputTokens": return aggregateLlm(await llmSpans(store, run.runId)).outputTokens;
127
+ case "failureClass": return runFailureClass(run) === "success" ? 1 : 0;
128
+ }
129
+ };
130
+ }
89
131
  //#endregion
90
- export { judgeSpans as a, runsForScenario as c, hasCapturedToolArgs as i, toolSpans as l, argHash as n, llmSpans as o, groupBy as r, runFailureClass as s, aggregateLlm as t };
132
+ export { hasCapturedToolArgs as a, llmSpans as c, runsForScenario as d, toolSpans as f, groupBy as i, runFailureClass as l, aggregateLlm as n, isRunMetric as o, argHash as r, judgeSpans as s, RUN_METRICS as t, runMetricExtractor as u };
91
133
 
92
- //# sourceMappingURL=query-CHmMP42p.js.map
134
+ //# sourceMappingURL=query-_5g6re3_.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"query-_5g6re3_.js","names":[],"sources":["../src/trace/query.ts"],"sourcesContent":["/**\n * Typed query helpers over TraceStore.\n *\n * Not a full SQL engine — a minimal, composable set of operators that\n * cover the canned-pipeline use cases. For ad-hoc analytics, persist to\n * NDJSON and point DuckDB at it; the schema is stable so external SQL\n * tooling works out of the box.\n */\n\nimport { ValidationError } from '../errors'\nimport { canonicalString } from '../ledger-core/canonical'\nimport type { FailureClass, JudgeSpan, LlmSpan, Run, ToolSpan } from './schema'\nimport { isJudgeSpan, isLlmSpan, isToolSpan } from './schema'\nimport type { TraceStore } from './store'\n\nexport async function runsForScenario(store: TraceStore, scenarioId: string): Promise<Run[]> {\n return store.listRuns({ scenarioId })\n}\n\nexport async function llmSpans(store: TraceStore, runId?: string): Promise<LlmSpan[]> {\n const spans = await store.spans({ runId, kind: 'llm' })\n return spans.filter(isLlmSpan)\n}\n\nexport async function toolSpans(\n store: TraceStore,\n runId?: string,\n toolName?: string,\n): Promise<ToolSpan[]> {\n const spans = await store.spans({ runId, kind: 'tool', toolName })\n return spans.filter(isToolSpan)\n}\n\n/** Query judge-kind spans from the trace store, optionally scoped to a single run. */\nexport async function judgeSpans(store: TraceStore, runId?: string): Promise<JudgeSpan[]> {\n const spans = await store.spans({ runId, kind: 'judge' })\n return spans.filter(isJudgeSpan)\n}\n\n/** Group spans by any key selector. */\nexport function groupBy<T, K extends string | number>(items: T[], key: (t: T) => K): Map<K, T[]> {\n const map = new Map<K, T[]>()\n for (const item of items) {\n const k = key(item)\n let bucket = map.get(k)\n if (!bucket) {\n bucket = []\n map.set(k, bucket)\n }\n bucket.push(item)\n }\n return map\n}\n\n/**\n * Key tool arguments for de-duplication: RFC 8785 canonical JSON, so key\n * order cannot split one call into two keys. Uncaptured args (`undefined`)\n * key to the string `'undefined'` so every call still keys to a string; an\n * args value with no canonical JSON form (a nested `undefined`, a function, a\n * non-finite number) throws `LedgerCanonicalizationError`.\n */\nexport function argHash(args: unknown): string {\n if (args === undefined) return 'undefined'\n return canonicalString(args)\n}\n\n/** Whether argument-based comparisons are valid for this tool call. */\nexport function hasCapturedToolArgs(span: ToolSpan): boolean {\n return span.argsCaptured !== false\n}\n\n/** Sum an LLM-span array into aggregate token + cost. */\nexport function aggregateLlm(spans: LlmSpan[]): {\n inputTokens: number\n outputTokens: number\n cachedTokens: number\n cacheWriteTokens: number\n reasoningTokens: number\n costUsd: number\n} {\n return spans.reduce(\n (acc, s) => ({\n inputTokens: acc.inputTokens + (s.inputTokens ?? 0),\n outputTokens: acc.outputTokens + (s.outputTokens ?? 0),\n cachedTokens: acc.cachedTokens + (s.cachedTokens ?? 0),\n cacheWriteTokens: acc.cacheWriteTokens + (s.cacheWriteTokens ?? 0),\n reasoningTokens: acc.reasoningTokens + (s.reasoningTokens ?? 0),\n costUsd: acc.costUsd + (s.costUsd ?? 0),\n }),\n {\n inputTokens: 0,\n outputTokens: 0,\n cachedTokens: 0,\n cacheWriteTokens: 0,\n reasoningTokens: 0,\n costUsd: 0,\n },\n )\n}\n\n/** Pick the outcome's failure class when present, else derive 'success' from run status. */\nexport function runFailureClass(run: Run): FailureClass {\n if (run.outcome?.failureClass) return run.outcome.failureClass\n if (run.status === 'completed' && run.outcome?.pass !== false) return 'success'\n if (run.status === 'aborted') return 'budget_exceeded'\n return 'unknown'\n}\n\n/**\n * Metrics `regressionView`, `correlationStudy`, and `calibrationCurve` can\n * read from a run without a caller-supplied extractor. The type derives from\n * this array, so a new metric cannot be declared without an extractor arm.\n */\nexport const RUN_METRICS = [\n 'score',\n 'overallScore',\n 'pass',\n 'durationMs',\n 'costUsd',\n 'inputTokens',\n 'outputTokens',\n 'failureClass',\n] as const\n\nexport type RunMetric = (typeof RUN_METRICS)[number]\n\nexport function isRunMetric(metric: string): metric is RunMetric {\n return (RUN_METRICS as readonly string[]).includes(metric)\n}\n\n/**\n * The extractor for one built-in metric. `null` means this run carries no\n * value for the metric — the caller drops that run from the sample.\n *\n * Throws `ValidationError` on a metric name this package does not define:\n * an unrecognized name would otherwise read as \"every run is missing this\n * value\" and produce an empty study instead of a refusal.\n */\nexport function runMetricExtractor(\n metric: string,\n): (run: Run, store: TraceStore) => Promise<number | null> {\n if (!isRunMetric(metric)) {\n throw new ValidationError(\n `unknown run metric '${metric}' — pass an \\`extract\\` function or use one of: ${RUN_METRICS.join(', ')}`,\n )\n }\n return async (run, store) => {\n switch (metric) {\n case 'score':\n case 'overallScore':\n return run.outcome?.score ?? null\n case 'pass':\n return run.outcome?.pass === true ? 1 : 0\n case 'durationMs':\n return run.endedAt && run.startedAt ? run.endedAt - run.startedAt : null\n case 'costUsd':\n return aggregateLlm(await llmSpans(store, run.runId)).costUsd\n case 'inputTokens':\n return aggregateLlm(await llmSpans(store, run.runId)).inputTokens\n case 'outputTokens':\n return aggregateLlm(await llmSpans(store, run.runId)).outputTokens\n case 'failureClass':\n return runFailureClass(run) === 'success' ? 1 : 0\n }\n }\n}\n"],"mappings":";;;;;;;;;;;;AAeA,eAAsB,gBAAgB,OAAmB,YAAoC;CAC3F,OAAO,MAAM,SAAS,EAAE,WAAW,CAAC;AACtC;AAEA,eAAsB,SAAS,OAAmB,OAAoC;CAEpF,QAAO,MADa,MAAM,MAAM;EAAE;EAAO,MAAM;CAAM,CAAC,EAAA,CACzC,OAAO,SAAS;AAC/B;AAEA,eAAsB,UACpB,OACA,OACA,UACqB;CAErB,QAAO,MADa,MAAM,MAAM;EAAE;EAAO,MAAM;EAAQ;CAAS,CAAC,EAAA,CACpD,OAAO,UAAU;AAChC;;AAGA,eAAsB,WAAW,OAAmB,OAAsC;CAExF,QAAO,MADa,MAAM,MAAM;EAAE;EAAO,MAAM;CAAQ,CAAC,EAAA,CAC3C,OAAO,WAAW;AACjC;;AAGA,SAAgB,QAAsC,OAAY,KAA+B;CAC/F,MAAM,sBAAM,IAAI,IAAY;CAC5B,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,IAAI,IAAI,IAAI;EAClB,IAAI,SAAS,IAAI,IAAI,CAAC;EACtB,IAAI,CAAC,QAAQ;GACX,SAAS,CAAC;GACV,IAAI,IAAI,GAAG,MAAM;EACnB;EACA,OAAO,KAAK,IAAI;CAClB;CACA,OAAO;AACT;;;;;;;;AASA,SAAgB,QAAQ,MAAuB;CAC7C,IAAI,SAAS,KAAA,GAAW,OAAO;CAC/B,OAAO,gBAAgB,IAAI;AAC7B;;AAGA,SAAgB,oBAAoB,MAAyB;CAC3D,OAAO,KAAK,iBAAiB;AAC/B;;AAGA,SAAgB,aAAa,OAO3B;CACA,OAAO,MAAM,QACV,KAAK,OAAO;EACX,aAAa,IAAI,eAAe,EAAE,eAAe;EACjD,cAAc,IAAI,gBAAgB,EAAE,gBAAgB;EACpD,cAAc,IAAI,gBAAgB,EAAE,gBAAgB;EACpD,kBAAkB,IAAI,oBAAoB,EAAE,oBAAoB;EAChE,iBAAiB,IAAI,mBAAmB,EAAE,mBAAmB;EAC7D,SAAS,IAAI,WAAW,EAAE,WAAW;CACvC,IACA;EACE,aAAa;EACb,cAAc;EACd,cAAc;EACd,kBAAkB;EAClB,iBAAiB;EACjB,SAAS;CACX,CACF;AACF;;AAGA,SAAgB,gBAAgB,KAAwB;CACtD,IAAI,IAAI,SAAS,cAAc,OAAO,IAAI,QAAQ;CAClD,IAAI,IAAI,WAAW,eAAe,IAAI,SAAS,SAAS,OAAO,OAAO;CACtE,IAAI,IAAI,WAAW,WAAW,OAAO;CACrC,OAAO;AACT;;;;;;AAOA,MAAa,cAAc;CACzB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAIA,SAAgB,YAAY,QAAqC;CAC/D,OAAQ,YAAkC,SAAS,MAAM;AAC3D;;;;;;;;;AAUA,SAAgB,mBACd,QACyD;CACzD,IAAI,CAAC,YAAY,MAAM,GACrB,MAAM,IAAI,gBACR,uBAAuB,OAAO,kDAAkD,YAAY,KAAK,IAAI,GACvG;CAEF,OAAO,OAAO,KAAK,UAAU;EAC3B,QAAQ,QAAR;GACE,KAAK;GACL,KAAK,gBACH,OAAO,IAAI,SAAS,SAAS;GAC/B,KAAK,QACH,OAAO,IAAI,SAAS,SAAS,OAAO,IAAI;GAC1C,KAAK,cACH,OAAO,IAAI,WAAW,IAAI,YAAY,IAAI,UAAU,IAAI,YAAY;GACtE,KAAK,WACH,OAAO,aAAa,MAAM,SAAS,OAAO,IAAI,KAAK,CAAC,CAAC,CAAC;GACxD,KAAK,eACH,OAAO,aAAa,MAAM,SAAS,OAAO,IAAI,KAAK,CAAC,CAAC,CAAC;GACxD,KAAK,gBACH,OAAO,aAAa,MAAM,SAAS,OAAO,IAAI,KAAK,CAAC,CAAC,CAAC;GACxD,KAAK,gBACH,OAAO,gBAAgB,GAAG,MAAM,YAAY,IAAI;EACpD;CACF;AACF"}
@@ -0,0 +1,21 @@
1
+ import { s as ValidationError } from "./errors-Dngq5h35.js";
2
+ //#region src/statistics/random.ts
3
+ /** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
4
+ * cryptographic. Exported so e-process shuffles and bootstrap resampling
5
+ * share ONE PRNG implementation. Every distinct 32-bit seed gives a distinct
6
+ * stream, including 0. */
7
+ function mulberry32(seed) {
8
+ if (!Number.isFinite(seed)) throw new ValidationError(`mulberry32: seed must be a finite number, got ${seed}`);
9
+ let s = seed | 0;
10
+ return () => {
11
+ s = s + 1831565813 | 0;
12
+ let t = s;
13
+ t = Math.imul(t ^ t >>> 15, t | 1);
14
+ t ^= t + Math.imul(t ^ t >>> 7, t | 61);
15
+ return ((t ^ t >>> 14) >>> 0) / 4294967296;
16
+ };
17
+ }
18
+ //#endregion
19
+ export { mulberry32 as t };
20
+
21
+ //# sourceMappingURL=random-Dn5fPWkt.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"random-Dn5fPWkt.js","names":[],"sources":["../src/statistics/random.ts"],"sourcesContent":["import { ValidationError } from '../errors'\n\n/** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not\n * cryptographic. Exported so e-process shuffles and bootstrap resampling\n * share ONE PRNG implementation. Every distinct 32-bit seed gives a distinct\n * stream, including 0. */\nexport function mulberry32(seed: number): () => number {\n if (!Number.isFinite(seed)) {\n throw new ValidationError(`mulberry32: seed must be a finite number, got ${seed}`)\n }\n let s = seed | 0\n return () => {\n s = (s + 0x6d2b79f5) | 0\n let t = s\n t = Math.imul(t ^ (t >>> 15), t | 1)\n t ^= t + Math.imul(t ^ (t >>> 7), t | 61)\n return ((t ^ (t >>> 14)) >>> 0) / 4294967296\n }\n}\n"],"mappings":";;;;;;AAMA,SAAgB,WAAW,MAA4B;CACrD,IAAI,CAAC,OAAO,SAAS,IAAI,GACvB,MAAM,IAAI,gBAAgB,iDAAiD,MAAM;CAEnF,IAAI,IAAI,OAAO;CACf,aAAa;EACX,IAAK,IAAI,aAAc;EACvB,IAAI,IAAI;EACR,IAAI,KAAK,KAAK,IAAK,MAAM,IAAK,IAAI,CAAC;EACnC,KAAK,IAAI,KAAK,KAAK,IAAK,MAAM,GAAI,IAAI,EAAE;EACxC,SAAS,IAAK,MAAM,QAAS,KAAK;CACpC;AACF"}
@@ -0,0 +1,17 @@
1
+ //#region src/record-id.ts
2
+ /**
3
+ * The ONE identifier mint for records this package emits — run ids, span ids,
4
+ * event ids, session ids.
5
+ *
6
+ * `globalThis.crypto.randomUUID` is available on every runtime this package
7
+ * supports (`engines.node >= 20`, and every browser that ships Web Crypto), so
8
+ * there is no weaker time-plus-`Math.random` path to fall back to. A private
9
+ * fallback is how two records end up with ids minted by different rules.
10
+ */
11
+ function newRecordId() {
12
+ return globalThis.crypto.randomUUID();
13
+ }
14
+ //#endregion
15
+ export { newRecordId as t };
16
+
17
+ //# sourceMappingURL=record-id-DUgsK5qp.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"record-id-DUgsK5qp.js","names":[],"sources":["../src/record-id.ts"],"sourcesContent":["/**\n * The ONE identifier mint for records this package emits — run ids, span ids,\n * event ids, session ids.\n *\n * `globalThis.crypto.randomUUID` is available on every runtime this package\n * supports (`engines.node >= 20`, and every browser that ships Web Crypto), so\n * there is no weaker time-plus-`Math.random` path to fall back to. A private\n * fallback is how two records end up with ids minted by different rules.\n */\nexport function newRecordId(): string {\n return globalThis.crypto.randomUUID()\n}\n"],"mappings":";;;;;;;;;;AASA,SAAgB,cAAsB;CACpC,OAAO,WAAW,OAAO,WAAW;AACtC"}
@@ -1,5 +1,5 @@
1
1
  import { c as VerificationError, s as ValidationError } from "./errors-Dngq5h35.js";
2
- import { c as mulberry32 } from "./internal-BDHPCnjk.js";
2
+ import { t as mulberry32 } from "./random-Dn5fPWkt.js";
3
3
  import { t as FAILURE_CLASSES } from "./schema-k6ZBftVv.js";
4
4
  import { r as observedSplitScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
5
5
  import { c as validateRunRecord } from "./run-record-BC0ebuRP.js";
@@ -523,4 +523,4 @@ function fmt(x) {
523
523
  //#endregion
524
524
  export { judgeReplayGate as i, evaluateReleaseConfidence as n, bootstrapCi as r, assertReleaseConfidence as t };
525
525
 
526
- //# sourceMappingURL=release-confidence-DKfD2RYU.js.map
526
+ //# sourceMappingURL=release-confidence-nGDJiiwc.js.map