@tangle-network/agent-eval 0.144.6 → 0.144.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (240) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/analyst/index.d.ts +470 -88
  6. package/dist/analyst/index.d.ts.map +1 -1
  7. package/dist/analyst/index.js +24 -5
  8. package/dist/analyst/index.js.map +1 -1
  9. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  10. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  11. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  12. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  13. package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BKENp2s5.js} +662 -605
  14. package/dist/benchmark-command-BKENp2s5.js.map +1 -0
  15. package/dist/benchmarks/index.d.ts +1 -1
  16. package/dist/benchmarks/index.js +1 -1
  17. package/dist/{benchmarks-BEOkuvIg.js → benchmarks-BlPmjd88.js} +4 -4
  18. package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-BlPmjd88.js.map} +1 -1
  19. package/dist/campaign/index.d.ts +7 -5
  20. package/dist/campaign/index.js +5 -3
  21. package/dist/{campaign-CXsdyym7.js → campaign--HVSuvV0.js} +17 -301
  22. package/dist/campaign--HVSuvV0.js.map +1 -0
  23. package/dist/cli.js +2 -2
  24. package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
  25. package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  26. package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
  27. package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  28. package/dist/contract/index.d.ts +9 -8
  29. package/dist/contract/index.d.ts.map +1 -1
  30. package/dist/contract/index.js +7 -6
  31. package/dist/contract/index.js.map +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/control.js +1 -1
  34. package/dist/counterfactual-CWPTrMH7.js +126 -0
  35. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  36. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  37. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  38. package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  39. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  40. package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
  41. package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  42. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  43. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  44. package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
  45. package/dist/engine-nB64f48I.d.ts.map +1 -0
  46. package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
  47. package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  48. package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  49. package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  50. package/dist/exec-BLtYZdWo.js +49 -0
  51. package/dist/exec-BLtYZdWo.js.map +1 -0
  52. package/dist/experiment/index.d.ts +802 -0
  53. package/dist/experiment/index.d.ts.map +1 -0
  54. package/dist/experiment/index.js +1108 -0
  55. package/dist/experiment/index.js.map +1 -0
  56. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  57. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  58. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  59. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  60. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
  61. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  62. package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
  63. package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  64. package/dist/fuzz.d.ts +1 -1
  65. package/dist/hosted/index.d.ts +3 -3
  66. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  67. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  68. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  69. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  70. package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
  71. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  72. package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
  73. package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  74. package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
  75. package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  76. package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
  77. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  78. package/dist/index.d.ts +214 -404
  79. package/dist/index.d.ts.map +1 -1
  80. package/dist/index.js +233 -649
  81. package/dist/index.js.map +1 -1
  82. package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
  83. package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  84. package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
  85. package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  86. package/dist/integrity-MLzHOfV9.js +141 -0
  87. package/dist/integrity-MLzHOfV9.js.map +1 -0
  88. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  89. package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
  90. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  91. package/dist/matrix/index.d.ts +2 -2
  92. package/dist/meta-eval/index.d.ts +1 -1
  93. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  94. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  95. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  96. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  97. package/dist/multishot/index.d.ts +2 -2
  98. package/dist/openapi.json +1 -1
  99. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  100. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  101. package/dist/pipelines/index.d.ts +2 -1
  102. package/dist/pipelines/index.d.ts.map +1 -1
  103. package/dist/pre-registration-DakwTRXk.js +96 -0
  104. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  105. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  106. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  107. package/dist/prime-protocol-BfSalTfR.js +453 -0
  108. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  109. package/dist/profile-cell.js +242 -1
  110. package/dist/profile-cell.js.map +1 -0
  111. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  112. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  113. package/dist/promotion-policy-CrLrmys8.js +682 -0
  114. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  115. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  116. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  117. package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
  118. package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  119. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  120. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  121. package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
  122. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  123. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  124. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  125. package/dist/reporting.d.ts +3 -3
  126. package/dist/reporting.js +2 -2
  127. package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
  128. package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  130. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  131. package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
  132. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  133. package/dist/rl.d.ts +17 -7
  134. package/dist/rl.d.ts.map +1 -1
  135. package/dist/rl.js +16 -7
  136. package/dist/rl.js.map +1 -1
  137. package/dist/rollout/index.d.ts +2 -2
  138. package/dist/rollout/index.js +2 -2
  139. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  140. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  141. package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
  142. package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  143. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  144. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  145. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  146. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  147. package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  148. package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  149. package/dist/sequential-D-BLJBKU.js +299 -0
  150. package/dist/sequential-D-BLJBKU.js.map +1 -0
  151. package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
  152. package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
  153. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  154. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  155. package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
  156. package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  157. package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
  158. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  159. package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CkaI2ly4.js} +19 -654
  160. package/dist/skillopt-optimization-method-CkaI2ly4.js.map +1 -0
  161. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  162. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  163. package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  164. package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  165. package/dist/steps-BArUxhna.d.ts +51 -0
  166. package/dist/steps-BArUxhna.d.ts.map +1 -0
  167. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  168. package/dist/store-DNe_Uv1Q.js.map +1 -0
  169. package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
  170. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  171. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  172. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  173. package/dist/supervisor-run/index.d.ts +2 -2
  174. package/dist/supervisor-run/index.js +1 -1
  175. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  176. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  177. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  178. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  179. package/dist/trace-repair/index.d.ts +2102 -0
  180. package/dist/trace-repair/index.d.ts.map +1 -0
  181. package/dist/trace-repair/index.js +3878 -0
  182. package/dist/trace-repair/index.js.map +1 -0
  183. package/dist/traces.d.ts +6 -6
  184. package/dist/traces.js +3 -2
  185. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  186. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  187. package/dist/trajectory-replay/index.d.ts +781 -0
  188. package/dist/trajectory-replay/index.d.ts.map +1 -0
  189. package/dist/trajectory-replay/index.js +2103 -0
  190. package/dist/trajectory-replay/index.js.map +1 -0
  191. package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
  192. package/dist/types-D216SgwM.d.ts.map +1 -0
  193. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  194. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  195. package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
  196. package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  197. package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
  198. package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  199. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  200. package/dist/verdict-DExhxfgR.d.ts +201 -0
  201. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  202. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  203. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  204. package/dist/wire/index.d.ts +2 -2
  205. package/dist/wire/index.js +1 -1
  206. package/docs/campaign-proposers.md +5 -0
  207. package/docs/charter.md +112 -0
  208. package/docs/experiment.md +104 -0
  209. package/docs/prime-analyst.md +1 -0
  210. package/docs/trace-analysis.md +26 -0
  211. package/docs/trace-repair-admission.md +194 -0
  212. package/docs/trace-repair-analyst-arms.md +121 -0
  213. package/docs/trace-repair-continuation.md +107 -0
  214. package/docs/trace-repair-grader.md +163 -0
  215. package/docs/trajectory-replay.md +110 -0
  216. package/docs/verification-strategies.md +103 -0
  217. package/package.json +19 -2
  218. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  219. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  220. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  221. package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
  222. package/dist/benchmark-command-CQd78YHt.js.map +0 -1
  223. package/dist/campaign-CXsdyym7.js.map +0 -1
  224. package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
  225. package/dist/index-4XwggC10.d.ts.map +0 -1
  226. package/dist/index-BIL5vxxt.d.ts.map +0 -1
  227. package/dist/integrity-fdt8XPAv.js.map +0 -1
  228. package/dist/llm-client-Dv5BiKLE.js.map +0 -1
  229. package/dist/replay-Krvb114g.d.ts.map +0 -1
  230. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  231. package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
  232. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  233. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  234. package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
  235. package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
  236. package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
  237. package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
  238. package/dist/types-XMVEdrE_.d.ts.map +0 -1
  239. package/dist/verdict-Dps8_okt.d.ts +0 -37
  240. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -1,7 +1,7 @@
1
1
  import { S as PaidCallResult, T as RunPaidCallInput, a as CostChannel, c as CostLedgerHandle, f as CostLedgerSummary, p as CostProvenance } from "./cost-ledger-Bv_e8XHY.js";
2
2
  import { u as RunTokenUsage } from "./run-record-DdSa93_W.js";
3
- import { U as LlmCallMetadata } from "./types-XMVEdrE_.js";
4
- import { m as ProposalFinding } from "./types-BhP9q0Fq.js";
3
+ import { U as LlmCallMetadata } from "./types-D216SgwM.js";
4
+ import { _ as ProposalFinding } from "./types-DF_Udrp-.js";
5
5
  //#region src/campaign/types.d.ts
6
6
  /** Stable identifier + kind tag for any scenario. Consumers
7
7
  * extend with their per-domain payload (persona, task, requirement, ...). */
@@ -624,4 +624,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
624
624
  }
625
625
  //#endregion
626
626
  export { LabeledScenarioWrite as A, ScoredSurfaceOutcome as B, JudgeDimension as C, LabeledScenarioSampleArgs as D, LabeledScenarioRecord as E, ProposeContext as F, labelTrustRank as G, SurfaceProposer as H, ProposedCandidate as I, RedactionStatus as L, OptimizerConfig as M, ParetoParent as N, LabeledScenarioSource as O, ProposalTrackContext as P, Scenario as R, JudgeConfig as S, LabelTrust as T, TraceSpan as U, SessionScript as V, isProposedCandidate as W, GateDecision as _, CampaignResult as a, GenerationRecord as b, CampaignTraceWriter as c, DispatchContext as d, DispatchFn as f, GateContribution as g, GateContext as h, CampaignCostMeter as i, MutableSurface as j, LabeledScenarioStore as k, CodeSurface as l, GateCheckStatus as m, CampaignArtifactWriter as n, CampaignScenarioIdentity as o, Gate as p, CampaignCellResult as r, CampaignTokenUsage as s, CampaignAggregates as t, ComponentSurface as u, GateResult as v, JudgeScore as w, JudgeAggregate as x, GenerationCandidate as y, ScenarioAggregate as z };
627
- //# sourceMappingURL=types-DOZyvsFU.d.ts.map
627
+ //# sourceMappingURL=types-DYuNHo9R.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"types-DOZyvsFU.d.ts","names":[],"sources":["../src/campaign/types.ts"],"mappings":";;;;;;;UAgCiB;EACf;EACA;EACA;;;;;EAKA;;;UAIe,iCAAiC,KAAK;EACrD;;;;;UAMe;EACf;EACA;EACA;EACA;EACA,QAAQ;EACR,OAAO;EACP,WAAW;EACX,MAAM;;EAEN;;EAEA;;;;;;;;EAQA;;;;KAKU,WAAW,kBAAkB,UAAU,cACjD,UAAU,WACV,KAAK,oBACF,QAAQ;;;;UAOI,cAAc,WAAW;EACxC;EACA;EACA;;EAEA;;;EAGA,sBAAsB,UAAU,WAAW,sBAAsB,UAAU,cAAc;;UAK1E;;EAEf;;EAEA;;;;;;;;;UAUe,YAAY,WAAW,kBAAkB,WAAW;EACnE;EACA,YAAY;;;;EAIZ;;;EAGA,MAAM;IACJ,UAAU;IACV,UAAU;IACV,QAAQ;;IAER,aAAa;IACb;IACA,WAAW;MACT,aAAa,QAAQ;EACzB,aAAa,UAAU;;;;;;;;;;;UAYR;EACf,YAAY;EACZ;EACA;;EAEA,UAAU;;;;EAIV;;;EAGA;;EAEA;;EAEA,WAAW,eAAe;;;;;;;UAUX;WACN;;;WAGA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;;WAGA;aACE;aACA;aACA;;;WAGF;;;UAIM;WACN;WACA,YAAY,SAAS;;;;;;;;;;;KAYpB,0BAA0B,mBAAmB;;;;;;;UAQxC;EACf,SAAS;;EAET;;;;EAIA;;;;iBAKc,oBACd,OAAO,iBAAiB,oBACvB,SAAS;;;;;;;;;UAkBK;EACf,SAAS;EACT;;;EAGA,YAAY;;;EAGZ;;EAEA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EACA;;;;;UAMe;;;EAGf;;EAEA;EACA;EACA;EACA,YAAY;;;;;EAKZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;EAC1E;IACE;IACA;;;;;UAMa,eAAe,YAAY;WACjC,gBAAgB;WAChB,SAAS,cAAc;WACvB,UAAU,cAAc;;WAExB;WACA;WACA,QAAQ;;WAER,QAAQ;;;WAGR,kBAAkB;;;WAGlB,mBAAmB;;;;WAInB;;;;;;;WAOA,gBAAgB,cAAc;;WAE9B,aAAa;WACb;;;;;;;;;;;;;;UAeM,gBAAgB,YAAY;EAC3C;;;;;EAKA,QAAQ,KAAK,eAAe,aAAa,QAAQ,MAAM,iBAAiB;;;EAGxE,QAAQ;IAAQ,SAAS,cAAc;;IAAwB;IAAe;;;UAG/D;EACf;EACA;EACA,mBAAmB,qBAAqB;;UAGzB,wBAAwB;EACvC,UAAU;;;KAMA;;KAGA;UAEK;EACf;EACA,QAAQ;EACR;;UAGe,YAAY,WAAW,kBAAkB;EACxD,oBAAoB,YAAY;EAChC,oBAAoB,YAAY;;EAEhC,aAAa,YAAY,eAAe;;;;;EAKxC,sBAAsB,YAAY,eAAe;;;;;;;EAOjD,yBAAyB,YAAY,eAAe;;;EAGpD,uBAAuB,YAAY;EACnC,WAAW;EACX;IAAQ;IAAmB;;;EAE3B,aAAa;EACb;EACA,QAAQ;;UAGO;EACf,UAAU;EACV;EACA,mBAAmB;EACnB;;;UAIe,KAAK,qBAAqB,kBAAkB,WAAW;EACtE;EACA,OAAO,KAAK,YAAY,WAAW,aAAa,QAAQ;;;;UAOzC;EACf,KAAK,cAAc,aAAa,0BAA0B;EAC1D,SAAS;;UAGM;EACf,IAAI,aAAa;EACjB,aAAa,aAAa;;;;UAKX;EACf,MAAM,cAAc,kBAAkB,aAAa;EACnD,UAAU,cAAc,iBAAiB;;;;;;KAO/B,qBAAqB;;;;;UAMhB;;EAEf,YAAY,GACV,OAAO,KAAK,iBAAiB;IAC3B,UAAU;MAEX,QAAQ,eAAe;;;;;KAQhB;KAOA;;;;;;;;;;;;;;;KAgBA;;iBASI,eAAe,OAAO;;;;UAOrB,qBAAqB,kBAAkB,WAAW,UAAU;EAC3E,UAAU;EACV,UAAU;EACV,aAAa,eAAe;EAC5B,QAAQ;EACR;EACA;EACA,iBAAiB;;;;;EAKjB,aAAa;;EAEb;;UAGe,sBAAsB,kBAAkB,WAAW,UAAU,6BACpE,qBAAqB,WAAW;;EAExC;;;EAGA;;UAGe;EACf;;EAEA;;;;EAIA;EACA;IACE;IACA,SAAS,wBAAwB;IACjC;IACA;;;;;IAKA,WAAW;;;UAIE;EACf,QAAQ,OAAO,uBAAuB;EACtC,OAAO,MAAM,4BAA4B,QAAQ;EACjD,QAAQ;IACN;IACA;IACA,UAAU;;;IAGV,SAAS,OAAO;;;UAMH,mBAAmB;;;EAGlC;EACA;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa,eAAe;;;EAG5B;;EAEA,gBAAgB;;EAEhB;;;EAGA,YAAY;;;EAGZ;;;EAGA;EACA;EACA;EACA;;EAEA;;EAEA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,YAAY;EACZ;;;;;;UAOe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;EAIA;;;;EAIA;IACE;IACA;IACA,iBAAiB;MAAQ;MAAgB;;;;;EAI3C,YAAY;;;;;;;;;;EAUZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;;;EAG1E;;;;EAIA;;UAGe;EACf,SAAS,eAAe;EACxB,YAAY,eAAe;;EAE3B,MAAM;;EAEN;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,eAAe,qBAAqB,kBAAkB,WAAW;;EAEhF;;EAEA;EACA;;EAEA;EACA;EACA;EACA;EACA,OAAO,MAAM,mBAAmB;EAChC,YAAY;EACZ;IACE,aAAa;IACb;;EAEF,OAAO;EACP;EACA;EACA,iBAAiB;;;EAGjB,WAAW,MAAM,2BAA2B,KAAK"}
1
+ {"version":3,"file":"types-DYuNHo9R.d.ts","names":[],"sources":["../src/campaign/types.ts"],"mappings":";;;;;;;UAgCiB;EACf;EACA;EACA;;;;;EAKA;;;UAIe,iCAAiC,KAAK;EACrD;;;;;UAMe;EACf;EACA;EACA;EACA;EACA,QAAQ;EACR,OAAO;EACP,WAAW;EACX,MAAM;;EAEN;;EAEA;;;;;;;;EAQA;;;;KAKU,WAAW,kBAAkB,UAAU,cACjD,UAAU,WACV,KAAK,oBACF,QAAQ;;;;UAOI,cAAc,WAAW;EACxC;EACA;EACA;;EAEA;;;EAGA,sBAAsB,UAAU,WAAW,sBAAsB,UAAU,cAAc;;UAK1E;;EAEf;;EAEA;;;;;;;;;UAUe,YAAY,WAAW,kBAAkB,WAAW;EACnE;EACA,YAAY;;;;EAIZ;;;EAGA,MAAM;IACJ,UAAU;IACV,UAAU;IACV,QAAQ;;IAER,aAAa;IACb;IACA,WAAW;MACT,aAAa,QAAQ;EACzB,aAAa,UAAU;;;;;;;;;;;UAYR;EACf,YAAY;EACZ;EACA;;EAEA,UAAU;;;;EAIV;;;EAGA;;EAEA;;EAEA,WAAW,eAAe;;;;;;;UAUX;WACN;;;WAGA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;;WAGA;aACE;aACA;aACA;;;WAGF;;;UAIM;WACN;WACA,YAAY,SAAS;;;;;;;;;;;KAYpB,0BAA0B,mBAAmB;;;;;;;UAQxC;EACf,SAAS;;EAET;;;;EAIA;;;;iBAKc,oBACd,OAAO,iBAAiB,oBACvB,SAAS;;;;;;;;;UAkBK;EACf,SAAS;EACT;;;EAGA,YAAY;;;EAGZ;;EAEA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EACA;;;;;UAMe;;;EAGf;;EAEA;EACA;EACA;EACA,YAAY;;;;;EAKZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;EAC1E;IACE;IACA;;;;;UAMa,eAAe,YAAY;WACjC,gBAAgB;WAChB,SAAS,cAAc;WACvB,UAAU,cAAc;;WAExB;WACA;WACA,QAAQ;;WAER,QAAQ;;;WAGR,kBAAkB;;;WAGlB,mBAAmB;;;;WAInB;;;;;;;WAOA,gBAAgB,cAAc;;WAE9B,aAAa;WACb;;;;;;;;;;;;;;UAeM,gBAAgB,YAAY;EAC3C;;;;;EAKA,QAAQ,KAAK,eAAe,aAAa,QAAQ,MAAM,iBAAiB;;;EAGxE,QAAQ;IAAQ,SAAS,cAAc;;IAAwB;IAAe;;;UAG/D;EACf;EACA;EACA,mBAAmB,qBAAqB;;UAGzB,wBAAwB;EACvC,UAAU;;;KAMA;;KAGA;UAEK;EACf;EACA,QAAQ;EACR;;UAGe,YAAY,WAAW,kBAAkB;EACxD,oBAAoB,YAAY;EAChC,oBAAoB,YAAY;;EAEhC,aAAa,YAAY,eAAe;;;;;EAKxC,sBAAsB,YAAY,eAAe;;;;;;;EAOjD,yBAAyB,YAAY,eAAe;;;EAGpD,uBAAuB,YAAY;EACnC,WAAW;EACX;IAAQ;IAAmB;;;EAE3B,aAAa;EACb;EACA,QAAQ;;UAGO;EACf,UAAU;EACV;EACA,mBAAmB;EACnB;;;UAIe,KAAK,qBAAqB,kBAAkB,WAAW;EACtE;EACA,OAAO,KAAK,YAAY,WAAW,aAAa,QAAQ;;;;UAOzC;EACf,KAAK,cAAc,aAAa,0BAA0B;EAC1D,SAAS;;UAGM;EACf,IAAI,aAAa;EACjB,aAAa,aAAa;;;;UAKX;EACf,MAAM,cAAc,kBAAkB,aAAa;EACnD,UAAU,cAAc,iBAAiB;;;;;;KAO/B,qBAAqB;;;;;UAMhB;;EAEf,YAAY,GACV,OAAO,KAAK,iBAAiB;IAC3B,UAAU;MAEX,QAAQ,eAAe;;;;;KAQhB;KAOA;;;;;;;;;;;;;;;KAgBA;;iBASI,eAAe,OAAO;;;;UAOrB,qBAAqB,kBAAkB,WAAW,UAAU;EAC3E,UAAU;EACV,UAAU;EACV,aAAa,eAAe;EAC5B,QAAQ;EACR;EACA;EACA,iBAAiB;;;;;EAKjB,aAAa;;EAEb;;UAGe,sBAAsB,kBAAkB,WAAW,UAAU,6BACpE,qBAAqB,WAAW;;EAExC;;;EAGA;;UAGe;EACf;;EAEA;;;;EAIA;EACA;IACE;IACA,SAAS,wBAAwB;IACjC;IACA;;;;;IAKA,WAAW;;;UAIE;EACf,QAAQ,OAAO,uBAAuB;EACtC,OAAO,MAAM,4BAA4B,QAAQ;EACjD,QAAQ;IACN;IACA;IACA,UAAU;;;IAGV,SAAS,OAAO;;;UAMH,mBAAmB;;;EAGlC;EACA;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa,eAAe;;;EAG5B;;EAEA,gBAAgB;;EAEhB;;;EAGA,YAAY;;;EAGZ;;;EAGA;EACA;EACA;EACA;;EAEA;;EAEA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,YAAY;EACZ;;;;;;UAOe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;EAIA;;;;EAIA;IACE;IACA;IACA,iBAAiB;MAAQ;MAAgB;;;;;EAI3C,YAAY;;;;;;;;;;EAUZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;;;EAG1E;;;;EAIA;;UAGe;EACf,SAAS,eAAe;EACxB,YAAY,eAAe;;EAE3B,MAAM;;EAEN;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,eAAe,qBAAqB,kBAAkB,WAAW;;EAEhF;;EAEA;EACA;;EAEA;EACA;EACA;EACA;EACA,OAAO,MAAM,mBAAmB;EAChC,YAAY;EACZ;IACE,aAAa;IACb;;EAEF,OAAO;EACP;EACA;EACA,iBAAiB;;;EAGjB,WAAW,MAAM,2BAA2B,KAAK"}
@@ -1 +1 @@
1
- {"version":3,"file":"usage-receipt-EVI8B8Xu.js","names":[],"sources":["../src/analyst/types.ts","../src/analyst/usage-receipt.ts"],"sourcesContent":["/**\n * Analyst contract — the missing orchestration layer over agent-eval's\n * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,\n * SemanticConceptJudge, JudgeFn, ...).\n *\n * Each existing primitive returns its own output shape. The Analyst\n * contract is the single envelope every primitive lifts into, so a\n * registry can run N analysts against a run and a single renderer can\n * compose findings without knowing which analyzer produced them.\n *\n * The contract is intentionally domain-agnostic: nothing here knows\n * about code, voice, RAG, or any particular agent stack. Analysts\n * declare what INPUT KIND they need (a trace store, an artifact dir,\n * a RunRecord, a JudgeInput, or `custom`), and the registry routes\n * the matching input from `AnalystRunInputs`.\n */\n\nimport { createHash } from 'node:crypto'\nimport type { CostLedgerHandle } from '../cost-ledger'\nimport type { RunCostProvenance, RunRecord, RunTokenUsage } from '../run-record'\nimport type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { JudgeInput } from '../types'\nimport type { ChatClient } from './chat-client'\n\n/**\n * Unified envelope every analyst emits. Schema-versioned so renderers\n * and time-series diffs survive future field additions.\n */\nexport interface AnalystFinding {\n schema_version: '1.0.0'\n /**\n * Stable hash over identity-defining fields (analyst_id + canonical\n * claim + area + optional subject). Two findings from two runs that\n * \"are the same finding\" share this id — that's what `diffFindings`\n * uses to compute appeared/disappeared sets across runs.\n */\n finding_id: string\n analyst_id: string\n produced_at: string\n severity: AnalystSeverity\n /**\n * Coarse classification. Renderers group by this. Free-form so\n * domain-specific analysts can introduce categories without a\n * schema change ('agent-reasoning', 'verification', 'cost',\n * 'tool-use', 'safety', 'latency', 'data-quality', ...).\n */\n area: string\n claim: string\n rationale?: string\n evidence_refs: EvidenceRef[]\n recommended_action?: string\n validation_plan?: string\n /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */\n confidence: number\n /**\n * Optional subject the finding is about — leaf id, agent id, request\n * id. Included in finding_id when present so per-subject findings\n * diff cleanly across runs.\n */\n subject?: string\n /** True when this finding was lifted from a judge result rather than observed\n * directly in a trace or artifact. Descriptive only: proposal access is\n * controlled by `ProposalFinding.proposal_origin`. */\n derived_from_judge?: boolean\n /** Analyst-private extras; renderers ignore unless they know the analyst. */\n metadata?: Record<string, unknown>\n}\n\nexport type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info'\n\n/** Data sources that candidate generation may intentionally learn from. */\nexport type ProposalFindingOrigin = 'search' | 'production'\n\n/** A finding explicitly admitted as candidate-generation input. */\nexport type ProposalFinding = AnalystFinding & {\n readonly proposal_origin: ProposalFindingOrigin\n}\n\nexport interface EvidenceRef {\n /**\n * Where the evidence lives. `span` and `event` refer to OTLP trace\n * elements; `artifact` to a file inside the run's artifact tree;\n * `finding` to another AnalystFinding (cross-analyst chaining);\n * `metric` to a named scalar reading the renderer knows how to read.\n */\n kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric'\n uri: string\n excerpt?: string\n}\n\n// ── Analyst contract ─────────────────────────────────────────────────\n\n/**\n * The discriminator the registry uses to pass the right input.\n * `custom` is the escape hatch — analysts that need something else\n * (e.g. an embedding cache, a partner SDK handle) read it from\n * `AnalystRunInputs.custom[<analyst id>]`.\n */\nexport type AnalystInputKind =\n | 'trace-store'\n | 'artifact-dir'\n | 'run-record'\n | 'judge-input'\n | 'custom'\n\nexport interface AnalystCost {\n /** `deterministic` analysts MUST NOT call the LLM. */\n kind: 'deterministic' | 'llm'\n /** Optional declared upper bound; the registry can enforce a budget. */\n est_usd_per_run?: number\n /** Models the analyst expects to use (informational). */\n models?: string[]\n /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */\n settlement_timeout_ms?: number\n}\n\nexport interface AnalystRequirements {\n /** Min number of shots / samples the analyst needs to produce signal. */\n min_shots?: number\n /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */\n capabilities?: string[]\n}\n\n/**\n * What's passed to every analyst call. The registry resolves which\n * field the analyst's `inputKind` selects and asserts it's present.\n */\nexport interface AnalystRunInputs {\n traceStore?: TraceAnalysisStore\n artifactDir?: string\n runRecord?: RunRecord\n judgeInput?: JudgeInput\n /** Keyed by analyst id; populated by callers that registered custom analysts. */\n custom?: Record<string, unknown>\n}\n\nexport interface AnalystContext {\n runId: string\n /** Stable correlation id so logs from a single registry.run() share a tag. */\n correlationId: string\n /** Enforced wall-clock deadline (epoch ms). */\n deadlineMs?: number\n /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */\n budgetUsd?: number\n /** Shared paid-call account when the analyst runs inside a larger campaign. */\n costLedger?: CostLedgerHandle\n /** Attribution phase used when writing to the shared paid-call account. */\n costPhase?: string\n /**\n * Shared chat client. Analysts that call an LLM go through this so\n * the operator picks transport (sandbox-sdk | router | cli-bridge |\n * direct-provider | mock) at the registry boundary without touching\n * analyst code.\n */\n chat?: ChatClient\n /**\n * Findings from a prior run the operator wants the analyst to see as\n * retrieval context. Kinds that take advantage of cross-run memory\n * (failure-mode \"I saw this cluster last run\", knowledge-gap \"the wiki\n * page I asked for is still missing\") render these into the actor's\n * working set. Filtering is the operator's job: pass the slice that\n * matches the analyst's id, or pass everything and let the kind\n * filter. Empty / absent means no cross-run context.\n */\n priorFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Findings emitted by analysts that completed earlier in this registry run.\n * This is separate from `priorFindings`: upstream findings are dependency\n * context for the current pass, while prior findings are cross-run memory.\n * The registry populates this only when `RegistryRunOpts.chainFindings` is on.\n */\n upstreamFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Report metered work independently of findings. This keeps an empty finding\n * set from erasing token/cost telemetry. Multiple receipts are accumulated.\n */\n recordUsage?: (receipt: AnalystUsageReceipt) => void\n /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */\n tags?: Record<string, string>\n /** Logger callback — analysts SHOULD prefer this over console.* for testability. */\n log?: (msg: string, fields?: Record<string, unknown>) => void\n /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */\n signal?: AbortSignal\n}\n\n/**\n * The minimal contract. Concrete analysts can refine `TInput` so\n * implementations stay type-safe (e.g. a trace analyst's `TInput` is\n * `TraceAnalysisStore`); the registry passes the right field from\n * `AnalystRunInputs` based on `inputKind`.\n */\nexport interface Analyst<TInput = unknown> {\n /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */\n readonly id: string\n /** Human-readable. One sentence. */\n readonly description: string\n readonly inputKind: AnalystInputKind\n readonly cost: AnalystCost\n readonly requires?: AnalystRequirements\n /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */\n readonly version: string\n analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>\n}\n\n/** Metered work performed by one analyst call. */\nexport interface AnalystUsageReceipt {\n /** Number of model-usage records observed at the provider boundary. */\n calls: number | null\n /** Null when the provider did not return token accounting. */\n tokens: RunTokenUsage | null\n /** Observed, estimated, or explicitly uncaptured dollar cost. */\n cost: RunCostProvenance\n /** Known lower bound when one or more calls have uncaptured cost. */\n knownCostUsd?: number\n /**\n * Token counts the provider reported on only one side. Present exactly when\n * `tokens` is null and at least one side WAS reported: `RunTokenUsage` has no\n * nullable side, so a one-sided count cannot live in `tokens` without writing\n * a zero nobody measured. Read it as a lower bound, never as a total — the\n * field exists so a null `tokens` cannot hide a real count.\n */\n partialTokens?: { input: number | null; output: number | null }\n /**\n * True when the token counts were DERIVED by the transport (from character\n * lengths, say) rather than measured by the model provider. `cost.kind` is\n * `estimated` both for a rate estimate over exact tokens and for one over\n * derived tokens; this is the field that separates them.\n */\n tokensEstimated?: boolean\n}\n\n// ── finding_id stability ─────────────────────────────────────────────\n\n/**\n * Compute the stable finding_id from the identity-defining fields.\n * Default implementation hashes {analyst_id, area, subject, normalized claim}.\n * Analysts that emit findings whose claim text varies per run (timestamps,\n * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,\n * or (b) move the variable part into `rationale`/`metadata` and keep the\n * `claim` static.\n */\nexport function computeFindingId(input: {\n analyst_id: string\n area: string\n subject?: string\n claim: string\n /** Override the claim for hashing — use when the displayed claim has run-specific bits. */\n id_basis?: string\n}): string {\n const basis = JSON.stringify({\n a: input.analyst_id,\n r: input.area,\n s: input.subject ?? '',\n c: normalizeClaim(input.id_basis ?? input.claim),\n })\n return `f_${createHash('sha256').update(basis).digest('hex').slice(0, 20)}`\n}\n\nfunction normalizeClaim(c: string): string {\n // Lowercase, collapse whitespace, strip trailing punctuation. Goal:\n // \"Leaf X failed install\" and \"Leaf X failed install.\" hash the same.\n return c\n .toLowerCase()\n .replace(/\\s+/g, ' ')\n .replace(/[.!?;:,]+$/g, '')\n .trim()\n}\n\n/**\n * Convenience factory: produce a fully-formed AnalystFinding with the\n * id computed automatically. Analyst code stays terse.\n */\nexport function makeFinding(\n init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): AnalystFinding {\n const { id_basis, produced_at, ...rest } = init\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({\n analyst_id: rest.analyst_id,\n area: rest.area,\n subject: rest.subject,\n claim: rest.claim,\n id_basis,\n }),\n produced_at: produced_at ?? new Date().toISOString(),\n ...rest,\n }\n}\n\n/** Build a finding whose source is explicitly allowed during candidate generation. */\nexport function makeProposalFinding(\n init: Omit<ProposalFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): ProposalFinding {\n const { proposal_origin, ...finding } = init\n return { ...makeFinding(finding), proposal_origin }\n}\n\n// ── Registry result envelope ────────────────────────────────────────\n\nexport interface AnalystRunSummary {\n analyst_id: string\n status: 'ok' | 'skipped' | 'failed'\n /** Why skipped — missing input, budget exceeded, capability unmet. */\n reason?: string\n findings_count: number\n latency_ms: number\n /** Additive model usage and cost provenance for this analyst. */\n usage: AnalystUsageReceipt\n /** When `status='failed'`: the error class + message, never the full stack. */\n error?: { class: string; message: string }\n}\n\nexport interface AnalystRunResult {\n run_id: string\n correlation_id: string\n started_at: string\n ended_at: string\n findings: AnalystFinding[]\n per_analyst: AnalystRunSummary[]\n /** Total LLM cost in USD across all analysts in this registry.run(). */\n total_cost_usd: number\n /**\n * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only\n * the known subtotal and must not be treated as the run's total spend.\n */\n total_cost_provenance?: RunCostProvenance\n}\n\n// ── Streaming event envelope ────────────────────────────────────────\n\n/**\n * Events emitted by `AnalystRegistry.runStream(...)` in real time as\n * the registry executes. UIs subscribe via `for await (const ev of\n * registry.runStream(...))`; `registry.run(...)` is a thin collector\n * over the same stream, so the two surfaces share their invariants.\n *\n * Per-finding events are intentionally omitted — analyzers are batch\n * operations (a recursive engine returns the full `findings:json[]` at the\n * end of the responder), so streaming inside one analyst would only\n * emit partial JSON consumers can't render. The kind-completion event\n * is the right granularity; subscribers wanting per-finding rendering\n * iterate `event.findings` themselves.\n */\nexport type AnalystRunEvent =\n | {\n type: 'run-started'\n run_id: string\n correlation_id: string\n started_at: string\n /** The ordered list of analyst ids the registry will run. */\n analyst_ids: ReadonlyArray<string>\n }\n | {\n type: 'analyst-skipped'\n summary: AnalystRunSummary\n }\n | {\n type: 'analyst-started'\n analyst_id: string\n started_at: string\n }\n | {\n type: 'analyst-completed'\n /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */\n summary: AnalystRunSummary\n findings: ReadonlyArray<AnalystFinding>\n }\n | {\n type: 'run-completed'\n result: AnalystRunResult\n }\n","import type { CostChannel, CostLedgerFilter, CostLedgerHandle } from '../cost-ledger'\nimport type { AnalystUsageReceipt } from './types'\n\nexport const DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS = 5_000\n\n/** Convert one ledger channel's complete call set into one analyst receipt. */\nexport function usageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n filter: CostChannel | CostLedgerFilter = 'analyst',\n): AnalystUsageReceipt {\n const resolvedFilter = typeof filter === 'string' ? { channel: filter } : filter\n const summary = ledger.summary(resolvedFilter)\n const receipts = ledger.list(resolvedFilter)\n const hasReasoningUsage = receipts.some((receipt) => receipt.reasoningTokens !== undefined)\n const hasCacheWriteUsage = receipts.some((receipt) => receipt.cacheWriteTokens !== undefined)\n const cost = summary.costProvenance\n return {\n calls: summary.totalCalls + summary.pendingCalls,\n tokens: summary.usageComplete\n ? {\n input: summary.inputTokens,\n output: summary.outputTokens,\n ...(hasReasoningUsage ? { reasoning: summary.reasoningTokens ?? 0 } : {}),\n ...(summary.cachedTokens > 0 ? { cached: summary.cachedTokens } : {}),\n ...(hasCacheWriteUsage ? { cacheWrite: summary.cacheWriteTokens ?? 0 } : {}),\n }\n : null,\n cost,\n ...(cost.kind === 'uncaptured' ? { knownCostUsd: summary.totalCostUsd } : {}),\n }\n}\n\nexport interface SettledUsageReceipt {\n settled: boolean\n pendingCalls: number\n receipt: AnalystUsageReceipt\n}\n\n/** Wait a bounded time for late provider receipts, then take one immutable snapshot. */\nexport async function settleUsageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n options: CostLedgerFilter & { timeoutMs?: number } = {},\n): Promise<SettledUsageReceipt> {\n const { timeoutMs: requestedTimeoutMs, ...requestedFilter } = options\n const filter: CostLedgerFilter = {\n channel: requestedFilter.channel ?? 'analyst',\n ...(requestedFilter.phase === undefined ? {} : { phase: requestedFilter.phase }),\n ...(requestedFilter.tags === undefined ? {} : { tags: requestedFilter.tags }),\n }\n const timeoutMs = validateUsageSettlementTimeout(requestedTimeoutMs)\n const initial = ledger.summary(filter)\n const waitResult =\n initial.pendingCalls === 0\n ? true\n : ledger.waitForIdle\n ? await ledger.waitForIdle({ timeoutMs })\n : false\n const pendingCalls = ledger.summary(filter).pendingCalls\n return {\n settled: waitResult && pendingCalls === 0,\n pendingCalls,\n receipt: usageReceiptFromCostLedger(ledger, filter),\n }\n}\n\nexport function validateUsageSettlementTimeout(timeoutMs?: number): number {\n const resolved = timeoutMs ?? DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS\n if (!Number.isSafeInteger(resolved) || resolved < 0 || resolved > 2_147_483_647) {\n throw new TypeError(\n 'settlementTimeoutMs must be a non-negative safe integer no greater than 2147483647',\n )\n }\n return resolved\n}\n\nexport function assertValidAnalystUsageReceipt(\n receipt: AnalystUsageReceipt,\n context = 'AnalystContext.recordUsage',\n): void {\n if (receipt.calls !== null && (!Number.isSafeInteger(receipt.calls) || receipt.calls < 0)) {\n throw new Error(`${context}: calls must be a non-negative safe integer or null`)\n }\n if (receipt.tokens) {\n assertNonNegativeSafeInteger(receipt.tokens.input, 'tokens.input', context)\n assertNonNegativeSafeInteger(receipt.tokens.output, 'tokens.output', context)\n if (receipt.tokens.reasoning !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.reasoning, 'tokens.reasoning', context)\n if (receipt.tokens.reasoning > receipt.tokens.output) {\n throw new Error(`${context}: tokens.reasoning must not exceed tokens.output`)\n }\n }\n if (receipt.tokens.cached !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cached, 'tokens.cached', context)\n }\n if (receipt.tokens.cacheWrite !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cacheWrite, 'tokens.cacheWrite', context)\n }\n }\n if (receipt.cost.kind !== 'uncaptured') {\n assertNonNegativeFinite(receipt.cost.usd, 'cost.usd', context)\n } else if (receipt.cost.usd !== null) {\n throw new Error(`${context}: uncaptured cost.usd must be null`)\n }\n if (receipt.knownCostUsd !== undefined) {\n assertNonNegativeFinite(receipt.knownCostUsd, 'knownCostUsd', context)\n }\n if (receipt.partialTokens) {\n const { input, output } = receipt.partialTokens\n if (receipt.tokens) {\n throw new Error(`${context}: partialTokens must be absent when tokens is complete`)\n }\n if (input === null && output === null) {\n throw new Error(`${context}: partialTokens must carry at least one reported side`)\n }\n if (input !== null) assertNonNegativeSafeInteger(input, 'partialTokens.input', context)\n if (output !== null) assertNonNegativeSafeInteger(output, 'partialTokens.output', context)\n }\n}\n\nfunction assertNonNegativeSafeInteger(value: number, field: string, context: string): void {\n if (!Number.isSafeInteger(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative safe integer`)\n }\n}\n\nfunction assertNonNegativeFinite(value: number, field: string, context: string): void {\n if (!Number.isFinite(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative finite number`)\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;AAiPA,SAAgB,iBAAiB,OAOtB;CACT,MAAM,QAAQ,KAAK,UAAU;EAC3B,GAAG,MAAM;EACT,GAAG,MAAM;EACT,GAAG,MAAM,WAAW;EACpB,GAAG,eAAe,MAAM,YAAY,MAAM,KAAK;CACjD,CAAC;CACD,OAAO,KAAK,WAAW,QAAQ,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1E;AAEA,SAAS,eAAe,GAAmB;CAGzC,OAAO,EACJ,YAAY,CAAC,CACb,QAAQ,QAAQ,GAAG,CAAC,CACpB,QAAQ,eAAe,EAAE,CAAC,CAC1B,KAAK;AACV;;;;;AAMA,SAAgB,YACd,MAIgB;CAChB,MAAM,EAAE,UAAU,aAAa,GAAG,SAAS;CAC3C,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAC3B,YAAY,KAAK;GACjB,MAAM,KAAK;GACX,SAAS,KAAK;GACd,OAAO,KAAK;GACZ;EACF,CAAC;EACD,aAAa,gCAAe,IAAI,KAAK,EAAA,CAAE,YAAY;EACnD,GAAG;CACL;AACF;;AAGA,SAAgB,oBACd,MAIiB;CACjB,MAAM,EAAE,iBAAiB,GAAG,YAAY;CACxC,OAAO;EAAE,GAAG,YAAY,OAAO;EAAG;CAAgB;AACpD;;ACxSA,SAAgB,2BACd,QACA,SAAyC,WACpB;CACrB,MAAM,iBAAiB,OAAO,WAAW,WAAW,EAAE,SAAS,OAAO,IAAI;CAC1E,MAAM,UAAU,OAAO,QAAQ,cAAc;CAC7C,MAAM,WAAW,OAAO,KAAK,cAAc;CAC3C,MAAM,oBAAoB,SAAS,MAAM,YAAY,QAAQ,oBAAoB,KAAA,CAAS;CAC1F,MAAM,qBAAqB,SAAS,MAAM,YAAY,QAAQ,qBAAqB,KAAA,CAAS;CAC5F,MAAM,OAAO,QAAQ;CACrB,OAAO;EACL,OAAO,QAAQ,aAAa,QAAQ;EACpC,QAAQ,QAAQ,gBACZ;GACE,OAAO,QAAQ;GACf,QAAQ,QAAQ;GAChB,GAAI,oBAAoB,EAAE,WAAW,QAAQ,mBAAmB,EAAE,IAAI,CAAC;GACvE,GAAI,QAAQ,eAAe,IAAI,EAAE,QAAQ,QAAQ,aAAa,IAAI,CAAC;GACnE,GAAI,qBAAqB,EAAE,YAAY,QAAQ,oBAAoB,EAAE,IAAI,CAAC;EAC5E,IACA;EACJ;EACA,GAAI,KAAK,SAAS,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CAC7E;AACF;;AASA,eAAsB,iCACpB,QACA,UAAqD,CAAC,GACxB;CAC9B,MAAM,EAAE,WAAW,oBAAoB,GAAG,oBAAoB;CAC9D,MAAM,SAA2B;EAC/B,SAAS,gBAAgB,WAAW;EACpC,GAAI,gBAAgB,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,gBAAgB,MAAM;EAC9E,GAAI,gBAAgB,SAAS,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM,gBAAgB,KAAK;CAC7E;CACA,MAAM,YAAY,+BAA+B,kBAAkB;CAEnE,MAAM,aADU,OAAO,QAAQ,MAEvB,CAAC,CAAC,iBAAiB,IACrB,OACA,OAAO,cACL,MAAM,OAAO,YAAY,EAAE,UAAU,CAAC,IACtC;CACR,MAAM,eAAe,OAAO,QAAQ,MAAM,CAAC,CAAC;CAC5C,OAAO;EACL,SAAS,cAAc,iBAAiB;EACxC;EACA,SAAS,2BAA2B,QAAQ,MAAM;CACpD;AACF;AAEA,SAAgB,+BAA+B,WAA4B;CACzE,MAAM,WAAW,aAAA;CACjB,IAAI,CAAC,OAAO,cAAc,QAAQ,KAAK,WAAW,KAAK,WAAW,YAChE,MAAM,IAAI,UACR,oFACF;CAEF,OAAO;AACT;AAEA,SAAgB,+BACd,SACA,UAAU,8BACJ;CACN,IAAI,QAAQ,UAAU,SAAS,CAAC,OAAO,cAAc,QAAQ,KAAK,KAAK,QAAQ,QAAQ,IACrF,MAAM,IAAI,MAAM,GAAG,QAAQ,oDAAoD;CAEjF,IAAI,QAAQ,QAAQ;EAClB,6BAA6B,QAAQ,OAAO,OAAO,gBAAgB,OAAO;EAC1E,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAC5E,IAAI,QAAQ,OAAO,cAAc,KAAA,GAAW;GAC1C,6BAA6B,QAAQ,OAAO,WAAW,oBAAoB,OAAO;GAClF,IAAI,QAAQ,OAAO,YAAY,QAAQ,OAAO,QAC5C,MAAM,IAAI,MAAM,GAAG,QAAQ,iDAAiD;EAEhF;EACA,IAAI,QAAQ,OAAO,WAAW,KAAA,GAC5B,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAE9E,IAAI,QAAQ,OAAO,eAAe,KAAA,GAChC,6BAA6B,QAAQ,OAAO,YAAY,qBAAqB,OAAO;CAExF;CACA,IAAI,QAAQ,KAAK,SAAS,cACxB,wBAAwB,QAAQ,KAAK,KAAK,YAAY,OAAO;MACxD,IAAI,QAAQ,KAAK,QAAQ,MAC9B,MAAM,IAAI,MAAM,GAAG,QAAQ,mCAAmC;CAEhE,IAAI,QAAQ,iBAAiB,KAAA,GAC3B,wBAAwB,QAAQ,cAAc,gBAAgB,OAAO;CAEvE,IAAI,QAAQ,eAAe;EACzB,MAAM,EAAE,OAAO,WAAW,QAAQ;EAClC,IAAI,QAAQ,QACV,MAAM,IAAI,MAAM,GAAG,QAAQ,uDAAuD;EAEpF,IAAI,UAAU,QAAQ,WAAW,MAC/B,MAAM,IAAI,MAAM,GAAG,QAAQ,sDAAsD;EAEnF,IAAI,UAAU,MAAM,6BAA6B,OAAO,uBAAuB,OAAO;EACtF,IAAI,WAAW,MAAM,6BAA6B,QAAQ,wBAAwB,OAAO;CAC3F;AACF;AAEA,SAAS,6BAA6B,OAAe,OAAe,SAAuB;CACzF,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,QAAQ,GAC1C,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,qCAAqC;AAE9E;AAEA,SAAS,wBAAwB,OAAe,OAAe,SAAuB;CACpF,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,GACrC,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,sCAAsC;AAE/E"}
1
+ {"version":3,"file":"usage-receipt-EVI8B8Xu.js","names":[],"sources":["../src/analyst/types.ts","../src/analyst/usage-receipt.ts"],"sourcesContent":["/**\n * Analyst contract — the missing orchestration layer over agent-eval's\n * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,\n * SemanticConceptJudge, JudgeFn, ...).\n *\n * Each existing primitive returns its own output shape. The Analyst\n * contract is the single envelope every primitive lifts into, so a\n * registry can run N analysts against a run and a single renderer can\n * compose findings without knowing which analyzer produced them.\n *\n * The contract is intentionally domain-agnostic: nothing here knows\n * about code, voice, RAG, or any particular agent stack. Analysts\n * declare what INPUT KIND they need (a trace store, an artifact dir,\n * a RunRecord, a JudgeInput, or `custom`), and the registry routes\n * the matching input from `AnalystRunInputs`.\n */\n\nimport { createHash } from 'node:crypto'\nimport type { CostLedgerHandle } from '../cost-ledger'\nimport type { RunCostProvenance, RunRecord, RunTokenUsage } from '../run-record'\nimport type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { JudgeInput } from '../types'\nimport type { ChatClient } from './chat-client'\n\n/**\n * Unified envelope every analyst emits. Schema-versioned so renderers\n * and time-series diffs survive future field additions.\n */\nexport interface AnalystFinding {\n schema_version: '1.0.0'\n /**\n * Stable hash over identity-defining fields (analyst_id + canonical\n * claim + area + optional subject). Two findings from two runs that\n * \"are the same finding\" share this id — that's what `diffFindings`\n * uses to compute appeared/disappeared sets across runs.\n */\n finding_id: string\n analyst_id: string\n produced_at: string\n severity: AnalystSeverity\n /**\n * Coarse classification. Renderers group by this. Free-form so\n * domain-specific analysts can introduce categories without a\n * schema change ('agent-reasoning', 'verification', 'cost',\n * 'tool-use', 'safety', 'latency', 'data-quality', ...).\n */\n area: string\n claim: string\n rationale?: string\n evidence_refs: EvidenceRef[]\n recommended_action?: string\n validation_plan?: string\n /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */\n confidence: number\n /**\n * Optional subject the finding is about — leaf id, agent id, request\n * id. Included in finding_id when present so per-subject findings\n * diff cleanly across runs.\n */\n subject?: string\n /** True when this finding was lifted from a judge result rather than observed\n * directly in a trace or artifact. Descriptive only: proposal access is\n * controlled by `ProposalFinding.proposal_origin`. */\n derived_from_judge?: boolean\n /** Analyst-private extras; renderers ignore unless they know the analyst. */\n metadata?: Record<string, unknown>\n}\n\nexport type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info'\n\n/** Data sources that candidate generation may intentionally learn from. */\nexport type ProposalFindingOrigin = 'search' | 'production'\n\n/** A finding explicitly admitted as candidate-generation input. */\nexport type ProposalFinding = AnalystFinding & {\n readonly proposal_origin: ProposalFindingOrigin\n}\n\nexport interface EvidenceRef {\n /**\n * Where the evidence lives. `span` and `event` refer to OTLP trace\n * elements; `artifact` to a file inside the run's artifact tree;\n * `finding` to another AnalystFinding (cross-analyst chaining);\n * `metric` to a named scalar reading the renderer knows how to read.\n */\n kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric'\n uri: string\n excerpt?: string\n}\n\n// ── Analyst contract ─────────────────────────────────────────────────\n\n/**\n * The discriminator the registry uses to pass the right input.\n * `custom` is the escape hatch — analysts that need something else\n * (e.g. an embedding cache, a partner SDK handle) read it from\n * `AnalystRunInputs.custom[<analyst id>]`.\n */\nexport type AnalystInputKind =\n | 'trace-store'\n | 'artifact-dir'\n | 'run-record'\n | 'judge-input'\n | 'custom'\n\nexport interface AnalystCost {\n /** `deterministic` analysts MUST NOT call the LLM. */\n kind: 'deterministic' | 'llm'\n /** Optional declared upper bound; the registry can enforce a budget. */\n est_usd_per_run?: number\n /** Models the analyst expects to use (informational). */\n models?: string[]\n /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */\n settlement_timeout_ms?: number\n}\n\nexport interface AnalystRequirements {\n /** Min number of shots / samples the analyst needs to produce signal. */\n min_shots?: number\n /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */\n capabilities?: string[]\n}\n\n/**\n * What's passed to every analyst call. The registry resolves which\n * field the analyst's `inputKind` selects and asserts it's present.\n */\nexport interface AnalystRunInputs {\n traceStore?: TraceAnalysisStore\n artifactDir?: string\n runRecord?: RunRecord\n judgeInput?: JudgeInput\n /** Keyed by analyst id; populated by callers that registered custom analysts. */\n custom?: Record<string, unknown>\n}\n\nexport interface AnalystContext {\n runId: string\n /** Stable correlation id so logs from a single registry.run() share a tag. */\n correlationId: string\n /** Enforced wall-clock deadline (epoch ms). */\n deadlineMs?: number\n /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */\n budgetUsd?: number\n /** Shared paid-call account when the analyst runs inside a larger campaign. */\n costLedger?: CostLedgerHandle\n /** Attribution phase used when writing to the shared paid-call account. */\n costPhase?: string\n /**\n * Shared chat client. Analysts that call an LLM go through this so\n * the operator picks transport (sandbox-sdk | router | cli-bridge |\n * direct-provider | mock) at the registry boundary without touching\n * analyst code.\n */\n chat?: ChatClient\n /**\n * Findings from a prior run the operator wants the analyst to see as\n * retrieval context. Kinds that take advantage of cross-run memory\n * (failure-mode \"I saw this cluster last run\", knowledge-gap \"the wiki\n * page I asked for is still missing\") render these into the actor's\n * working set. Filtering is the operator's job: pass the slice that\n * matches the analyst's id, or pass everything and let the kind\n * filter. Empty / absent means no cross-run context.\n */\n priorFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Findings emitted by analysts that completed earlier in this registry run.\n * This is separate from `priorFindings`: upstream findings are dependency\n * context for the current pass, while prior findings are cross-run memory.\n * The registry populates this only when `RegistryRunOpts.chainFindings` is on.\n */\n upstreamFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Report metered work independently of findings. This keeps an empty finding\n * set from erasing token/cost telemetry. Multiple receipts are accumulated.\n */\n recordUsage?: (receipt: AnalystUsageReceipt) => void\n /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */\n tags?: Record<string, string>\n /** Logger callback — analysts SHOULD prefer this over console.* for testability. */\n log?: (msg: string, fields?: Record<string, unknown>) => void\n /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */\n signal?: AbortSignal\n /**\n * Optional live-execution port. A runtime that owns a sandbox or checkout\n * fills it so an analyst can execute a bounded probe against the run's\n * produced state instead of reasoning about it from the trace alone. This\n * package defines only the port: no field here reaches for an agent loop,\n * and an absent probe means the analyst works from recorded evidence.\n */\n probe?: ExecutionProbe\n}\n\n// ── Live-execution port ─────────────────────────────────────────────\n\n/** One bounded command an analyst asks the probe to run. */\nexport interface ExecutionProbeRequest {\n command: string\n /** Working directory inside the probed environment. */\n cwd?: string\n /** Hard wall-clock deadline for this one execution. */\n timeoutMs: number\n /** Bytes of combined output retained; the prober truncates beyond it. */\n maxOutputBytes?: number\n signal?: AbortSignal\n}\n\n/**\n * Typed outcome of one probe execution. `succeeded: false` is a PROBE failure\n * (the environment could not run the command); a command that ran and exited\n * non-zero is a successful observation with a non-zero `exitCode`.\n */\nexport type ExecutionProbeOutcome =\n | {\n succeeded: true\n exitCode: number\n stdout: string\n stderr: string\n durationMs: number\n /** True when output was cut at `maxOutputBytes`. */\n truncated: boolean\n }\n | { succeeded: false; error: { class: string; message: string } }\n\n/**\n * The seam a runtime fills to let analysts observe produced state live.\n * Implementations own sandboxing, credentials, and cleanup; analysts only\n * submit bounded requests and read typed outcomes.\n */\nexport interface ExecutionProbe {\n /** One plain sentence naming what is being probed (e.g. a sandbox id). */\n readonly description: string\n execute(request: ExecutionProbeRequest): Promise<ExecutionProbeOutcome>\n}\n\n/**\n * The minimal contract. Concrete analysts can refine `TInput` so\n * implementations stay type-safe (e.g. a trace analyst's `TInput` is\n * `TraceAnalysisStore`); the registry passes the right field from\n * `AnalystRunInputs` based on `inputKind`.\n */\nexport interface Analyst<TInput = unknown> {\n /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */\n readonly id: string\n /** Human-readable. One sentence. */\n readonly description: string\n readonly inputKind: AnalystInputKind\n readonly cost: AnalystCost\n readonly requires?: AnalystRequirements\n /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */\n readonly version: string\n analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>\n}\n\n/** Metered work performed by one analyst call. */\nexport interface AnalystUsageReceipt {\n /** Number of model-usage records observed at the provider boundary. */\n calls: number | null\n /** Null when the provider did not return token accounting. */\n tokens: RunTokenUsage | null\n /** Observed, estimated, or explicitly uncaptured dollar cost. */\n cost: RunCostProvenance\n /** Known lower bound when one or more calls have uncaptured cost. */\n knownCostUsd?: number\n /**\n * Token counts the provider reported on only one side. Present exactly when\n * `tokens` is null and at least one side WAS reported: `RunTokenUsage` has no\n * nullable side, so a one-sided count cannot live in `tokens` without writing\n * a zero nobody measured. Read it as a lower bound, never as a total — the\n * field exists so a null `tokens` cannot hide a real count.\n */\n partialTokens?: { input: number | null; output: number | null }\n /**\n * True when the token counts were DERIVED by the transport (from character\n * lengths, say) rather than measured by the model provider. `cost.kind` is\n * `estimated` both for a rate estimate over exact tokens and for one over\n * derived tokens; this is the field that separates them.\n */\n tokensEstimated?: boolean\n}\n\n// ── finding_id stability ─────────────────────────────────────────────\n\n/**\n * Compute the stable finding_id from the identity-defining fields.\n * Default implementation hashes {analyst_id, area, subject, normalized claim}.\n * Analysts that emit findings whose claim text varies per run (timestamps,\n * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,\n * or (b) move the variable part into `rationale`/`metadata` and keep the\n * `claim` static.\n */\nexport function computeFindingId(input: {\n analyst_id: string\n area: string\n subject?: string\n claim: string\n /** Override the claim for hashing — use when the displayed claim has run-specific bits. */\n id_basis?: string\n}): string {\n const basis = JSON.stringify({\n a: input.analyst_id,\n r: input.area,\n s: input.subject ?? '',\n c: normalizeClaim(input.id_basis ?? input.claim),\n })\n return `f_${createHash('sha256').update(basis).digest('hex').slice(0, 20)}`\n}\n\nfunction normalizeClaim(c: string): string {\n // Lowercase, collapse whitespace, strip trailing punctuation. Goal:\n // \"Leaf X failed install\" and \"Leaf X failed install.\" hash the same.\n return c\n .toLowerCase()\n .replace(/\\s+/g, ' ')\n .replace(/[.!?;:,]+$/g, '')\n .trim()\n}\n\n/**\n * Convenience factory: produce a fully-formed AnalystFinding with the\n * id computed automatically. Analyst code stays terse.\n */\nexport function makeFinding(\n init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): AnalystFinding {\n const { id_basis, produced_at, ...rest } = init\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({\n analyst_id: rest.analyst_id,\n area: rest.area,\n subject: rest.subject,\n claim: rest.claim,\n id_basis,\n }),\n produced_at: produced_at ?? new Date().toISOString(),\n ...rest,\n }\n}\n\n/** Build a finding whose source is explicitly allowed during candidate generation. */\nexport function makeProposalFinding(\n init: Omit<ProposalFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): ProposalFinding {\n const { proposal_origin, ...finding } = init\n return { ...makeFinding(finding), proposal_origin }\n}\n\n// ── Registry result envelope ────────────────────────────────────────\n\nexport interface AnalystRunSummary {\n analyst_id: string\n status: 'ok' | 'skipped' | 'failed'\n /** Why skipped — missing input, budget exceeded, capability unmet. */\n reason?: string\n findings_count: number\n latency_ms: number\n /** Additive model usage and cost provenance for this analyst. */\n usage: AnalystUsageReceipt\n /** When `status='failed'`: the error class + message, never the full stack. */\n error?: { class: string; message: string }\n}\n\nexport interface AnalystRunResult {\n run_id: string\n correlation_id: string\n started_at: string\n ended_at: string\n findings: AnalystFinding[]\n per_analyst: AnalystRunSummary[]\n /** Total LLM cost in USD across all analysts in this registry.run(). */\n total_cost_usd: number\n /**\n * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only\n * the known subtotal and must not be treated as the run's total spend.\n */\n total_cost_provenance?: RunCostProvenance\n}\n\n// ── Streaming event envelope ────────────────────────────────────────\n\n/**\n * Events emitted by `AnalystRegistry.runStream(...)` in real time as\n * the registry executes. UIs subscribe via `for await (const ev of\n * registry.runStream(...))`; `registry.run(...)` is a thin collector\n * over the same stream, so the two surfaces share their invariants.\n *\n * Per-finding events are intentionally omitted — analyzers are batch\n * operations (a recursive engine returns the full `findings:json[]` at the\n * end of the responder), so streaming inside one analyst would only\n * emit partial JSON consumers can't render. The kind-completion event\n * is the right granularity; subscribers wanting per-finding rendering\n * iterate `event.findings` themselves.\n */\nexport type AnalystRunEvent =\n | {\n type: 'run-started'\n run_id: string\n correlation_id: string\n started_at: string\n /** The ordered list of analyst ids the registry will run. */\n analyst_ids: ReadonlyArray<string>\n }\n | {\n type: 'analyst-skipped'\n summary: AnalystRunSummary\n }\n | {\n type: 'analyst-started'\n analyst_id: string\n started_at: string\n }\n | {\n type: 'analyst-completed'\n /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */\n summary: AnalystRunSummary\n findings: ReadonlyArray<AnalystFinding>\n }\n | {\n type: 'run-completed'\n result: AnalystRunResult\n }\n","import type { CostChannel, CostLedgerFilter, CostLedgerHandle } from '../cost-ledger'\nimport type { AnalystUsageReceipt } from './types'\n\nexport const DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS = 5_000\n\n/** Convert one ledger channel's complete call set into one analyst receipt. */\nexport function usageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n filter: CostChannel | CostLedgerFilter = 'analyst',\n): AnalystUsageReceipt {\n const resolvedFilter = typeof filter === 'string' ? { channel: filter } : filter\n const summary = ledger.summary(resolvedFilter)\n const receipts = ledger.list(resolvedFilter)\n const hasReasoningUsage = receipts.some((receipt) => receipt.reasoningTokens !== undefined)\n const hasCacheWriteUsage = receipts.some((receipt) => receipt.cacheWriteTokens !== undefined)\n const cost = summary.costProvenance\n return {\n calls: summary.totalCalls + summary.pendingCalls,\n tokens: summary.usageComplete\n ? {\n input: summary.inputTokens,\n output: summary.outputTokens,\n ...(hasReasoningUsage ? { reasoning: summary.reasoningTokens ?? 0 } : {}),\n ...(summary.cachedTokens > 0 ? { cached: summary.cachedTokens } : {}),\n ...(hasCacheWriteUsage ? { cacheWrite: summary.cacheWriteTokens ?? 0 } : {}),\n }\n : null,\n cost,\n ...(cost.kind === 'uncaptured' ? { knownCostUsd: summary.totalCostUsd } : {}),\n }\n}\n\nexport interface SettledUsageReceipt {\n settled: boolean\n pendingCalls: number\n receipt: AnalystUsageReceipt\n}\n\n/** Wait a bounded time for late provider receipts, then take one immutable snapshot. */\nexport async function settleUsageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n options: CostLedgerFilter & { timeoutMs?: number } = {},\n): Promise<SettledUsageReceipt> {\n const { timeoutMs: requestedTimeoutMs, ...requestedFilter } = options\n const filter: CostLedgerFilter = {\n channel: requestedFilter.channel ?? 'analyst',\n ...(requestedFilter.phase === undefined ? {} : { phase: requestedFilter.phase }),\n ...(requestedFilter.tags === undefined ? {} : { tags: requestedFilter.tags }),\n }\n const timeoutMs = validateUsageSettlementTimeout(requestedTimeoutMs)\n const initial = ledger.summary(filter)\n const waitResult =\n initial.pendingCalls === 0\n ? true\n : ledger.waitForIdle\n ? await ledger.waitForIdle({ timeoutMs })\n : false\n const pendingCalls = ledger.summary(filter).pendingCalls\n return {\n settled: waitResult && pendingCalls === 0,\n pendingCalls,\n receipt: usageReceiptFromCostLedger(ledger, filter),\n }\n}\n\nexport function validateUsageSettlementTimeout(timeoutMs?: number): number {\n const resolved = timeoutMs ?? DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS\n if (!Number.isSafeInteger(resolved) || resolved < 0 || resolved > 2_147_483_647) {\n throw new TypeError(\n 'settlementTimeoutMs must be a non-negative safe integer no greater than 2147483647',\n )\n }\n return resolved\n}\n\nexport function assertValidAnalystUsageReceipt(\n receipt: AnalystUsageReceipt,\n context = 'AnalystContext.recordUsage',\n): void {\n if (receipt.calls !== null && (!Number.isSafeInteger(receipt.calls) || receipt.calls < 0)) {\n throw new Error(`${context}: calls must be a non-negative safe integer or null`)\n }\n if (receipt.tokens) {\n assertNonNegativeSafeInteger(receipt.tokens.input, 'tokens.input', context)\n assertNonNegativeSafeInteger(receipt.tokens.output, 'tokens.output', context)\n if (receipt.tokens.reasoning !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.reasoning, 'tokens.reasoning', context)\n if (receipt.tokens.reasoning > receipt.tokens.output) {\n throw new Error(`${context}: tokens.reasoning must not exceed tokens.output`)\n }\n }\n if (receipt.tokens.cached !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cached, 'tokens.cached', context)\n }\n if (receipt.tokens.cacheWrite !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cacheWrite, 'tokens.cacheWrite', context)\n }\n }\n if (receipt.cost.kind !== 'uncaptured') {\n assertNonNegativeFinite(receipt.cost.usd, 'cost.usd', context)\n } else if (receipt.cost.usd !== null) {\n throw new Error(`${context}: uncaptured cost.usd must be null`)\n }\n if (receipt.knownCostUsd !== undefined) {\n assertNonNegativeFinite(receipt.knownCostUsd, 'knownCostUsd', context)\n }\n if (receipt.partialTokens) {\n const { input, output } = receipt.partialTokens\n if (receipt.tokens) {\n throw new Error(`${context}: partialTokens must be absent when tokens is complete`)\n }\n if (input === null && output === null) {\n throw new Error(`${context}: partialTokens must carry at least one reported side`)\n }\n if (input !== null) assertNonNegativeSafeInteger(input, 'partialTokens.input', context)\n if (output !== null) assertNonNegativeSafeInteger(output, 'partialTokens.output', context)\n }\n}\n\nfunction assertNonNegativeSafeInteger(value: number, field: string, context: string): void {\n if (!Number.isSafeInteger(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative safe integer`)\n }\n}\n\nfunction assertNonNegativeFinite(value: number, field: string, context: string): void {\n if (!Number.isFinite(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative finite number`)\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;AAmSA,SAAgB,iBAAiB,OAOtB;CACT,MAAM,QAAQ,KAAK,UAAU;EAC3B,GAAG,MAAM;EACT,GAAG,MAAM;EACT,GAAG,MAAM,WAAW;EACpB,GAAG,eAAe,MAAM,YAAY,MAAM,KAAK;CACjD,CAAC;CACD,OAAO,KAAK,WAAW,QAAQ,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1E;AAEA,SAAS,eAAe,GAAmB;CAGzC,OAAO,EACJ,YAAY,CAAC,CACb,QAAQ,QAAQ,GAAG,CAAC,CACpB,QAAQ,eAAe,EAAE,CAAC,CAC1B,KAAK;AACV;;;;;AAMA,SAAgB,YACd,MAIgB;CAChB,MAAM,EAAE,UAAU,aAAa,GAAG,SAAS;CAC3C,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAC3B,YAAY,KAAK;GACjB,MAAM,KAAK;GACX,SAAS,KAAK;GACd,OAAO,KAAK;GACZ;EACF,CAAC;EACD,aAAa,gCAAe,IAAI,KAAK,EAAA,CAAE,YAAY;EACnD,GAAG;CACL;AACF;;AAGA,SAAgB,oBACd,MAIiB;CACjB,MAAM,EAAE,iBAAiB,GAAG,YAAY;CACxC,OAAO;EAAE,GAAG,YAAY,OAAO;EAAG;CAAgB;AACpD;;AC1VA,SAAgB,2BACd,QACA,SAAyC,WACpB;CACrB,MAAM,iBAAiB,OAAO,WAAW,WAAW,EAAE,SAAS,OAAO,IAAI;CAC1E,MAAM,UAAU,OAAO,QAAQ,cAAc;CAC7C,MAAM,WAAW,OAAO,KAAK,cAAc;CAC3C,MAAM,oBAAoB,SAAS,MAAM,YAAY,QAAQ,oBAAoB,KAAA,CAAS;CAC1F,MAAM,qBAAqB,SAAS,MAAM,YAAY,QAAQ,qBAAqB,KAAA,CAAS;CAC5F,MAAM,OAAO,QAAQ;CACrB,OAAO;EACL,OAAO,QAAQ,aAAa,QAAQ;EACpC,QAAQ,QAAQ,gBACZ;GACE,OAAO,QAAQ;GACf,QAAQ,QAAQ;GAChB,GAAI,oBAAoB,EAAE,WAAW,QAAQ,mBAAmB,EAAE,IAAI,CAAC;GACvE,GAAI,QAAQ,eAAe,IAAI,EAAE,QAAQ,QAAQ,aAAa,IAAI,CAAC;GACnE,GAAI,qBAAqB,EAAE,YAAY,QAAQ,oBAAoB,EAAE,IAAI,CAAC;EAC5E,IACA;EACJ;EACA,GAAI,KAAK,SAAS,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CAC7E;AACF;;AASA,eAAsB,iCACpB,QACA,UAAqD,CAAC,GACxB;CAC9B,MAAM,EAAE,WAAW,oBAAoB,GAAG,oBAAoB;CAC9D,MAAM,SAA2B;EAC/B,SAAS,gBAAgB,WAAW;EACpC,GAAI,gBAAgB,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,gBAAgB,MAAM;EAC9E,GAAI,gBAAgB,SAAS,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM,gBAAgB,KAAK;CAC7E;CACA,MAAM,YAAY,+BAA+B,kBAAkB;CAEnE,MAAM,aADU,OAAO,QAAQ,MAEvB,CAAC,CAAC,iBAAiB,IACrB,OACA,OAAO,cACL,MAAM,OAAO,YAAY,EAAE,UAAU,CAAC,IACtC;CACR,MAAM,eAAe,OAAO,QAAQ,MAAM,CAAC,CAAC;CAC5C,OAAO;EACL,SAAS,cAAc,iBAAiB;EACxC;EACA,SAAS,2BAA2B,QAAQ,MAAM;CACpD;AACF;AAEA,SAAgB,+BAA+B,WAA4B;CACzE,MAAM,WAAW,aAAA;CACjB,IAAI,CAAC,OAAO,cAAc,QAAQ,KAAK,WAAW,KAAK,WAAW,YAChE,MAAM,IAAI,UACR,oFACF;CAEF,OAAO;AACT;AAEA,SAAgB,+BACd,SACA,UAAU,8BACJ;CACN,IAAI,QAAQ,UAAU,SAAS,CAAC,OAAO,cAAc,QAAQ,KAAK,KAAK,QAAQ,QAAQ,IACrF,MAAM,IAAI,MAAM,GAAG,QAAQ,oDAAoD;CAEjF,IAAI,QAAQ,QAAQ;EAClB,6BAA6B,QAAQ,OAAO,OAAO,gBAAgB,OAAO;EAC1E,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAC5E,IAAI,QAAQ,OAAO,cAAc,KAAA,GAAW;GAC1C,6BAA6B,QAAQ,OAAO,WAAW,oBAAoB,OAAO;GAClF,IAAI,QAAQ,OAAO,YAAY,QAAQ,OAAO,QAC5C,MAAM,IAAI,MAAM,GAAG,QAAQ,iDAAiD;EAEhF;EACA,IAAI,QAAQ,OAAO,WAAW,KAAA,GAC5B,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAE9E,IAAI,QAAQ,OAAO,eAAe,KAAA,GAChC,6BAA6B,QAAQ,OAAO,YAAY,qBAAqB,OAAO;CAExF;CACA,IAAI,QAAQ,KAAK,SAAS,cACxB,wBAAwB,QAAQ,KAAK,KAAK,YAAY,OAAO;MACxD,IAAI,QAAQ,KAAK,QAAQ,MAC9B,MAAM,IAAI,MAAM,GAAG,QAAQ,mCAAmC;CAEhE,IAAI,QAAQ,iBAAiB,KAAA,GAC3B,wBAAwB,QAAQ,cAAc,gBAAgB,OAAO;CAEvE,IAAI,QAAQ,eAAe;EACzB,MAAM,EAAE,OAAO,WAAW,QAAQ;EAClC,IAAI,QAAQ,QACV,MAAM,IAAI,MAAM,GAAG,QAAQ,uDAAuD;EAEpF,IAAI,UAAU,QAAQ,WAAW,MAC/B,MAAM,IAAI,MAAM,GAAG,QAAQ,sDAAsD;EAEnF,IAAI,UAAU,MAAM,6BAA6B,OAAO,uBAAuB,OAAO;EACtF,IAAI,WAAW,MAAM,6BAA6B,QAAQ,wBAAwB,OAAO;CAC3F;AACF;AAEA,SAAS,6BAA6B,OAAe,OAAe,SAAuB;CACzF,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,QAAQ,GAC1C,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,qCAAqC;AAE9E;AAEA,SAAS,wBAAwB,OAAe,OAAe,SAAuB;CACpF,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,GACrC,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,sCAAsC;AAE/E"}
@@ -0,0 +1,201 @@
1
+ //#region src/verification-strategy.d.ts
2
+ /**
3
+ * Verification-strategy family — the shared vocabulary for "what certified
4
+ * this result", covering tasks with an answer key and tasks without one.
5
+ *
6
+ * Charter Wave 5 (docs/charter.md): an unsolved problem has no held-out
7
+ * suite by definition, so the held-out suite must be ONE member of a
8
+ * strategy family, not the family itself. This module carries the taxonomy
9
+ * and the port shape only. A checker is an injected executable boundary
10
+ * that returns a typed outcome, so a consumer binds its own kernel (a Lean
11
+ * toolchain, a metamorphic harness, a replication runner) and this package
12
+ * never chooses one. Instruments, never methods.
13
+ *
14
+ * Full family doc, per-member failure modes in prose, and the BCWW pilot
15
+ * worked example: docs/verification-strategies.md.
16
+ *
17
+ * Leaf module: no imports from this package. `src/verdict.ts` and
18
+ * `src/rl/verifiable-reward.ts` both build on these types.
19
+ */
20
+ /**
21
+ * One member of the verification-strategy family.
22
+ *
23
+ * Every member has a documented failure mode — the specific way it can
24
+ * certify a wrong result. A consumer that reads a certification must weigh
25
+ * the member's failure mode, not treat "certified" as one bit.
26
+ *
27
+ * - `'compile'` — typecheck / build / lint passed. Deterministic.
28
+ * Failure mode: code that compiles is not code that is correct; the
29
+ * weakest deterministic member.
30
+ * - `'test'` — unit / integration / held-out suite pass-rate. Deterministic.
31
+ * Failure mode: assumes an answer key exists; certifies nothing outside
32
+ * the suite's coverage, and a stubbed integration reports green.
33
+ * - `'schema'` — structured output validates. Deterministic.
34
+ * Failure mode: shape is not meaning; a well-formed wrong answer passes.
35
+ * - `'sandbox'` — sandbox execution exit code. Deterministic.
36
+ * Failure mode: an exit code compresses the whole run to one bit; a
37
+ * faked success exits 0.
38
+ * - `'judge'` — LLM judge score. Probabilistic.
39
+ * Failure mode: drifts across model versions and is Goodhart-gameable by
40
+ * the graded policy.
41
+ * - `'composite'` — weighted blend across members. Determinism inherited
42
+ * from its members.
43
+ * Failure mode: scalar collapse — the blend hides which member carried
44
+ * the score and which failed.
45
+ * - `'proof-kernel'` — a proof assistant's kernel accepted a formal proof.
46
+ * Deterministic.
47
+ * Failure mode: the formalization gap. The kernel certifies the formal
48
+ * statement, never that the formal statement matches the informal claim;
49
+ * a proved theorem about the wrong statement is a wrong result with a
50
+ * valid certificate. Discharge it with the blind statement-equivalence
51
+ * protocol (`src/equivalence-check.ts`).
52
+ * - `'invariant'` — invariant / metamorphic properties held on the result.
53
+ * Deterministic.
54
+ * Failure mode: weak invariants pass everything. An invariant set earns
55
+ * weight only through seeded-bug calibration — a mutation the set
56
+ * provably rejects; an uncalibrated set is a rubber stamp.
57
+ * - `'replication'` — an independent re-execution reproduced the result
58
+ * from pinned inputs. Deterministic given the pins.
59
+ * Failure mode: replication re-runs the method, so it catches
60
+ * nondeterminism and environment drift, never a methodological error the
61
+ * method itself carries.
62
+ * - `'agreement'` — independently-derived results agree. Probabilistic.
63
+ * Failure mode: the shared blind spot. Derivers that share training
64
+ * corpora, priors, or the same misread of the source agree for the same
65
+ * wrong reason; the blindness provenance of each arm is what the
66
+ * certificate rests on.
67
+ */
68
+ type VerificationStrategySource = 'compile' | 'test' | 'schema' | 'sandbox' | 'judge' | 'composite' | 'proof-kernel' | 'invariant' | 'replication' | 'agreement';
69
+ /**
70
+ * What a strategy member can honestly claim about its outcomes.
71
+ *
72
+ * `determinism: 'inherited'` exists only for `'composite'`: a blend takes
73
+ * the class of its members. The reward axis stays binary
74
+ * (`VerifiableReward.determinism`); this registry describes the family,
75
+ * it never scores a run.
76
+ */
77
+ interface VerificationStrategyProfile {
78
+ /** Determinism class a well-formed checker of this member can claim. */
79
+ determinism: 'deterministic' | 'probabilistic' | 'inherited';
80
+ /** The documented way this member certifies a wrong result. */
81
+ failureMode: string;
82
+ }
83
+ /**
84
+ * The family registry. `Record` over the union keeps it exhaustive: adding
85
+ * a member to `VerificationStrategySource` without a profile here fails to
86
+ * compile. The failure mode travels with the taxonomy so a reader of a
87
+ * certification can surface it without this package's docs at hand.
88
+ */
89
+ declare const VERIFICATION_STRATEGIES: Record<VerificationStrategySource, VerificationStrategyProfile>;
90
+ /** Every family member, derived from the registry so it cannot drift. */
91
+ declare const VERIFICATION_STRATEGY_SOURCES: readonly VerificationStrategySource[];
92
+ /**
93
+ * Exact identity of the thing that checked. "lean4 4.33.0 with Mathlib
94
+ * db584cd6d46c" and "lean4, some version" are different certificates: the
95
+ * first is re-runnable, the second is a story.
96
+ */
97
+ interface CheckerIdentity {
98
+ /** Toolchain or model name, e.g. `'lean4'` or a judge model id. */
99
+ name: string;
100
+ /** Exact version or snapshot, e.g. `'4.33.0'` or a model snapshot id. */
101
+ version: string;
102
+ /**
103
+ * Content pins the version alone does not capture, e.g.
104
+ * `{ mathlib: 'db584cd6d46c' }`. Keyed by dependency name.
105
+ */
106
+ pins?: Record<string, string>;
107
+ }
108
+ /**
109
+ * Typed outcome of an external checker call. Callers MUST inspect
110
+ * `succeeded` before touching `value`; a failed check keeps its full error
111
+ * text instead of collapsing to null.
112
+ */
113
+ type CheckerOutcome<Result> = {
114
+ succeeded: true;
115
+ value: Result;
116
+ } | {
117
+ succeeded: false;
118
+ error: string;
119
+ };
120
+ /**
121
+ * The port shape every strategy binds through: an injected executable
122
+ * boundary. This package defines the port and refuses to ship any
123
+ * implementation behind it — the consumer binds its own kernel, invariant
124
+ * harness, replication runner, or judge, and the binding carries its own
125
+ * identity and determinism claim.
126
+ *
127
+ * A checker that ran but could not decide returns
128
+ * `{ succeeded: false, error }` with the reason; `succeeded: true` is
129
+ * reserved for a discharged check.
130
+ */
131
+ interface StrategyChecker<Input, Result> {
132
+ /** Which family member this checker discharges obligations for. */
133
+ strategy: VerificationStrategySource;
134
+ identity: CheckerIdentity;
135
+ /** Determinism class this checker claims for its outcomes. */
136
+ determinism: 'deterministic' | 'probabilistic';
137
+ check(input: Input): Promise<CheckerOutcome<Result>>;
138
+ }
139
+ //#endregion
140
+ //#region src/verdict.d.ts
141
+ /**
142
+ * What certified a verdict — the epistemics a bare `valid` + `score` pair
143
+ * cannot carry. A kernel-checked proof and an LLM judge can produce the
144
+ * same `{ valid: true, score: 1 }`; this record is what tells them apart.
145
+ *
146
+ * Every field is required on purpose. A certification without a checker
147
+ * identity or an evidence digest is exactly the unverifiable claim this
148
+ * record exists to kill.
149
+ */
150
+ interface VerdictCertification {
151
+ /**
152
+ * Which verification-strategy member produced the certificate. Each
153
+ * member's failure mode is documented on the family type and in
154
+ * `VERIFICATION_STRATEGIES` (src/verification-strategy.ts).
155
+ */
156
+ strategy: VerificationStrategySource;
157
+ /**
158
+ * Exact checker identity, e.g. `{ name: 'lean4', version: '4.33.0',
159
+ * pins: { mathlib: 'db584cd6d46c' } }` for a kernel, or a judge model
160
+ * id + snapshot for a judge.
161
+ */
162
+ checker: CheckerIdentity;
163
+ /**
164
+ * Every step the certificate rests on that the checker did NOT verify,
165
+ * e.g. 'diagonal embedding argued in prose' or 'agreement arms share
166
+ * training-corpus blind spots'. An empty array is the producer's
167
+ * explicit, auditable claim that there are none — it is not a default.
168
+ */
169
+ assumptions: string[];
170
+ /** Digest (e.g. sha256) of the evidence artifact the checker consumed or produced. */
171
+ evidenceDigest: string;
172
+ }
173
+ /**
174
+ * Minimal verdict shape — `valid` + `score` are required; `scores` +
175
+ * `notes` are optional surface. Validators that need richer shapes
176
+ * parameterise `Validator<Output, MyVerdict>` with their own type.
177
+ *
178
+ * Need structured extras? Extend DefaultVerdict with typed fields — never
179
+ * serialize extras into `notes`.
180
+ */
181
+ interface DefaultVerdict {
182
+ /** Whether the output meets the validator's pass criteria. */
183
+ valid: boolean;
184
+ /** Aggregate score in [0, 1]. Drivers use this for winner selection. */
185
+ score: number;
186
+ /** Per-dimension scores. Free-form; weighted into `score` by the validator. */
187
+ scores?: Record<string, number>;
188
+ /** Human-readable rationale; surfaces in trace + final-result `winner.verdict`. */
189
+ notes?: string;
190
+ /**
191
+ * What certified this verdict, when anything did. Absent = uncertified:
192
+ * scored, but no strategy vouches for it. A consumer that cares reads
193
+ * `certification.strategy` to tell a kernel-checked verdict from a
194
+ * judge-scored one; a consumer that does not read it sees the exact
195
+ * verdict it always saw.
196
+ */
197
+ certification?: VerdictCertification;
198
+ }
199
+ //#endregion
200
+ export { StrategyChecker as a, VerificationStrategyProfile as c, CheckerOutcome as i, VerificationStrategySource as l, VerdictCertification as n, VERIFICATION_STRATEGIES as o, CheckerIdentity as r, VERIFICATION_STRATEGY_SOURCES as s, DefaultVerdict as t };
201
+ //# sourceMappingURL=verdict-DExhxfgR.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"verdict-DExhxfgR.d.ts","names":[],"sources":["../src/verification-strategy.ts","../src/verdict.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;KAmEY;;;;;;;;;UAoBK;;EAEf;;EAEA;;;;;;;;cASW,yBAAyB,OACpC,4BACA;;cAkDW,wCAEC;;;;;;UAOG;;EAEf;;EAEA;;;;;EAKA,OAAO;;;;;;;KAQG,eAAe;EACrB;EAAiB,OAAO;;EACxB;EAAkB;;;;;;;;;;;;;UAaP,gBAAgB,OAAO;;EAEtC,UAAU;EACV,UAAU;;EAEV;EACA,MAAM,OAAO,QAAQ,QAAQ,eAAe;;;;;;;;;;;;;UC5K7B;;;;;;EAMf,UAAU;;;;;;EAMV,SAAS;;;;;;;EAOT;;EAEA;;;;;;;;;;UAWe;;EAEf;;EAEA;;EAEA,SAAS;;EAET;;;;;;;;EAQA,gBAAgB"}
@@ -0,0 +1,159 @@
1
+ import { appendFileSync, existsSync, readFileSync } from "node:fs";
2
+ import { createHash } from "node:crypto";
3
+ //#region src/verdict-cache.ts
4
+ /**
5
+ * Content-addressed judge-verdict caching.
6
+ *
7
+ * LAW: cache JUDGE VERDICTS only — judging the same artifact with the same
8
+ * judge+rubric is pure. NEVER cache agent rollouts. (A router that cached
9
+ * identical fanout prompts silently destroyed best-of-N diversity; rollout
10
+ * caching reintroduces that failure class. Judging has no diversity to
11
+ * destroy — same artifact + same rubric ⇒ same verdict is the desired
12
+ * property, not a bug.)
13
+ *
14
+ * The cache key is a sha-256 over the canonical JSON of everything that can
15
+ * change a verdict: the artifact content, the scenario id, the judge name,
16
+ * the full dimension list (key + description — the description IS the rubric
17
+ * text shown to the judge), and a caller-supplied `judgeVersion`.
18
+ * `judgeVersion` is REQUIRED: a judge whose prompt/model/ensemble changes
19
+ * without a version bump would otherwise silently serve stale verdicts.
20
+ *
21
+ * Strict canonicalization (`canonicalJson`) throws on undefined / function /
22
+ * symbol / non-finite numbers — an artifact that cannot be unambiguously
23
+ * serialized cannot be content-addressed, and coercing it would let two
24
+ * different artifacts collide on one key.
25
+ */
26
+ function canonicalizeAt(value, path) {
27
+ if (value === null) return "null";
28
+ switch (typeof value) {
29
+ case "boolean": return value ? "true" : "false";
30
+ case "number":
31
+ if (!Number.isFinite(value)) throw new Error(`canonicalJson: non-finite number (${value}) at ${path} — ambiguity is an error, not a coercion`);
32
+ return JSON.stringify(value);
33
+ case "string": return JSON.stringify(value);
34
+ case "undefined":
35
+ case "function":
36
+ case "symbol": throw new Error(`canonicalJson: ${typeof value} at ${path} — ambiguity is an error, not a coercion`);
37
+ case "bigint": throw new Error(`canonicalJson: bigint at ${path} — not representable in JSON`);
38
+ case "object": break;
39
+ }
40
+ const obj = value;
41
+ if (typeof obj.toJSON === "function") return canonicalizeAt(obj.toJSON(), path);
42
+ if (Array.isArray(obj)) return `[${obj.map((item, i) => canonicalizeAt(item, `${path}[${i}]`)).join(",")}]`;
43
+ if (obj instanceof Map || obj instanceof Set) throw new Error(`canonicalJson: ${obj instanceof Map ? "Map" : "Set"} at ${path} — would serialize as '{}'; convert to a plain object/array first`);
44
+ return `{${Object.keys(obj).sort().map((k) => `${JSON.stringify(k)}:${canonicalizeAt(obj[k], `${path}.${k}`)}`).join(",")}}`;
45
+ }
46
+ /**
47
+ * Stable JSON stringify: object keys sorted recursively, so two semantically
48
+ * equal values produce byte-identical output regardless of key insertion
49
+ * order. Throws on undefined / function / symbol / NaN / ±Infinity / bigint /
50
+ * Map / Set — anything JSON.stringify would coerce or drop silently.
51
+ *
52
+ * Distinct from `pre-registration.ts`'s `canonicalize`/`hashJson`, which are
53
+ * permissive (coercion allowed) and async (web-crypto). Use THIS pair when a
54
+ * hash collision or silent coercion would corrupt a cache key or attestation.
55
+ */
56
+ function canonicalJson(value) {
57
+ return canonicalizeAt(value, "$");
58
+ }
59
+ /** Hex sha-256 over `canonicalJson(value)`. The content address used by the
60
+ * verdict cache and report attestation. */
61
+ function contentHash(value) {
62
+ return createHash("sha256").update(canonicalJson(value)).digest("hex");
63
+ }
64
+ /** Process-local Map-backed store. */
65
+ function inMemoryVerdictCache() {
66
+ const entries = /* @__PURE__ */ new Map();
67
+ return {
68
+ get: (key) => entries.get(key),
69
+ set: (key, score) => {
70
+ entries.set(key, score);
71
+ }
72
+ };
73
+ }
74
+ function parseCacheLine(line, path, lineNo) {
75
+ let parsed;
76
+ try {
77
+ parsed = JSON.parse(line);
78
+ } catch (err) {
79
+ throw new Error(`fileVerdictCache: corrupt JSONL at ${path}:${lineNo} — ${err instanceof Error ? err.message : String(err)}`);
80
+ }
81
+ const rec = parsed;
82
+ if (typeof rec !== "object" || rec === null || typeof rec.key !== "string" || typeof rec.score !== "object" || rec.score === null || typeof rec.score.composite !== "number" || typeof rec.score.dimensions !== "object") throw new Error(`fileVerdictCache: invalid record shape at ${path}:${lineNo} — expected {key, score:{dimensions, composite, notes}}`);
83
+ return rec;
84
+ }
85
+ /**
86
+ * JSONL-file-backed store: the full file is loaded into an in-memory index at
87
+ * construction; every `set` appends one line synchronously (durable before
88
+ * the verdict is returned). A corrupt or malformed line throws at load with
89
+ * file:line — a skipped line would silently re-judge (cost) or, worse, mask
90
+ * a half-written file that needs operator attention.
91
+ */
92
+ function fileVerdictCache(path) {
93
+ const entries = /* @__PURE__ */ new Map();
94
+ if (existsSync(path)) {
95
+ const lines = readFileSync(path, "utf8").split("\n");
96
+ for (let i = 0; i < lines.length; i++) {
97
+ const line = lines[i];
98
+ if (line === void 0 || line.trim() === "") continue;
99
+ const rec = parseCacheLine(line, path, i + 1);
100
+ entries.set(rec.key, rec.score);
101
+ }
102
+ }
103
+ return {
104
+ get: (key) => entries.get(key),
105
+ set: (key, score) => {
106
+ appendFileSync(path, `${JSON.stringify({
107
+ key,
108
+ score
109
+ })}\n`, "utf8");
110
+ entries.set(key, score);
111
+ }
112
+ };
113
+ }
114
+ /**
115
+ * Wrap a `JudgeConfig` so repeat judgments of the same artifact are served
116
+ * from the store instead of re-invoking `score()`. The wrapper is generic
117
+ * over the judge's own type parameters and preserves `appliesTo` — it is a
118
+ * drop-in replacement anywhere a `JudgeConfig` is accepted.
119
+ *
120
+ * A judge that throws is NOT cached: the error propagates and the next
121
+ * attempt re-judges (caching a failure would pin a transient outage forever).
122
+ */
123
+ function cachedJudge(judge, store, options) {
124
+ if (typeof options.judgeVersion !== "string" || options.judgeVersion.trim() === "") throw new Error("cachedJudge: judgeVersion is required and must be a non-empty string");
125
+ const stats = {
126
+ hits: 0,
127
+ misses: 0
128
+ };
129
+ const wrapped = {
130
+ name: judge.name,
131
+ dimensions: judge.dimensions,
132
+ judgeVersion: options.judgeVersion,
133
+ async score(input) {
134
+ const key = contentHash({
135
+ artifact: canonicalJson(input.artifact),
136
+ scenarioId: input.scenario.id,
137
+ judgeName: judge.name,
138
+ dimensions: judge.dimensions,
139
+ judgeVersion: options.judgeVersion
140
+ });
141
+ const cached = await store.get(key);
142
+ if (cached !== void 0) {
143
+ stats.hits += 1;
144
+ return cached;
145
+ }
146
+ const score = await judge.score(input);
147
+ await store.set(key, score);
148
+ stats.misses += 1;
149
+ return score;
150
+ },
151
+ stats: () => ({ ...stats })
152
+ };
153
+ if (judge.appliesTo) wrapped.appliesTo = judge.appliesTo;
154
+ return wrapped;
155
+ }
156
+ //#endregion
157
+ export { inMemoryVerdictCache as a, fileVerdictCache as i, canonicalJson as n, contentHash as r, cachedJudge as t };
158
+
159
+ //# sourceMappingURL=verdict-cache-BCcOh0kF.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"verdict-cache-BCcOh0kF.js","names":[],"sources":["../src/verdict-cache.ts"],"sourcesContent":["/**\n * Content-addressed judge-verdict caching.\n *\n * LAW: cache JUDGE VERDICTS only — judging the same artifact with the same\n * judge+rubric is pure. NEVER cache agent rollouts. (A router that cached\n * identical fanout prompts silently destroyed best-of-N diversity; rollout\n * caching reintroduces that failure class. Judging has no diversity to\n * destroy — same artifact + same rubric ⇒ same verdict is the desired\n * property, not a bug.)\n *\n * The cache key is a sha-256 over the canonical JSON of everything that can\n * change a verdict: the artifact content, the scenario id, the judge name,\n * the full dimension list (key + description — the description IS the rubric\n * text shown to the judge), and a caller-supplied `judgeVersion`.\n * `judgeVersion` is REQUIRED: a judge whose prompt/model/ensemble changes\n * without a version bump would otherwise silently serve stale verdicts.\n *\n * Strict canonicalization (`canonicalJson`) throws on undefined / function /\n * symbol / non-finite numbers — an artifact that cannot be unambiguously\n * serialized cannot be content-addressed, and coercing it would let two\n * different artifacts collide on one key.\n */\n\nimport { createHash } from 'node:crypto'\nimport { appendFileSync, existsSync, readFileSync } from 'node:fs'\nimport type { JudgeConfig, JudgeScore, Scenario } from './campaign/types'\n\n// ── canonical JSON + content hash ─────────────────────────────────────────\n\nfunction canonicalizeAt(value: unknown, path: string): string {\n if (value === null) return 'null'\n switch (typeof value) {\n case 'boolean':\n return value ? 'true' : 'false'\n case 'number':\n if (!Number.isFinite(value)) {\n throw new Error(\n `canonicalJson: non-finite number (${value}) at ${path} — ambiguity is an error, not a coercion`,\n )\n }\n return JSON.stringify(value)\n case 'string':\n return JSON.stringify(value)\n case 'undefined':\n case 'function':\n case 'symbol':\n throw new Error(\n `canonicalJson: ${typeof value} at ${path} — ambiguity is an error, not a coercion`,\n )\n case 'bigint':\n throw new Error(`canonicalJson: bigint at ${path} — not representable in JSON`)\n case 'object':\n break\n }\n const obj = value as Record<string, unknown>\n // Honor toJSON (Date → ISO string) before structural checks — without it a\n // Date would canonicalize to '{}' and every timestamp would collide.\n if (typeof obj.toJSON === 'function') {\n return canonicalizeAt((obj as { toJSON(): unknown }).toJSON(), path)\n }\n if (Array.isArray(obj)) {\n return `[${obj.map((item, i) => canonicalizeAt(item, `${path}[${i}]`)).join(',')}]`\n }\n if (obj instanceof Map || obj instanceof Set) {\n throw new Error(\n `canonicalJson: ${obj instanceof Map ? 'Map' : 'Set'} at ${path} — would serialize as '{}'; convert to a plain object/array first`,\n )\n }\n const keys = Object.keys(obj).sort()\n const parts = keys.map((k) => `${JSON.stringify(k)}:${canonicalizeAt(obj[k], `${path}.${k}`)}`)\n return `{${parts.join(',')}}`\n}\n\n/**\n * Stable JSON stringify: object keys sorted recursively, so two semantically\n * equal values produce byte-identical output regardless of key insertion\n * order. Throws on undefined / function / symbol / NaN / ±Infinity / bigint /\n * Map / Set — anything JSON.stringify would coerce or drop silently.\n *\n * Distinct from `pre-registration.ts`'s `canonicalize`/`hashJson`, which are\n * permissive (coercion allowed) and async (web-crypto). Use THIS pair when a\n * hash collision or silent coercion would corrupt a cache key or attestation.\n */\nexport function canonicalJson(value: unknown): string {\n return canonicalizeAt(value, '$')\n}\n\n/** Hex sha-256 over `canonicalJson(value)`. The content address used by the\n * verdict cache and report attestation. */\nexport function contentHash(value: unknown): string {\n return createHash('sha256').update(canonicalJson(value)).digest('hex')\n}\n\n// ── store contract ─────────────────────────────────────────────────────────\n\n/** Pluggable verdict store. Sync or async on both legs — `cachedJudge`\n * awaits the results either way. */\nexport interface VerdictCacheStore {\n get(key: string): Promise<JudgeScore | undefined> | JudgeScore | undefined\n set(key: string, score: JudgeScore): Promise<void> | void\n}\n\n/** Process-local Map-backed store. */\nexport function inMemoryVerdictCache(): VerdictCacheStore {\n const entries = new Map<string, JudgeScore>()\n return {\n get: (key) => entries.get(key),\n set: (key, score) => {\n entries.set(key, score)\n },\n }\n}\n\ninterface VerdictCacheLine {\n key: string\n score: JudgeScore\n}\n\nfunction parseCacheLine(line: string, path: string, lineNo: number): VerdictCacheLine {\n let parsed: unknown\n try {\n parsed = JSON.parse(line)\n } catch (err) {\n throw new Error(\n `fileVerdictCache: corrupt JSONL at ${path}:${lineNo} — ${err instanceof Error ? err.message : String(err)}`,\n )\n }\n const rec = parsed as Partial<VerdictCacheLine>\n if (\n typeof rec !== 'object' ||\n rec === null ||\n typeof rec.key !== 'string' ||\n typeof rec.score !== 'object' ||\n rec.score === null ||\n typeof rec.score.composite !== 'number' ||\n typeof rec.score.dimensions !== 'object'\n ) {\n throw new Error(\n `fileVerdictCache: invalid record shape at ${path}:${lineNo} — expected {key, score:{dimensions, composite, notes}}`,\n )\n }\n return rec as VerdictCacheLine\n}\n\n/**\n * JSONL-file-backed store: the full file is loaded into an in-memory index at\n * construction; every `set` appends one line synchronously (durable before\n * the verdict is returned). A corrupt or malformed line throws at load with\n * file:line — a skipped line would silently re-judge (cost) or, worse, mask\n * a half-written file that needs operator attention.\n */\nexport function fileVerdictCache(path: string): VerdictCacheStore {\n const entries = new Map<string, JudgeScore>()\n if (existsSync(path)) {\n const lines = readFileSync(path, 'utf8').split('\\n')\n for (let i = 0; i < lines.length; i++) {\n const line = lines[i]\n if (line === undefined || line.trim() === '') continue\n const rec = parseCacheLine(line, path, i + 1)\n entries.set(rec.key, rec.score)\n }\n }\n return {\n get: (key) => entries.get(key),\n set: (key, score) => {\n appendFileSync(path, `${JSON.stringify({ key, score })}\\n`, 'utf8')\n entries.set(key, score)\n },\n }\n}\n\n// ── cached judge wrapper ───────────────────────────────────────────────────\n\nexport interface VerdictCacheStats {\n hits: number\n misses: number\n}\n\nexport interface CachedJudgeOptions {\n /** REQUIRED — part of the cache key. Bump on any change to the judge's\n * prompt, model, ensemble, or scoring logic; silent judge upgrades must\n * never serve stale verdicts. */\n judgeVersion: string\n}\n\n/** The wrapped judge: same `JudgeConfig` seam, plus hit/miss observability. */\nexport type CachedJudge<TArtifact, TScenario extends Scenario = Scenario> = JudgeConfig<\n TArtifact,\n TScenario\n> & {\n stats(): VerdictCacheStats\n}\n\n/**\n * Wrap a `JudgeConfig` so repeat judgments of the same artifact are served\n * from the store instead of re-invoking `score()`. The wrapper is generic\n * over the judge's own type parameters and preserves `appliesTo` — it is a\n * drop-in replacement anywhere a `JudgeConfig` is accepted.\n *\n * A judge that throws is NOT cached: the error propagates and the next\n * attempt re-judges (caching a failure would pin a transient outage forever).\n */\nexport function cachedJudge<TArtifact, TScenario extends Scenario = Scenario>(\n judge: JudgeConfig<TArtifact, TScenario>,\n store: VerdictCacheStore,\n options: CachedJudgeOptions,\n): CachedJudge<TArtifact, TScenario> {\n if (typeof options.judgeVersion !== 'string' || options.judgeVersion.trim() === '') {\n throw new Error('cachedJudge: judgeVersion is required and must be a non-empty string')\n }\n const stats: VerdictCacheStats = { hits: 0, misses: 0 }\n const wrapped: CachedJudge<TArtifact, TScenario> = {\n name: judge.name,\n dimensions: judge.dimensions,\n judgeVersion: options.judgeVersion,\n async score(input) {\n const key = contentHash({\n artifact: canonicalJson(input.artifact),\n scenarioId: input.scenario.id,\n judgeName: judge.name,\n dimensions: judge.dimensions,\n judgeVersion: options.judgeVersion,\n })\n const cached = await store.get(key)\n if (cached !== undefined) {\n stats.hits += 1\n return cached\n }\n const score = await judge.score(input)\n await store.set(key, score)\n stats.misses += 1\n return score\n },\n stats: () => ({ ...stats }),\n }\n if (judge.appliesTo) wrapped.appliesTo = judge.appliesTo\n return wrapped\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;AA6BA,SAAS,eAAe,OAAgB,MAAsB;CAC5D,IAAI,UAAU,MAAM,OAAO;CAC3B,QAAQ,OAAO,OAAf;EACE,KAAK,WACH,OAAO,QAAQ,SAAS;EAC1B,KAAK;GACH,IAAI,CAAC,OAAO,SAAS,KAAK,GACxB,MAAM,IAAI,MACR,qCAAqC,MAAM,OAAO,KAAK,yCACzD;GAEF,OAAO,KAAK,UAAU,KAAK;EAC7B,KAAK,UACH,OAAO,KAAK,UAAU,KAAK;EAC7B,KAAK;EACL,KAAK;EACL,KAAK,UACH,MAAM,IAAI,MACR,kBAAkB,OAAO,MAAM,MAAM,KAAK,yCAC5C;EACF,KAAK,UACH,MAAM,IAAI,MAAM,4BAA4B,KAAK,6BAA6B;EAChF,KAAK,UACH;CACJ;CACA,MAAM,MAAM;CAGZ,IAAI,OAAO,IAAI,WAAW,YACxB,OAAO,eAAgB,IAA8B,OAAO,GAAG,IAAI;CAErE,IAAI,MAAM,QAAQ,GAAG,GACnB,OAAO,IAAI,IAAI,KAAK,MAAM,MAAM,eAAe,MAAM,GAAG,KAAK,GAAG,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,GAAG,EAAE;CAEnF,IAAI,eAAe,OAAO,eAAe,KACvC,MAAM,IAAI,MACR,kBAAkB,eAAe,MAAM,QAAQ,MAAM,MAAM,KAAK,kEAClE;CAIF,OAAO,IAFM,OAAO,KAAK,GAAG,CAAC,CAAC,KACb,CAAC,CAAC,KAAK,MAAM,GAAG,KAAK,UAAU,CAAC,EAAE,GAAG,eAAe,IAAI,IAAI,GAAG,KAAK,GAAG,GAAG,GAC5E,CAAC,CAAC,KAAK,GAAG,EAAE;AAC7B;;;;;;;;;;;AAYA,SAAgB,cAAc,OAAwB;CACpD,OAAO,eAAe,OAAO,GAAG;AAClC;;;AAIA,SAAgB,YAAY,OAAwB;CAClD,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,cAAc,KAAK,CAAC,CAAC,CAAC,OAAO,KAAK;AACvE;;AAYA,SAAgB,uBAA0C;CACxD,MAAM,0BAAU,IAAI,IAAwB;CAC5C,OAAO;EACL,MAAM,QAAQ,QAAQ,IAAI,GAAG;EAC7B,MAAM,KAAK,UAAU;GACnB,QAAQ,IAAI,KAAK,KAAK;EACxB;CACF;AACF;AAOA,SAAS,eAAe,MAAc,MAAc,QAAkC;CACpF,IAAI;CACJ,IAAI;EACF,SAAS,KAAK,MAAM,IAAI;CAC1B,SAAS,KAAK;EACZ,MAAM,IAAI,MACR,sCAAsC,KAAK,GAAG,OAAO,KAAK,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG,GAC3G;CACF;CACA,MAAM,MAAM;CACZ,IACE,OAAO,QAAQ,YACf,QAAQ,QACR,OAAO,IAAI,QAAQ,YACnB,OAAO,IAAI,UAAU,YACrB,IAAI,UAAU,QACd,OAAO,IAAI,MAAM,cAAc,YAC/B,OAAO,IAAI,MAAM,eAAe,UAEhC,MAAM,IAAI,MACR,6CAA6C,KAAK,GAAG,OAAO,wDAC9D;CAEF,OAAO;AACT;;;;;;;;AASA,SAAgB,iBAAiB,MAAiC;CAChE,MAAM,0BAAU,IAAI,IAAwB;CAC5C,IAAI,WAAW,IAAI,GAAG;EACpB,MAAM,QAAQ,aAAa,MAAM,MAAM,CAAC,CAAC,MAAM,IAAI;EACnD,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;GACrC,MAAM,OAAO,MAAM;GACnB,IAAI,SAAS,KAAA,KAAa,KAAK,KAAK,MAAM,IAAI;GAC9C,MAAM,MAAM,eAAe,MAAM,MAAM,IAAI,CAAC;GAC5C,QAAQ,IAAI,IAAI,KAAK,IAAI,KAAK;EAChC;CACF;CACA,OAAO;EACL,MAAM,QAAQ,QAAQ,IAAI,GAAG;EAC7B,MAAM,KAAK,UAAU;GACnB,eAAe,MAAM,GAAG,KAAK,UAAU;IAAE;IAAK;GAAM,CAAC,EAAE,KAAK,MAAM;GAClE,QAAQ,IAAI,KAAK,KAAK;EACxB;CACF;AACF;;;;;;;;;;AAiCA,SAAgB,YACd,OACA,OACA,SACmC;CACnC,IAAI,OAAO,QAAQ,iBAAiB,YAAY,QAAQ,aAAa,KAAK,MAAM,IAC9E,MAAM,IAAI,MAAM,sEAAsE;CAExF,MAAM,QAA2B;EAAE,MAAM;EAAG,QAAQ;CAAE;CACtD,MAAM,UAA6C;EACjD,MAAM,MAAM;EACZ,YAAY,MAAM;EAClB,cAAc,QAAQ;EACtB,MAAM,MAAM,OAAO;GACjB,MAAM,MAAM,YAAY;IACtB,UAAU,cAAc,MAAM,QAAQ;IACtC,YAAY,MAAM,SAAS;IAC3B,WAAW,MAAM;IACjB,YAAY,MAAM;IAClB,cAAc,QAAQ;GACxB,CAAC;GACD,MAAM,SAAS,MAAM,MAAM,IAAI,GAAG;GAClC,IAAI,WAAW,KAAA,GAAW;IACxB,MAAM,QAAQ;IACd,OAAO;GACT;GACA,MAAM,QAAQ,MAAM,MAAM,MAAM,KAAK;GACrC,MAAM,MAAM,IAAI,KAAK,KAAK;GAC1B,MAAM,UAAU;GAChB,OAAO;EACT;EACA,cAAc,EAAE,GAAG,MAAM;CAC3B;CACA,IAAI,MAAM,WAAW,QAAQ,YAAY,MAAM;CAC/C,OAAO;AACT"}
@@ -1,7 +1,7 @@
1
1
  import { c as CostLedgerHandle } from "../cost-ledger-Bv_e8XHY.js";
2
- import { Z as LlmRouteRequirements, q as LlmClientOptions } from "../types-XMVEdrE_.js";
2
+ import { Z as LlmRouteRequirements, q as LlmClientOptions } from "../types-D216SgwM.js";
3
3
  import { s as TraceStore } from "../store-CT9YIIve.js";
4
- import { _ as FeedbackTrajectoryStore } from "../feedback-trajectory-GgoS0-MK.js";
4
+ import { _ as FeedbackTrajectoryStore } from "../feedback-trajectory-Rh280oXo.js";
5
5
  import { z } from "zod";
6
6
  import { ServerType } from "@hono/node-server";
7
7
  import { Hono } from "hono";
@@ -1,2 +1,2 @@
1
- import { A as TraceEventSchema, C as HealthResponseSchema, D as RubricDimensionSchema, E as ListRubricsResponseSchema, F as hashRubric, M as TracesIngestResponseSchema, N as VersionResponseSchema, O as RubricInfoSchema, P as WIRE_VERSION, S as FeedbackTrajectorySchema, T as JudgeResultSchema, _ as ErrorResponseSchema, a as runRpcBatch, b as FeedbackIngestResponseSchema, c as WireError, d as handleListRubrics, f as handleTracesIngest, g as listBuiltinRubrics, h as getBuiltinRubric, i as dispatchRpc, j as TracesIngestRequestSchema, k as RubricSchema, l as handleFeedbackIngest, m as BUILTIN_RUBRICS, n as startServer, o as runRpcOnce, p as handleVersion, r as startServerAsync, s as buildOpenApi, t as createApp, u as handleJudge, v as FailureModeSchema, w as JudgeRequestSchema, x as FeedbackLabelSchema, y as FeedbackAttemptSchema } from "../server-D6XJQHw7.js";
1
+ import { A as TraceEventSchema, C as HealthResponseSchema, D as RubricDimensionSchema, E as ListRubricsResponseSchema, F as hashRubric, M as TracesIngestResponseSchema, N as VersionResponseSchema, O as RubricInfoSchema, P as WIRE_VERSION, S as FeedbackTrajectorySchema, T as JudgeResultSchema, _ as ErrorResponseSchema, a as runRpcBatch, b as FeedbackIngestResponseSchema, c as WireError, d as handleListRubrics, f as handleTracesIngest, g as listBuiltinRubrics, h as getBuiltinRubric, i as dispatchRpc, j as TracesIngestRequestSchema, k as RubricSchema, l as handleFeedbackIngest, m as BUILTIN_RUBRICS, n as startServer, o as runRpcOnce, p as handleVersion, r as startServerAsync, s as buildOpenApi, t as createApp, u as handleJudge, v as FailureModeSchema, w as JudgeRequestSchema, x as FeedbackLabelSchema, y as FeedbackAttemptSchema } from "../server-iu0ede49.js";
2
2
  export { BUILTIN_RUBRICS, ErrorResponseSchema, FailureModeSchema, FeedbackAttemptSchema, FeedbackIngestResponseSchema, FeedbackLabelSchema, FeedbackTrajectorySchema, HealthResponseSchema, JudgeRequestSchema, JudgeResultSchema, ListRubricsResponseSchema, RubricDimensionSchema, RubricInfoSchema, RubricSchema, TraceEventSchema, TracesIngestRequestSchema, TracesIngestResponseSchema, VersionResponseSchema, WIRE_VERSION, WireError, buildOpenApi, createApp, dispatchRpc, getBuiltinRubric, handleFeedbackIngest, handleJudge, handleListRubrics, handleTracesIngest, handleVersion, hashRubric, listBuiltinRubrics, runRpcBatch, runRpcOnce, startServer, startServerAsync };
@@ -443,6 +443,11 @@ const proposer: SurfaceProposer = {
443
443
  Return a label and rationale when they will help later analysis.
444
444
  Candidate creation must not read final test results.
445
445
 
446
+ `runOptimization()` rejects a candidate whose `surfaceHash` was already admitted.
447
+ This includes the baseline, an earlier generation, and another candidate in the same proposal.
448
+ The complete proposal is checked before candidate dispatch, so duplicates cannot consume candidate cells.
449
+ Use `reps` when one surface needs repeated measurements.
450
+
446
451
  ## Data And Cost Rules
447
452
 
448
453
  - Train and selection cases are visible to complete optimization methods.