@tangle-network/agent-eval 0.172.1 → 0.173.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (219) hide show
  1. package/CHANGELOG.md +19 -0
  2. package/README.md +17 -2
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/analyst/index.d.ts +13 -13
  5. package/dist/analyst/index.js +7 -6
  6. package/dist/analyst/index.js.map +1 -1
  7. package/dist/{attestation-CJBGmMVh.d.ts → attestation-c1QvaBdX.d.ts} +2 -2
  8. package/dist/{attestation-CJBGmMVh.d.ts.map → attestation-c1QvaBdX.d.ts.map} +1 -1
  9. package/dist/{backend-integrity-e79K3UPD.d.ts → backend-integrity-CeuTgqsd.d.ts} +3 -4
  10. package/dist/backend-integrity-CeuTgqsd.d.ts.map +1 -0
  11. package/dist/{benchmark-h-h4bfqj.d.ts → benchmark-BjLGkfnN.d.ts} +3 -3
  12. package/dist/{benchmark-h-h4bfqj.d.ts.map → benchmark-BjLGkfnN.d.ts.map} +1 -1
  13. package/dist/{benchmark-command--qeZUHbu.js → benchmark-command-9S20PRel.js} +9 -9
  14. package/dist/{benchmark-command--qeZUHbu.js.map → benchmark-command-9S20PRel.js.map} +1 -1
  15. package/dist/benchmarks/index.d.ts +5 -5
  16. package/dist/benchmarks/index.js +3 -3
  17. package/dist/{bounded-process-VIi0KSL2.js → bounded-process-BBZob7vl.js} +30 -7
  18. package/dist/bounded-process-BBZob7vl.js.map +1 -0
  19. package/dist/builder-eval/index.d.ts +3 -3
  20. package/dist/builder-eval/index.js +2 -2
  21. package/dist/campaign/index.d.ts +9 -9
  22. package/dist/campaign/index.js +6 -6
  23. package/dist/{campaign-Dp35pBbS.js → campaign-pxS0wmo4.js} +8 -8
  24. package/dist/{campaign-Dp35pBbS.js.map → campaign-pxS0wmo4.js.map} +1 -1
  25. package/dist/chat-client-Db4bqYfA.js +115 -0
  26. package/dist/chat-client-Db4bqYfA.js.map +1 -0
  27. package/dist/{chat-json-call-5Jxna-aV.js → chat-json-call-C26igCih.js} +16 -6
  28. package/dist/chat-json-call-C26igCih.js.map +1 -0
  29. package/dist/cli.js +31 -17
  30. package/dist/cli.js.map +1 -1
  31. package/dist/{client-BvwNkIRN.js → client-BlLY6o2w.js} +2 -2
  32. package/dist/{client-BvwNkIRN.js.map → client-BlLY6o2w.js.map} +1 -1
  33. package/dist/{client-Df7wdslk.d.ts → client-DlqdbM7n.d.ts} +4 -4
  34. package/dist/{client-Df7wdslk.d.ts.map → client-DlqdbM7n.d.ts.map} +1 -1
  35. package/dist/contract/index.d.ts +13 -13
  36. package/dist/contract/index.js +202 -10
  37. package/dist/contract/index.js.map +1 -1
  38. package/dist/{counterfactual-Bee5_BIn.d.ts → counterfactual-CLgrwhkY.d.ts} +4 -4
  39. package/dist/{counterfactual-Bee5_BIn.d.ts.map → counterfactual-CLgrwhkY.d.ts.map} +1 -1
  40. package/dist/{chat-client-DI79OPye.js → default-registry-B0bKikCb.js} +3 -39
  41. package/dist/default-registry-B0bKikCb.js.map +1 -0
  42. package/dist/{default-registry-XxedTLwu.d.ts → default-registry-BKwc8bN5.d.ts} +6 -6
  43. package/dist/{default-registry-XxedTLwu.d.ts.map → default-registry-BKwc8bN5.d.ts.map} +1 -1
  44. package/dist/{define-agent-eval-0wW7gFhr.d.ts → define-agent-eval-CY6qdlGV.d.ts} +6 -6
  45. package/dist/{define-agent-eval-0wW7gFhr.d.ts.map → define-agent-eval-CY6qdlGV.d.ts.map} +1 -1
  46. package/dist/{define-agent-eval-jS8xj_Q_.js → define-agent-eval-D_i_s69h.js} +7 -7
  47. package/dist/{define-agent-eval-jS8xj_Q_.js.map → define-agent-eval-D_i_s69h.js.map} +1 -1
  48. package/dist/{dspy-rlm-engine-CS3qcCEk.js → dspy-rlm-engine-D5byiHn9.js} +5 -6
  49. package/dist/dspy-rlm-engine-D5byiHn9.js.map +1 -0
  50. package/dist/{emitter-Bvnu0VzL.d.ts → emitter-Cs0egaFd.d.ts} +3 -3
  51. package/dist/{emitter-Bvnu0VzL.d.ts.map → emitter-Cs0egaFd.d.ts.map} +1 -1
  52. package/dist/{engine-BfRay1qD.d.ts → engine-DhFir3Ys.d.ts} +23 -8
  53. package/dist/{engine-BfRay1qD.d.ts.map → engine-DhFir3Ys.d.ts.map} +1 -1
  54. package/dist/{eval-campaign-JDTeE6Pl.js → eval-campaign-BeAjdhzC.js} +2 -2
  55. package/dist/{eval-campaign-JDTeE6Pl.js.map → eval-campaign-BeAjdhzC.js.map} +1 -1
  56. package/dist/{exact-types-BEecmnWm.d.ts → exact-types-BKOEILRP.d.ts} +2 -2
  57. package/dist/{exact-types-BEecmnWm.d.ts.map → exact-types-BKOEILRP.d.ts.map} +1 -1
  58. package/dist/experiment/index.d.ts +4 -4
  59. package/dist/{external-optimizer-contracts-szBJ_1vh.d.ts → external-optimizer-contracts-CQCpyrIL.d.ts} +2 -2
  60. package/dist/{external-optimizer-contracts-szBJ_1vh.d.ts.map → external-optimizer-contracts-CQCpyrIL.d.ts.map} +1 -1
  61. package/dist/{external-optimizer-process-CQxylYeG.js → external-optimizer-process-Cq_Pg15r.js} +2 -2
  62. package/dist/{external-optimizer-process-CQxylYeG.js.map → external-optimizer-process-Cq_Pg15r.js.map} +1 -1
  63. package/dist/{external-optimizer-subprocess-Cex8Da2i.js → external-optimizer-subprocess-DgNebftP.js} +2 -2
  64. package/dist/{external-optimizer-subprocess-Cex8Da2i.js.map → external-optimizer-subprocess-DgNebftP.js.map} +1 -1
  65. package/dist/failure-cluster-OldNRoAt.d.ts +154 -0
  66. package/dist/failure-cluster-OldNRoAt.d.ts.map +1 -0
  67. package/dist/{feedback-trajectory-DIqpCyF0.d.ts → feedback-trajectory-CMnv_uYs.d.ts} +6 -6
  68. package/dist/{feedback-trajectory-DIqpCyF0.d.ts.map → feedback-trajectory-CMnv_uYs.d.ts.map} +1 -1
  69. package/dist/{heldout-gate-JgNRDZwZ.d.ts → heldout-gate-Df5hsqmm.d.ts} +7 -7
  70. package/dist/{heldout-gate-JgNRDZwZ.d.ts.map → heldout-gate-Df5hsqmm.d.ts.map} +1 -1
  71. package/dist/hosted/index.d.ts +2 -2
  72. package/dist/hosted/index.js +1 -1
  73. package/dist/{index-DnglhM0A.d.ts → index-BQqOjerE.d.ts} +94 -13
  74. package/dist/index-BQqOjerE.d.ts.map +1 -0
  75. package/dist/{index-_vPrVMRX.d.ts → index-CFDffsKz.d.ts} +11 -11
  76. package/dist/{index-_vPrVMRX.d.ts.map → index-CFDffsKz.d.ts.map} +1 -1
  77. package/dist/{index-DDAPhUJJ.d.ts → index-D0Db5X-4.d.ts} +45 -9
  78. package/dist/index-D0Db5X-4.d.ts.map +1 -0
  79. package/dist/{index-DMoxLG8P.d.ts → index-e7LXeRVa.d.ts} +3 -3
  80. package/dist/{index-DMoxLG8P.d.ts.map → index-e7LXeRVa.d.ts.map} +1 -1
  81. package/dist/index.d.ts +130 -42
  82. package/dist/index.d.ts.map +1 -1
  83. package/dist/index.js +64 -41
  84. package/dist/index.js.map +1 -1
  85. package/dist/{insight-report-08F022xN.d.ts → insight-report-DETqPc_A.d.ts} +4 -4
  86. package/dist/{insight-report-08F022xN.d.ts.map → insight-report-DETqPc_A.d.ts.map} +1 -1
  87. package/dist/{integrity-B_EDELom.d.ts → integrity-BKTcA-HP.d.ts} +3 -3
  88. package/dist/{integrity-B_EDELom.d.ts.map → integrity-BKTcA-HP.d.ts.map} +1 -1
  89. package/dist/{kind-factory-DMeEoMQZ.js → kind-factory-gP6lDySe.js} +164 -149
  90. package/dist/kind-factory-gP6lDySe.js.map +1 -0
  91. package/dist/{llm-client-BFMRpmqb.js → llm-client-CxQtdtd6.js} +12 -5
  92. package/dist/llm-client-CxQtdtd6.js.map +1 -0
  93. package/dist/{llm-judge-aQHIk5_-.js → llm-judge-B2YxbAJb.js} +81 -7
  94. package/dist/{llm-judge-aQHIk5_-.js.map → llm-judge-B2YxbAJb.js.map} +1 -1
  95. package/dist/{matrix-Ch8JO1pG.d.ts → matrix-DGu8KhSs.d.ts} +2 -2
  96. package/dist/{matrix-Ch8JO1pG.d.ts.map → matrix-DGu8KhSs.d.ts.map} +1 -1
  97. package/dist/meta-eval/index.d.ts +100 -5
  98. package/dist/meta-eval/index.d.ts.map +1 -1
  99. package/dist/meta-eval/index.js +200 -2
  100. package/dist/meta-eval/index.js.map +1 -1
  101. package/dist/{mint-DjfDUMHr.js → mint-vWOdD8Ae.js} +2 -2
  102. package/dist/{mint-DjfDUMHr.js.map → mint-vWOdD8Ae.js.map} +1 -1
  103. package/dist/multishot/golden/index.d.ts +1 -1
  104. package/dist/multishot/index.d.ts +2 -2
  105. package/dist/openapi.json +1 -1
  106. package/dist/pipelines/index.d.ts +5 -5
  107. package/dist/pipelines/index.js +3 -3
  108. package/dist/{pre-registration-DHz6P_6f.d.ts → pre-registration-BoI4ucR3.d.ts} +2 -2
  109. package/dist/{pre-registration-DHz6P_6f.d.ts.map → pre-registration-BoI4ucR3.d.ts.map} +1 -1
  110. package/dist/{produced-state-CxmbFxFd.js → produced-state-7VYDwtkk.js} +3 -3
  111. package/dist/{produced-state-CxmbFxFd.js.map → produced-state-7VYDwtkk.js.map} +1 -1
  112. package/dist/{promotion-policy-BBBcz5_3.d.ts → promotion-policy-CvMda3kU.d.ts} +2 -2
  113. package/dist/{promotion-policy-BBBcz5_3.d.ts.map → promotion-policy-CvMda3kU.d.ts.map} +1 -1
  114. package/dist/{provenance-Dp-vvyrU.d.ts → provenance-CRY67X50.d.ts} +39 -165
  115. package/dist/provenance-CRY67X50.d.ts.map +1 -0
  116. package/dist/{query-BPGMVlbM.js → query-D1nLIKt7.js} +2 -2
  117. package/dist/{query-BPGMVlbM.js.map → query-D1nLIKt7.js.map} +1 -1
  118. package/dist/{query-Na5gEIGd.d.ts → query-D6W6MaGx.d.ts} +3 -3
  119. package/dist/{query-Na5gEIGd.d.ts.map → query-D6W6MaGx.d.ts.map} +1 -1
  120. package/dist/{registry-xEb_xfns.d.ts → registry-7pOUBrtX.d.ts} +4 -4
  121. package/dist/{registry-xEb_xfns.d.ts.map → registry-7pOUBrtX.d.ts.map} +1 -1
  122. package/dist/{release-confidence-D6lQw_o7.d.ts → release-confidence-BAcNYOf1.d.ts} +4 -4
  123. package/dist/{release-confidence-D6lQw_o7.d.ts.map → release-confidence-BAcNYOf1.d.ts.map} +1 -1
  124. package/dist/{release-confidence-CzUHc4z4.js → release-confidence-BsGEg_xg.js} +3 -3
  125. package/dist/{release-confidence-CzUHc4z4.js.map → release-confidence-BsGEg_xg.js.map} +1 -1
  126. package/dist/reporting.d.ts +3 -3
  127. package/dist/reporting.js +1 -1
  128. package/dist/{researcher-CMUTQXD7.d.ts → researcher-jsW1X94L.d.ts} +7 -8
  129. package/dist/researcher-jsW1X94L.d.ts.map +1 -0
  130. package/dist/{reward-hacking-SkxYgT0x.js → reward-hacking-CKW4teig.js} +2 -2
  131. package/dist/{reward-hacking-SkxYgT0x.js.map → reward-hacking-CKW4teig.js.map} +1 -1
  132. package/dist/{reward-hacking-CgPRUesA.d.ts → reward-hacking-ZXEi9VCq.d.ts} +2 -2
  133. package/dist/{reward-hacking-CgPRUesA.d.ts.map → reward-hacking-ZXEi9VCq.d.ts.map} +1 -1
  134. package/dist/rl.d.ts +7 -7
  135. package/dist/rl.js +4 -4
  136. package/dist/rollout/index.d.ts +1 -1
  137. package/dist/rollout/index.js +2 -2
  138. package/dist/{rollout-Crypdx8s.js → rollout-C-znbbYg.js} +2 -2
  139. package/dist/{rollout-Crypdx8s.js.map → rollout-C-znbbYg.js.map} +1 -1
  140. package/dist/{rubric-predictive-validity-DluJLCKQ.d.ts → rubric-predictive-validity-Dl1dvKCv.d.ts} +2 -2
  141. package/dist/{rubric-predictive-validity-DluJLCKQ.d.ts.map → rubric-predictive-validity-Dl1dvKCv.d.ts.map} +1 -1
  142. package/dist/{run-record-DQjRcYwA.d.ts → run-record-DTv1MdjK.d.ts} +2 -2
  143. package/dist/{run-record-DQjRcYwA.d.ts.map → run-record-DTv1MdjK.d.ts.map} +1 -1
  144. package/dist/{run-record-DLORoL7t.js → run-record-ZIsR9Fif.js} +2 -2
  145. package/dist/{run-record-DLORoL7t.js.map → run-record-ZIsR9Fif.js.map} +1 -1
  146. package/dist/{schema-DID1Cqct.d.ts → schema-CR5cpjQ3.d.ts} +57 -2
  147. package/dist/{schema-DID1Cqct.d.ts.map → schema-CR5cpjQ3.d.ts.map} +1 -1
  148. package/dist/{schema-CdIX2aHu.js → schema-CSf6qWgZ.js} +35 -1
  149. package/dist/{schema-CdIX2aHu.js.map → schema-CSf6qWgZ.js.map} +1 -1
  150. package/dist/semantic-concept-judge-Ct3QU7t5.js +780 -0
  151. package/dist/semantic-concept-judge-Ct3QU7t5.js.map +1 -0
  152. package/dist/{series-convergence-D9WgpXGi.d.ts → series-convergence-DeG33RpC.d.ts} +2 -2
  153. package/dist/{series-convergence-D9WgpXGi.d.ts.map → series-convergence-DeG33RpC.d.ts.map} +1 -1
  154. package/dist/{server-CCEnywOR.js → server-BR6onwZB.js} +2 -2
  155. package/dist/{server-CCEnywOR.js.map → server-BR6onwZB.js.map} +1 -1
  156. package/dist/{skillopt-optimization-method-LHi02MzH.js → skillopt-optimization-method-BzdphODy.js} +6 -6
  157. package/dist/{skillopt-optimization-method-LHi02MzH.js.map → skillopt-optimization-method-BzdphODy.js.map} +1 -1
  158. package/dist/{statistical-heldout-Yldkntvy.d.ts → statistical-heldout-DTyB_6-1.d.ts} +3 -3
  159. package/dist/{statistical-heldout-Yldkntvy.d.ts.map → statistical-heldout-DTyB_6-1.d.ts.map} +1 -1
  160. package/dist/{store-Cq9oOrI1.d.ts → store-BErPvYBr.d.ts} +2 -2
  161. package/dist/{store-Cq9oOrI1.d.ts.map → store-BErPvYBr.d.ts.map} +1 -1
  162. package/dist/{store-otlp-CHjBvWQY.js → store-otlp-Dow0pk_5.js} +2 -2
  163. package/dist/{store-otlp-CHjBvWQY.js.map → store-otlp-Dow0pk_5.js.map} +1 -1
  164. package/dist/{store-tool-spans-BvdUbeOB.d.ts → store-tool-spans-CCZNsihA.d.ts} +8 -8
  165. package/dist/{store-tool-spans-BvdUbeOB.d.ts.map → store-tool-spans-CCZNsihA.d.ts.map} +1 -1
  166. package/dist/{store-tool-spans-B9o6tU8f.js → store-tool-spans-CeNj_m2L.js} +3 -3
  167. package/dist/{store-tool-spans-B9o6tU8f.js.map → store-tool-spans-CeNj_m2L.js.map} +1 -1
  168. package/dist/storyboard/index.d.ts +1 -1
  169. package/dist/{summary-report-DRstQNBX.d.ts → summary-report-gMrbYawB.d.ts} +3 -3
  170. package/dist/{summary-report-DRstQNBX.d.ts.map → summary-report-gMrbYawB.d.ts.map} +1 -1
  171. package/dist/{task-failure-attributes-CBGtLS_H.js → task-failure-attributes-CZjZeBsY.js} +3 -3
  172. package/dist/{task-failure-attributes-CBGtLS_H.js.map → task-failure-attributes-CZjZeBsY.js.map} +1 -1
  173. package/dist/{tool-groups-DjwlMBvW.d.ts → tool-groups-Cp4Xdzrp.d.ts} +3 -3
  174. package/dist/tool-groups-Cp4Xdzrp.d.ts.map +1 -0
  175. package/dist/{tool-waste-CwGHzBzX.js → tool-waste-B9tdWV6g.js} +253 -6
  176. package/dist/tool-waste-B9tdWV6g.js.map +1 -0
  177. package/dist/{tool-waste-BrmLKxMw.d.ts → tool-waste-D23I0zWm.d.ts} +4 -4
  178. package/dist/{tool-waste-BrmLKxMw.d.ts.map → tool-waste-D23I0zWm.d.ts.map} +1 -1
  179. package/dist/trace-repair/index.d.ts +2 -2
  180. package/dist/traces.d.ts +10 -10
  181. package/dist/traces.js +7 -7
  182. package/dist/{trajectory-r1bQqvBQ.d.ts → trajectory-D7qrNvaN.d.ts} +3 -3
  183. package/dist/{trajectory-r1bQqvBQ.d.ts.map → trajectory-D7qrNvaN.d.ts.map} +1 -1
  184. package/dist/trajectory-replay/index.d.ts +3 -3
  185. package/dist/{types-CCZ34qmV.d.ts → types-BDV4PiMR.d.ts} +3 -3
  186. package/dist/{types-CCZ34qmV.d.ts.map → types-BDV4PiMR.d.ts.map} +1 -1
  187. package/dist/{types-nokrtr7M.d.ts → types-Ba5UQyVD.d.ts} +4 -4
  188. package/dist/{types-nokrtr7M.d.ts.map → types-Ba5UQyVD.d.ts.map} +1 -1
  189. package/dist/{types-DMoNFDWi.d.ts → types-DN2WdT5S.d.ts} +3 -3
  190. package/dist/{types-DMoNFDWi.d.ts.map → types-DN2WdT5S.d.ts.map} +1 -1
  191. package/dist/types-gvRsyJLh.d.ts +831 -0
  192. package/dist/types-gvRsyJLh.d.ts.map +1 -0
  193. package/dist/wire/index.d.ts +3 -3
  194. package/dist/wire/index.js +1 -1
  195. package/docs/plants.md +69 -1
  196. package/docs/public-api.md +6 -4
  197. package/docs/trace-analysis.md +55 -8
  198. package/package.json +1 -1
  199. package/dist/backend-integrity-e79K3UPD.d.ts.map +0 -1
  200. package/dist/bounded-process-VIi0KSL2.js.map +0 -1
  201. package/dist/chat-client-DI79OPye.js.map +0 -1
  202. package/dist/chat-json-call-5Jxna-aV.js.map +0 -1
  203. package/dist/dspy-rlm-engine-CS3qcCEk.js.map +0 -1
  204. package/dist/failure-cluster-6YSvsKlp.d.ts +0 -58
  205. package/dist/failure-cluster-6YSvsKlp.d.ts.map +0 -1
  206. package/dist/index-DDAPhUJJ.d.ts.map +0 -1
  207. package/dist/index-DnglhM0A.d.ts.map +0 -1
  208. package/dist/kind-factory-DMeEoMQZ.js.map +0 -1
  209. package/dist/llm-client-BFMRpmqb.js.map +0 -1
  210. package/dist/provenance-Dp-vvyrU.d.ts.map +0 -1
  211. package/dist/raw-provider-sink-BU29Sh8h.d.ts +0 -134
  212. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +0 -1
  213. package/dist/researcher-CMUTQXD7.d.ts.map +0 -1
  214. package/dist/semantic-concept-judge-I36eejJx.js +0 -382
  215. package/dist/semantic-concept-judge-I36eejJx.js.map +0 -1
  216. package/dist/tool-groups-DjwlMBvW.d.ts.map +0 -1
  217. package/dist/tool-waste-CwGHzBzX.js.map +0 -1
  218. package/dist/types-Bfk0uxRj.d.ts +0 -443
  219. package/dist/types-Bfk0uxRj.d.ts.map +0 -1
@@ -1,9 +1,9 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
2
  import { r as canonicalString } from "./canonical-DPyQ_rpt.js";
3
- import { J as recoverTruncatedJson, n as JudgeParseError } from "./llm-judge-aQHIk5_-.js";
3
+ import { Y as recoverTruncatedJson, n as JudgeParseError } from "./llm-judge-B2YxbAJb.js";
4
4
  import { i as CostLedger } from "./cost-ledger-B1qx30B4.js";
5
5
  import { t as certificationEvidenceDigest } from "./verdict-BQ3pCFf8.js";
6
- import { f as ModelSubstitutionError, h as assertServedModel, o as costReceiptFromLlm, s as costReceiptFromLlmError, u as maximumChargeForLlmRequest } from "./llm-client-BFMRpmqb.js";
6
+ import { f as ModelSubstitutionError, h as assertServedModel, o as costReceiptFromLlm, s as costReceiptFromLlmError, u as maximumChargeForLlmRequest } from "./llm-client-CxQtdtd6.js";
7
7
  import { createHash, randomUUID } from "node:crypto";
8
8
  import { harnessSupportsModel } from "@tangle-network/agent-interface";
9
9
  //#region src/agent-profile.ts
@@ -583,4 +583,4 @@ function extractProducedState(events) {
583
583
  //#endregion
584
584
  export { CODING_HARNESSES as a, agentProfileId as c, harnessAxisOf as d, verifyCompletion as i, agentProfileModelId as l, completionVerdict as n, HARNESS_NATIVE_MODEL as o, createLlmCorrectnessChecker as r, agentProfileHash as s, extractProducedState as t, expandProfileAxes as u };
585
585
 
586
- //# sourceMappingURL=produced-state-CxmbFxFd.js.map
586
+ //# sourceMappingURL=produced-state-7VYDwtkk.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"produced-state-CxmbFxFd.js","names":[],"sources":["../src/agent-profile.ts","../src/completion-verifier.ts","../src/produced-state.ts"],"sourcesContent":["import { createHash } from 'node:crypto'\nimport type { AgentProfile } from '@tangle-network/agent-interface'\nimport { type HarnessType, harnessSupportsModel } from '@tangle-network/agent-interface'\nimport { ValidationError } from './errors'\nimport { canonicalString } from './ledger-core/canonical'\n\nexport type { AgentProfile, HarnessType } from '@tangle-network/agent-interface'\n\n/**\n * The agentic coding harnesses an eval sweeps by default — the ones we care about\n * ranking. This is the SINGLE source of that list; consumers import it instead of\n * re-declaring their own (a re-declared list is how the fleet drifts). Pass an\n * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known\n * harness) to widen beyond these.\n */\nexport const CODING_HARNESSES: readonly HarnessType[] = [\n 'opencode',\n 'claude-code',\n 'codex',\n 'kimi-code',\n]\n\nexport interface ProfileAxisSpec {\n /** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the\n * harness and model vary. `model.default` is the fallback model. */\n base: AgentProfile\n /** Harnesses to cross. Default: {@link CODING_HARNESSES}. */\n harnesses?: readonly HarnessType[]\n /** Models to cross. Default: `[base.model.default]` — one model, i.e. today's\n * single-model behaviour, so omitting this never changes an existing run. */\n models?: readonly string[]\n /** Force every (harness, model) pair verbatim, even ones the harness can't run —\n * for deliberately testing failure modes. Default (false): SNAP instead — a\n * vendor-locked harness runs only the swept models in its family, or its native\n * default when it supports none, so no harness is dropped and none gets a\n * guaranteed-failing foreign-model cell. */\n keepIncompatible?: boolean\n}\n\n/** Model sentinel for a vendor-locked harness that supports none of the swept models:\n * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness\n * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi\n * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship\n * table that would rot as router catalogs change. */\nexport const HARNESS_NATIVE_MODEL = 'default'\n\n/**\n * Expand a base profile across the harness × model matrix into the `AgentProfile[]`\n * that `runProfileMatrix` / `selfImprove` score — the ONE place \"which harnesses ×\n * which models do we evaluate\" lives, so no product hand-rolls its own harness list\n * or column→profile mapping (the pattern that let those copies drift and silently\n * break the harness pivot).\n *\n * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,\n * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both\n * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row\n * and results join back by harness/model via {@link harnessAxisOf} with no\n * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or\n * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every\n * requested harness runs; `keepIncompatible` forces every pair verbatim.\n *\n * Omit `harnesses`/`models` to sweep the full default set — the \"turn it on for\n * everything we care about\" switch, identical in shape whether one harness or all.\n */\nexport function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[] {\n const harnesses = spec.harnesses ?? CODING_HARNESSES\n if (harnesses.length === 0) throw new ValidationError('expandProfileAxes: no harnesses to sweep')\n const baseModel = spec.base.model?.default\n const models = spec.models ?? (baseModel ? [baseModel] : [])\n if (models.length === 0) {\n throw new ValidationError(\n 'expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied',\n )\n }\n const out: AgentProfile[] = []\n const seen = new Set<string>()\n for (const harness of harnesses) {\n // A universal (router-backed) harness — opencode/pi/claudish — runs every swept\n // model. A vendor-locked harness — codex/claude-code/kimi-code — runs only the\n // swept models in its own family; when it supports NONE of them it snaps to its\n // native default (the `HARNESS_NATIVE_MODEL` sentinel it resolves at runtime)\n // rather than being dropped, so every requested harness still appears in the\n // sweep on a model it can actually run — e.g. sweeping `deepseek/x` puts opencode\n // on deepseek and kimi-code on its own Kimi model, a real head-to-head.\n // `keepIncompatible` forces every (harness, model) pair verbatim (failure-mode runs).\n const supported = spec.keepIncompatible\n ? models\n : models.filter((model) => harnessSupportsModel(harness, model))\n const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL]\n for (const model of effective) {\n const profile: AgentProfile = {\n ...spec.base,\n harness,\n name: `${spec.base.name ?? 'agent'}/${harness}/${model}`,\n model: { ...spec.base.model, default: model },\n metadata: { ...(spec.base.metadata ?? {}), harness, harnessModel: model },\n }\n const id = agentProfileId(profile)\n if (seen.has(id)) continue\n seen.add(id)\n out.push(profile)\n }\n }\n if (out.length === 0) {\n // Unreachable in normal use — snapping guarantees ≥1 cell per harness — but keep a\n // fail-closed guard so a future refactor can't silently produce an empty sweep.\n throw new ValidationError(\n `expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(', ')}], models=[${models.join(', ')}]).`,\n )\n }\n return out\n}\n\n/**\n * Read the (harness, model) a matrix cell ran under, off a profile or a result row's\n * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile\n * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by\n * this instead of recomputing an id (recomputing the wrong key is what broke the pivot\n * in the hand-rolled copies).\n */\nexport function harnessAxisOf(\n profile: Pick<AgentProfile, 'metadata'>,\n): { harness: HarnessType; model: string } | undefined {\n const m = profile.metadata as Record<string, unknown> | undefined\n const harness = m?.harness\n const model = m?.harnessModel\n if (typeof harness === 'string' && typeof model === 'string') {\n return { harness: harness as HarnessType, model }\n }\n return undefined\n}\n\n/**\n * Collision-resistant, path-safe, human-readable profile id for eval artifacts.\n * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix\n * keys, and directory names where two profiles must not collapse onto one row.\n * The suffix is the first 64 bits of the behaviour hash, enough for ordinary\n * eval matrices while keeping filenames readable.\n */\nexport function agentProfileId(profile: AgentProfile): string {\n const label = pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? 'profile'\n return `${label}-${agentProfileHash(profile).slice(0, 16)}`\n}\n\n/**\n * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete\n * model id because run records reject bare/missing model aliases.\n */\nexport function agentProfileModelId(profile: AgentProfile): string {\n const model = profile.model?.default?.trim()\n if (!model) {\n const label = agentProfileDisplayLabel(profile) ?? 'unnamed profile'\n throw new ValidationError(\n `AgentProfile \"${label}\" has no model.default — cannot record eval run`,\n )\n }\n return model\n}\n\nfunction agentProfileDisplayLabel(profile: AgentProfile): string | undefined {\n return profile.name?.trim() || profile.version?.trim() || undefined\n}\n\nfunction pathSafeProfileLabel(label: string | undefined): string | undefined {\n const safe = label\n ?.trim()\n .replace(/[^A-Za-z0-9._-]+/g, '-')\n .replace(/-+/g, '-')\n .replace(/^-|-$/g, '')\n return safe || undefined\n}\n\nfunction compact<T extends Record<string, unknown>>(input: T): Partial<T> {\n const out: Record<string, unknown> = {}\n for (const [key, value] of Object.entries(input)) {\n if (value !== undefined) out[key] = value\n }\n return out as Partial<T>\n}\n\n/**\n * Deterministic behaviour identity for the canonical\n * `@tangle-network/agent-interface` AgentProfile.\n *\n * `name` and `description` are labels and do not affect the hash. Profile\n * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,\n * and extensions do affect the hash. Resource array order is hash-bearing\n * because mount order can change agent behaviour. Undefined fields are treated\n * as absent; explicit `null` fields remain hash-bearing.\n */\nexport function agentProfileHash(profile: AgentProfile): string {\n const model = agentProfileModelId(profile)\n const behaviour = compact({\n ...profile,\n name: undefined,\n description: undefined,\n tags: profile.tags ? [...profile.tags].sort() : undefined,\n model: compact({ ...profile.model, default: model }),\n })\n return createHash('sha256').update(canonicalString(behaviour)).digest('hex')\n}\n","/**\n * Completion verifier — the task-completion oracle.\n *\n * Answers the only eval question that is not a proxy: did the agent actually\n * COMPLETE the task — produce every required deliverable, persisted and\n * correct — rather than describe what should be done. A fluent transcript\n * that never produces the artifact scores zero here.\n *\n * Per requirement, a two-stage check:\n * 1. Structural — a produced item (vault artifact / approved proposal /\n * tool call) of the right kind is matched against the requirement and\n * carries non-empty content. Deterministic; no LLM.\n * 2. Correctness — only if structurally present AND the matched item\n * carries content, one targeted check decides whether that item\n * actually fulfils the requirement. A hallucinated artifact fails here;\n * an absent one already failed stage 1.\n *\n * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —\n * checker failures — are excluded from the denominator, never scored as\n * zeros). Quality dimensions are meaningless on an incomplete task — callers\n * gate on `fullyComplete` / `completionRate` before scoring quality.\n */\n\nimport { randomUUID } from 'node:crypto'\nimport type { ChatClient, ChatRequest } from './analyst/chat-client'\nimport type { Artifact } from './artifact-validator'\nimport { CostLedger, type CostLedgerHandle } from './cost-ledger'\nimport { assertServedModel, ModelSubstitutionError } from './integrity/served-model'\nimport { recoverTruncatedJson } from './json-recovery'\nimport { JudgeParseError } from './judges'\nimport {\n costReceiptFromLlm,\n costReceiptFromLlmError,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport { packageVersion } from './package-version'\nimport type { RawProviderEvent, RawProviderSink } from './trace/raw-provider-sink'\nimport {\n certificationEvidenceDigest,\n type DefaultVerdict,\n type VerdictCertification,\n} from './verdict'\nimport type { CheckerIdentity, VerificationStrategySource } from './verification-strategy'\n\n/** What kind of produced state can satisfy a requirement structurally. */\nexport type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any'\n\nexport interface CompletionRequirement {\n /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */\n reqId: string\n /** Human-readable description of the required deliverable. */\n title: string\n /** Optional kind/category hint, matched against a produced item's kind. */\n category?: string\n /** What produced state satisfies this requirement. Defaults to 'any'. */\n satisfiedBy?: SatisfiedBy\n}\n\nexport interface TaskGold {\n taskId: string\n requirements: CompletionRequirement[]\n}\n\nexport interface ProducedProposal {\n id: string\n title: string\n status: 'pending' | 'approved' | 'rejected'\n /** Optional persisted body — when present, enables a correctness check. */\n content?: string\n}\n\n/** Everything observable about what a run actually produced. */\nexport interface ProducedState {\n /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */\n artifacts: Artifact[]\n /** Proposals / filings the agent created. */\n proposals: ProducedProposal[]\n /** Names of tools the agent invoked. */\n toolCalls: string[]\n}\n\nexport interface RequirementCheck {\n reqId: string\n title: string\n /** A produced item of the right kind matched the requirement, non-empty. */\n structurallyPresent: boolean\n /**\n * Whether the matched item actually fulfils the requirement. `null` when\n * not structurally present, when the matched item carries no content\n * to assess, or when the correctness check itself failed (`unmeasured`).\n */\n correct: boolean | null\n /** structurallyPresent && !unmeasured && correct !== false. */\n satisfied: boolean\n /**\n * Set when the correctness check itself errored (LLM call failure or an\n * unparseable response after retry). The requirement's fulfilment is\n * UNKNOWN — `correct` stays null, `satisfied` is false, and\n * `completionVerdict` excludes the row from `completionRate`'s\n * denominator. Never folded into a zero: a synthetic zero is\n * indistinguishable from a real failure (see `JudgeParseError`).\n */\n unmeasured?: true\n /** Why the correctness check could not be measured (present iff `unmeasured`). */\n unmeasuredReason?: string\n /** Human-readable evidence for the verdict. */\n evidence: string[]\n}\n\n/** Extends the substrate verdict spine: `valid` = `fullyComplete` and\n * `score` = `completionRate` — derived in `completionVerdict()`, the one\n * place those equalities hold by construction. */\nexport interface CompletionVerdict extends DefaultVerdict {\n taskId: string\n requirements: RequirementCheck[]\n /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */\n completionRate: number\n /** Every measurable requirement satisfied (false when anything is unmeasured). */\n fullyComplete: boolean\n /** Requirements whose correctness check errored — reported, never scored as zero. */\n unmeasuredCount: number\n}\n\n/**\n * Construct a `CompletionVerdict` from the per-requirement checks, deriving\n * `completionRate` / `fullyComplete` and the spine fields (`valid` =\n * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero\n * requirements — a verdict over nothing is a misconfiguration, mirroring\n * `verifyCompletion`'s gold-spec guard.\n */\nexport function completionVerdict(input: {\n taskId: string\n requirements: RequirementCheck[]\n /** What certified the correctness stage, when anything did. Omitted =\n * an uncertified verdict — the honest default for a bare checker. */\n certification?: VerdictCertification\n}): CompletionVerdict {\n if (input.requirements.length === 0) {\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`,\n )\n }\n const measurable = input.requirements.filter((r) => !r.unmeasured)\n const unmeasuredCount = input.requirements.length - measurable.length\n if (measurable.length === 0) {\n // Every check errored: this is an infrastructure failure, not a scored\n // run. A 0-rate verdict here would be a fabricated measurement.\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? 'unknown reason'})`,\n )\n }\n const satisfiedCount = measurable.filter((r) => r.satisfied).length\n const completionRate = satisfiedCount / measurable.length\n // A run with unmeasured rows can still report a rate over what WAS\n // measured, but must not claim full completion.\n const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length\n return {\n taskId: input.taskId,\n requirements: input.requirements,\n completionRate,\n fullyComplete,\n unmeasuredCount,\n valid: fullyComplete,\n score: completionRate,\n ...(input.certification === undefined ? {} : { certification: input.certification }),\n }\n}\n\n/**\n * What a correctness checker declares about itself so the completion\n * verdict can carry a certification: which strategy member it discharges,\n * its exact identity, and the steps its answers rest on unverified.\n */\nexport interface CorrectnessCheckerAttestation {\n strategy: VerificationStrategySource\n checker: CheckerIdentity\n assumptions: string[]\n}\n\n/**\n * Decides whether a produced item's content actually fulfils a requirement.\n * Injected so the structural verifier stays pure and unit-testable; the\n * production implementation is `createLlmCorrectnessChecker`.\n *\n * `attestation` is optional metadata on the function value: a checker that\n * carries one yields certified completion verdicts; a bare function yields\n * the same verdict uncertified. A plain arrow function remains a valid\n * checker.\n */\nexport interface CorrectnessChecker {\n (\n requirement: CompletionRequirement,\n content: string,\n ): Promise<{ correct: boolean; reason: string }>\n attestation?: CorrectnessCheckerAttestation\n}\n\nconst STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'of',\n 'for',\n 'and',\n 'or',\n 'to',\n 'in',\n 'on',\n 'with',\n 'by',\n])\n\n// Deliverable-FORM vocabulary — words that name the SHAPE of an output, not its\n// domain content. A correct \"swap-comparison view persisted as a ui/** artifact\"\n// is an OpenUI JSON whose body says nothing about \"artifact\" / \"persisted\" /\n// \"view\"; the discriminative tokens are the domain nouns (swap, comparison).\n// Stripped from the REQUIREMENT side of structural recall so a deliverable is\n// matched on what it IS about, not on the boilerplate describing its form. The\n// correctness checker strips the same class via TITLE_STOPWORDS. Anti-game holds:\n// the distinctive domain tokens remain, so an off-topic item still fails.\nconst REQUIREMENT_FORM_STOPWORDS = new Set([\n 'generated',\n 'generate',\n 'view',\n 'render',\n 'rendered',\n 'persisted',\n 'persist',\n 'artifact',\n 'file',\n 'document',\n 'note',\n 'proposal',\n 'deliverable',\n 'output',\n 'created',\n 'create',\n 'produce',\n 'produced',\n 'flag',\n])\n\nconst MATCH_THRESHOLD = 0.5\nconst MIN_CONTENT_CHARS = 50\n\nfunction tokens(s: string, extraStop?: Set<string>): Set<string> {\n return new Set(\n s\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)),\n )\n}\n\n/**\n * Recall of the requirement's tokens within a candidate's identifying text.\n * Recall, not Jaccard — a candidate's path/id legitimately carries extra\n * tokens the requirement does not name. The requirement side drops\n * deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.\n */\nfunction tokenRecall(requirementText: string, candidateText: string): number {\n const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS)\n if (req.size === 0) return 0\n const cand = tokens(candidateText)\n let hit = 0\n for (const t of req) if (cand.has(t)) hit++\n return hit / req.size\n}\n\ninterface Candidate {\n reqIndex: number\n /** Unique key for a produced item — each item satisfies at most one requirement. */\n itemKey: string\n score: number\n evidence: string\n /** Content to correctness-check, or null when the matched item has none. */\n content: string | null\n}\n\nfunction artifactCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n artifacts: Artifact[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n artifacts.forEach((a, i) => {\n if ((a.content ?? '').trim().length < MIN_CONTENT_CHARS) return\n // Match against the artifact CONTENT too, not just its path + kind — a\n // generated view / note whose path is generic still satisfies a requirement\n // when its body covers it (e.g. an OpenUI comparison grounded in the on-file\n // figures). Bounded slice keeps the recall text cheap; MATCH_THRESHOLD holds.\n let score = tokenRecall(\n reqText,\n `${a.path ?? ''} ${a.kind} ${(a.content ?? '').slice(0, 4000)}`,\n )\n if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) {\n score = Math.max(score, 1)\n }\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `artifact:${i}`,\n score,\n evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,\n content: a.content ?? null,\n })\n })\n return out\n}\n\nfunction proposalCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n proposals: ProducedProposal[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n for (const p of proposals) {\n // Pending or rejected work is not a completed deliverable.\n if (p.status !== 'approved') continue\n // A proposal needs an assessable BODY to be a deliverable. A bare title is\n // not completion: correctness cannot be judged on it, so a title-only match\n // would auto-pass the oracle (structurallyPresent && correct===null →\n // satisfied) with no verifiable content. Tool calls are the only\n // legitimately content-less deliverable (`toolCallCandidates`).\n const body = (p.content ?? '').trim()\n if (body.length < MIN_CONTENT_CHARS) continue\n // Match against the body as well as the (often short) title — a refusal /\n // flag / analysis proposal whose title is a label still satisfies a\n // descriptively-worded requirement when its content covers it. MATCH_THRESHOLD\n // + the requirement's distinctive tokens keep an off-topic proposal out;\n // correctness (a SEMANTIC checker, NOT this lexical pass) then judges\n // polarity/fulfilment, so a negation that merely contains the tokens fails.\n // Structural and correctness must use different evidence or the two-stage\n // check collapses to one lexical gate.\n const score = tokenRecall(reqText, `${p.title} ${body}`)\n if (score < MATCH_THRESHOLD) continue\n out.push({\n reqIndex,\n itemKey: `proposal:${p.id}`,\n score,\n evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,\n content: body,\n })\n }\n return out\n}\n\nfunction toolCallCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n toolCalls: string[],\n): Candidate[] {\n const out: Candidate[] = []\n toolCalls.forEach((name, i) => {\n const score = tokenRecall(req.title, name)\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `tool:${i}`,\n score,\n evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,\n content: null,\n })\n })\n return out\n}\n\n/**\n * Verify whether a run completed the task. `checkCorrectness` is injected —\n * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.\n *\n * Throws on a gold spec with no requirements: an eval task that requires\n * nothing is a misconfiguration, not a vacuously-complete task.\n */\nexport async function verifyCompletion(\n gold: TaskGold,\n state: ProducedState,\n checkCorrectness: CorrectnessChecker,\n): Promise<CompletionVerdict> {\n if (gold.requirements.length === 0) {\n throw new Error(\n `verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`,\n )\n }\n\n // Collect every above-threshold (requirement, produced-item) candidate, then\n // assign greedily by descending score: each requirement and each produced\n // item is used at most once. One deliverable fulfils one requirement.\n const candidates: Candidate[] = []\n gold.requirements.forEach((req, i) => {\n const by = req.satisfiedBy ?? 'any'\n if (by === 'artifact' || by === 'any') {\n candidates.push(...artifactCandidates(req, i, state.artifacts))\n }\n if (by === 'proposal' || by === 'any') {\n candidates.push(...proposalCandidates(req, i, state.proposals))\n }\n if (by === 'tool-call' || by === 'any') {\n candidates.push(...toolCallCandidates(req, i, state.toolCalls))\n }\n })\n candidates.sort((a, b) => b.score - a.score)\n\n const assigned = new Map<number, Candidate>()\n const itemTaken = new Set<string>()\n for (const c of candidates) {\n if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue\n assigned.set(c.reqIndex, c)\n itemTaken.add(c.itemKey)\n }\n\n const requirements: RequirementCheck[] = []\n for (let i = 0; i < gold.requirements.length; i++) {\n const req = gold.requirements[i]!\n const match = assigned.get(i)\n const evidence: string[] = []\n let correct: boolean | null = null\n let unmeasuredReason: string | undefined\n\n if (match) {\n evidence.push(match.evidence)\n if (match.content !== null) {\n try {\n const r = await checkCorrectness(req, match.content)\n correct = r.correct\n evidence.push(`correctness: ${r.correct ? 'pass' : 'fail'} — ${r.reason}`)\n } catch (err) {\n // The CHECKER failed, not the requirement. Recording this as a\n // zero would fabricate a model failure out of an infrastructure\n // one; the requirement is unmeasured and leaves the denominator.\n unmeasuredReason =\n err instanceof JudgeParseError\n ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}`\n : `checker call failed: ${err instanceof Error ? err.message : String(err)}`\n evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`)\n }\n } else {\n evidence.push('correctness: not assessed — matched item carries no content')\n }\n } else {\n const by = req.satisfiedBy ?? 'any'\n const kind = by === 'any' ? 'artifact/proposal/tool-call' : by\n evidence.push(`no produced ${kind} matched this requirement`)\n }\n\n const structurallyPresent = match !== undefined\n const unmeasured = unmeasuredReason !== undefined\n const satisfied = structurallyPresent && !unmeasured && correct !== false\n requirements.push({\n reqId: req.reqId,\n title: req.title,\n structurallyPresent,\n correct,\n satisfied,\n ...(unmeasured ? { unmeasured: true as const, unmeasuredReason } : {}),\n evidence,\n })\n }\n\n // The verdict blends the deterministic structural stage with the injected\n // correctness stage, so the certification names the checker's own member\n // and carries the structural stage's lexical nature as an assumption. A\n // checker without an attestation yields an uncertified verdict.\n const attestation = checkCorrectness.attestation\n const certification: VerdictCertification | undefined = attestation\n ? {\n strategy: attestation.strategy,\n checker: attestation.checker,\n assumptions: [\n 'structural matching is lexical token recall over produced items — the correctness stage only sees items it matched',\n ...attestation.assumptions,\n ],\n evidenceDigest: certificationEvidenceDigest({ taskId: gold.taskId, requirements }),\n }\n : undefined\n\n return completionVerdict({\n taskId: gold.taskId,\n requirements,\n ...(certification === undefined ? {} : { certification }),\n })\n}\n\nexport interface LlmCorrectnessCheckerOpts {\n model?: string\n /** Optional ledger for direct use. */\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /** Max chars of artifact content sent to the checker. */\n maxContentChars?: number\n /**\n * Checker LLM calls per requirement before giving up (parse failures and\n * call errors both consume attempts). The failure then surfaces as an\n * `unmeasured` requirement, never a zero.\n */\n maxAttempts?: number\n /**\n * Forensic capture of every checker request/response/error — without it a\n * checker failure is unauditable (the agent-turn raws never contain the\n * checker's own calls). Same sink contract as `LlmClient`.\n */\n rawSink?: RawProviderSink\n}\n\n/**\n * Parse the correctness checker's model response. Tolerates a response\n * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the\n * verdict boolean usually lands in the first few tokens, so a recovered\n * prefix with a boolean `correct` is a real measurement, not a guess.\n * Fails loud (JudgeParseError) when no boolean verdict is recoverable.\n */\nexport function parseCorrectnessResponse(raw: string): { correct: boolean; reason: string } {\n const readVerdict = (candidate: unknown): { correct: boolean; reason: string } | null => {\n if (candidate === null || typeof candidate !== 'object') return null\n const { correct, reason } = candidate as { correct?: unknown; reason?: unknown }\n if (typeof correct !== 'boolean') return null\n return { correct, reason: typeof reason === 'string' ? reason : '' }\n }\n\n const match = raw.match(/\\{[\\s\\S]*\\}/)\n if (match) {\n try {\n const strict = readVerdict(JSON.parse(match[0]))\n if (strict) return strict\n } catch {\n // fall through to truncation recovery\n }\n }\n // The strict path needs a closing `}`; a cap-hit response has none. Take\n // everything from the first `{` and auto-close it.\n const start = raw.indexOf('{')\n if (start !== -1) {\n const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)))\n if (recovered) return recovered\n }\n throw new JudgeParseError('correctness-checker', raw)\n}\n\n/**\n * Production `CorrectnessChecker` — one LLM call per matched artifact,\n * deterministic (temperature 0), structured JSON out. Judges fulfilment\n * only: a plan, a gesture, or a description of what should be done does not\n * fulfil a requirement — the artifact must BE the deliverable.\n */\nexport function createLlmCorrectnessChecker(\n chat: ChatClient,\n opts: LlmCorrectnessCheckerOpts = {},\n): CorrectnessChecker {\n const model = opts.model ?? 'claude-sonnet-4-6'\n const maxContentChars = opts.maxContentChars ?? 8000\n const maxAttempts = opts.maxAttempts ?? 2\n const costLedger = opts.costLedger ?? new CostLedger()\n const sink = opts.rawSink\n const record = async (event: RawProviderEvent): Promise<void> => {\n // Forensic capture is best-effort; the verdict is the system of record.\n try {\n await sink?.record(event)\n } catch {\n // Intentionally swallowed.\n }\n }\n const checker: CorrectnessChecker = async (requirement, content) => {\n const request = {\n model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}.',\n },\n {\n role: 'user' as const,\n content: `Requirement: ${requirement.title}\\n${\n requirement.category ? `Category: ${requirement.category}\\n` : ''\n }\\nProduced artifact:\\n${content.slice(0, maxContentChars)}`,\n },\n ],\n temperature: 0,\n maxTokens: 200,\n } satisfies ChatRequest\n let lastErr: unknown\n for (let attempt = 0; attempt < maxAttempts; attempt++) {\n const started = Date.now()\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'request',\n timestamp: started,\n requestBody: request,\n redactedFields: [],\n })\n try {\n const paid = await costLedger.runPaidCall({\n channel: 'verifier',\n phase: opts.costPhase ?? 'completion.correctness',\n actor: 'correctness-checker',\n model,\n maximumCharge:\n chat.maximumAttempts === undefined\n ? undefined\n : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),\n tags: {\n ...opts.costTags,\n requirementId: requirement.reqId,\n attempt: String(attempt),\n },\n signal: opts.signal,\n execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),\n receipt: costReceiptFromLlm,\n receiptFromError: costReceiptFromLlmError,\n })\n if (!paid.succeeded) throw paid.error\n const resp = paid.value\n // Hold the transport to its own word: a verdict produced by a model\n // other than the one requested is not this model's verdict. A\n // transport that echoes no id cannot be made to prove identity here —\n // callers needing that proof enable it at the client\n // (`LlmClientOptions.assertServedModel`) or gate on `assertModelsServed`.\n assertServedModel(model, resp.servedModel, {\n allowUnreported: true,\n context: `correctness checker for requirement ${requirement.reqId}`,\n })\n const raw = resp.content\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'response',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n responseBody: resp,\n redactedFields: [],\n })\n return parseCorrectnessResponse(raw)\n } catch (err) {\n lastErr = err\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n // Substitution is a verdict about identity, not a transient fault:\n // another attempt reaches the same wrong model and spends more tokens.\n if (err instanceof ModelSubstitutionError) throw err\n }\n }\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n }\n checker.attestation = {\n strategy: 'judge',\n checker: { name: 'llm-correctness-checker', version: model },\n assumptions: [\n 'served-model identity is accepted when the transport reports none (assertServedModel allowUnreported)',\n ],\n }\n return checker\n}\n\n/** Stopwords for requirement-title tokenization — drops the imperative verbs\n * ('review', 'update', …) common to deliverable titles so recall keys on the\n * substantive nouns, not the boilerplate ask. */\nconst TITLE_STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'and',\n 'or',\n 'for',\n 'to',\n 'of',\n 'in',\n 'on',\n 'with',\n 'review',\n 'update',\n 'new',\n 'proposed',\n])\n\n/**\n * Deterministic `CorrectnessChecker` — the no-LLM counterpart to\n * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its\n * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`\n * of the requirement title's significant tokens. No network.\n *\n * Polarity-blind: token recall credits a negation that contains the\n * requirement's tokens (\"I will NOT produce the comparison\" recalls every token\n * of \"produce the comparison\"). The structural match stage is ALSO lexical, so\n * pairing the two collapses to a single gameable gate. Use this only as an\n * opt-in structural pre-filter or for tasks whose requirements have no polarity\n * to invert; for produced-state grading the correctness checker MUST be semantic\n * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.\n */\nexport function createTokenRecallChecker(\n opts: { minRecall?: number; minContentLength?: number } = {},\n): CorrectnessChecker {\n const minRecall = opts.minRecall ?? 0.5\n const minLen = opts.minContentLength ?? 120\n const checker: CorrectnessChecker = async (requirement, content) => {\n const body = content.trim()\n if (body.length < minLen)\n return {\n correct: false,\n reason: `content too thin (${body.length} chars) to be the deliverable`,\n }\n const titleTokens = requirement.title\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t))\n if (titleTokens.length === 0)\n return {\n correct: true,\n reason: 'requirement title has no significant tokens — structural match accepted',\n }\n const lower = body.toLowerCase()\n const hits = titleTokens.filter((t) => lower.includes(t)).length\n const recall = hits / titleTokens.length\n return recall >= minRecall\n ? {\n correct: true,\n reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`,\n }\n : {\n correct: false,\n reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`,\n }\n }\n // 'schema': token recall checks shape, not meaning — the member whose\n // documented failure mode (\"a well-formed wrong answer passes\") is\n // exactly this checker's polarity blindness.\n checker.attestation = {\n strategy: 'schema',\n checker: { name: 'token-recall-checker', version: packageVersion() },\n assumptions: ['polarity-blind: a negation that recalls the requirement tokens passes'],\n }\n return checker\n}\n","/**\n * Produced-state extraction — normalize a run's runtime event stream into the\n * typed `ProducedState` the completion oracle consumes.\n *\n * `ProducedState` answers \"what did the agent actually produce\" — vault\n * artifacts, proposals, tool calls. The runtime emits these as a stream of\n * events; this module is the single normalization point from that stream to\n * the shape `verifyCompletion` expects.\n *\n * Input is structurally typed (`RuntimeEventLike`) so this module does not\n * depend on agent-runtime — agent-runtime's `RuntimeStreamEvent` satisfies it\n * structurally. The `content` on `ArtifactEventLike` and the whole\n * `proposal_created` variant are the runtime-side enrichments this contract\n * requires; the runtime emits them, this module consumes them.\n */\n\nimport type { Artifact } from './artifact-validator'\nimport type { ProducedProposal, ProducedState } from './completion-verifier'\n\n/** A tool the agent invoked. */\nexport interface ToolCallEventLike {\n type: 'tool_call'\n toolName: string\n}\n\n/**\n * An artifact the agent produced. `content` is the enriched field — the\n * runtime's base `artifact` event carries only metadata; the completion\n * oracle needs the body to verify the deliverable, so the runtime emits it.\n */\nexport interface ArtifactEventLike {\n type: 'artifact'\n artifactId: string\n name?: string\n mimeType?: string\n uri?: string\n content?: string\n}\n\n/** A proposal / filing the agent created. */\nexport interface ProposalEventLike {\n type: 'proposal_created'\n proposalId: string\n title: string\n status?: 'pending' | 'approved' | 'rejected'\n // body of the proposal (e.g. a submit_proposal `description`). When present,\n // the completion oracle correctness-checks it like artifact content; absent,\n // the proposal is graded presence-only.\n content?: string\n}\n\n/**\n * The subset of runtime stream events `extractProducedState` consumes.\n * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;\n * the `{ type: string }` catch-all keeps the input permissive so callers can\n * pass the whole unfiltered telemetry stream — unrecognized events are skipped.\n */\nexport type RuntimeEventLike =\n | ToolCallEventLike\n | ArtifactEventLike\n | ProposalEventLike\n | { type: string }\n\nfunction artifactKind(mimeType: string | undefined): string {\n if (!mimeType) return 'file'\n if (mimeType.includes('json')) return 'json'\n if (mimeType.startsWith('text/')) return 'text'\n return 'file'\n}\n\n/**\n * Normalize a run's runtime event stream into `ProducedState`.\n *\n * Pure and total — unrecognized event types are skipped. `toolCalls` is\n * deduplicated by name in first-seen order (completion cares about a tool's\n * presence, not its call count). An artifact with neither a name nor a uri\n * still yields an entry keyed by its `artifactId` so it is never silently\n * dropped; an artifact with no `content` yields empty content, which the\n * completion oracle's structural check then rejects on its own.\n */\nexport function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState {\n const artifacts: Artifact[] = []\n const proposals: ProducedProposal[] = []\n const toolCalls: string[] = []\n const seenTools = new Set<string>()\n\n for (const ev of events) {\n if (ev.type === 'tool_call') {\n const name = (ev as ToolCallEventLike).toolName\n if (name && !seenTools.has(name)) {\n seenTools.add(name)\n toolCalls.push(name)\n }\n } else if (ev.type === 'artifact') {\n const a = ev as ArtifactEventLike\n artifacts.push({\n kind: artifactKind(a.mimeType),\n path: a.name ?? a.uri ?? a.artifactId,\n content: a.content ?? '',\n })\n } else if (ev.type === 'proposal_created') {\n const p = ev as ProposalEventLike\n proposals.push({\n id: p.proposalId,\n title: p.title,\n status: p.status ?? 'pending',\n ...(p.content !== undefined ? { content: p.content } : {}),\n })\n }\n }\n\n return { artifacts, proposals, toolCalls }\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAa,mBAA2C;CACtD;CACA;CACA;CACA;AACF;;;;;;AAwBA,MAAa,uBAAuB;;;;;;;;;;;;;;;;;;;AAoBpC,SAAgB,kBAAkB,MAAuC;CACvE,MAAM,YAAY,KAAK,aAAa;CACpC,IAAI,UAAU,WAAW,GAAG,MAAM,IAAI,gBAAgB,0CAA0C;CAChG,MAAM,YAAY,KAAK,KAAK,OAAO;CACnC,MAAM,SAAS,KAAK,WAAW,YAAY,CAAC,SAAS,IAAI,CAAC;CAC1D,IAAI,OAAO,WAAW,GACpB,MAAM,IAAI,gBACR,kGACF;CAEF,MAAM,MAAsB,CAAC;CAC7B,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,WAAW,WAAW;EAS/B,MAAM,YAAY,KAAK,mBACnB,SACA,OAAO,QAAQ,UAAU,qBAAqB,SAAS,KAAK,CAAC;EACjE,MAAM,YAAY,UAAU,SAAS,IAAI,YAAY,CAAC,oBAAoB;EAC1E,KAAK,MAAM,SAAS,WAAW;GAC7B,MAAM,UAAwB;IAC5B,GAAG,KAAK;IACR;IACA,MAAM,GAAG,KAAK,KAAK,QAAQ,QAAQ,GAAG,QAAQ,GAAG;IACjD,OAAO;KAAE,GAAG,KAAK,KAAK;KAAO,SAAS;IAAM;IAC5C,UAAU;KAAE,GAAI,KAAK,KAAK,YAAY,CAAC;KAAI;KAAS,cAAc;IAAM;GAC1E;GACA,MAAM,KAAK,eAAe,OAAO;GACjC,IAAI,KAAK,IAAI,EAAE,GAAG;GAClB,KAAK,IAAI,EAAE;GACX,IAAI,KAAK,OAAO;EAClB;CACF;CACA,IAAI,IAAI,WAAW,GAGjB,MAAM,IAAI,gBACR,uDAAuD,UAAU,KAAK,IAAI,EAAE,aAAa,OAAO,KAAK,IAAI,EAAE,IAC7G;CAEF,OAAO;AACT;;;;;;;;AASA,SAAgB,cACd,SACqD;CACrD,MAAM,IAAI,QAAQ;CAClB,MAAM,UAAU,GAAG;CACnB,MAAM,QAAQ,GAAG;CACjB,IAAI,OAAO,YAAY,YAAY,OAAO,UAAU,UAClD,OAAO;EAAW;EAAwB;CAAM;AAGpD;;;;;;;;AASA,SAAgB,eAAe,SAA+B;CAE5D,OAAO,GADO,qBAAqB,yBAAyB,OAAO,CAAC,KAAK,UACzD,GAAG,iBAAiB,OAAO,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1D;;;;;AAMA,SAAgB,oBAAoB,SAA+B;CACjE,MAAM,QAAQ,QAAQ,OAAO,SAAS,KAAK;CAC3C,IAAI,CAAC,OAEH,MAAM,IAAI,gBACR,iBAFY,yBAAyB,OAAO,KAAK,kBAE1B,gDACzB;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,SAA2C;CAC3E,OAAO,QAAQ,MAAM,KAAK,KAAK,QAAQ,SAAS,KAAK,KAAK,KAAA;AAC5D;AAEA,SAAS,qBAAqB,OAA+C;CAM3E,OALa,OACT,KAAK,CAAC,CACP,QAAQ,qBAAqB,GAAG,CAAC,CACjC,QAAQ,OAAO,GAAG,CAAC,CACnB,QAAQ,UAAU,EAAE,KACR,KAAA;AACjB;AAEA,SAAS,QAA2C,OAAsB;CACxE,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,KAAK,GAC7C,IAAI,UAAU,KAAA,GAAW,IAAI,OAAO;CAEtC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,iBAAiB,SAA+B;CAC9D,MAAM,QAAQ,oBAAoB,OAAO;CACzC,MAAM,YAAY,QAAQ;EACxB,GAAG;EACH,MAAM,KAAA;EACN,aAAa,KAAA;EACb,MAAM,QAAQ,OAAO,CAAC,GAAG,QAAQ,IAAI,CAAC,CAAC,KAAK,IAAI,KAAA;EAChD,OAAO,QAAQ;GAAE,GAAG,QAAQ;GAAO,SAAS;EAAM,CAAC;CACrD,CAAC;CACD,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,gBAAgB,SAAS,CAAC,CAAC,CAAC,OAAO,KAAK;AAC7E;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACtEA,SAAgB,kBAAkB,OAMZ;CACpB,IAAI,MAAM,aAAa,WAAW,GAChC,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,+DAC3C;CAEF,MAAM,aAAa,MAAM,aAAa,QAAQ,MAAM,CAAC,EAAE,UAAU;CACjE,MAAM,kBAAkB,MAAM,aAAa,SAAS,WAAW;CAC/D,IAAI,WAAW,WAAW,GAGxB,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,yCAAyC,MAAM,aAAa,OAAO,8BAA8B,MAAM,aAAa,EAAE,EAAE,oBAAoB,iBAAiB,EACxM;CAEF,MAAM,iBAAiB,WAAW,QAAQ,MAAM,EAAE,SAAS,CAAC,CAAC;CAC7D,MAAM,iBAAiB,iBAAiB,WAAW;CAGnD,MAAM,gBAAgB,oBAAoB,KAAK,mBAAmB,WAAW;CAC7E,OAAO;EACL,QAAQ,MAAM;EACd,cAAc,MAAM;EACpB;EACA;EACA;EACA,OAAO;EACP,OAAO;EACP,GAAI,MAAM,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,eAAe,MAAM,cAAc;CACpF;AACF;AA+BA,MAAM,4BAAY,IAAI,IAAI;CACxB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAUD,MAAM,6CAA6B,IAAI,IAAI;CACzC;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAM,kBAAkB;AACxB,MAAM,oBAAoB;AAE1B,SAAS,OAAO,GAAW,WAAsC;CAC/D,OAAO,IAAI,IACT,EACG,YAAY,CAAC,CACb,MAAM,YAAY,CAAC,CACnB,QAAQ,MAAM,EAAE,SAAS,KAAK,CAAC,UAAU,IAAI,CAAC,KAAK,CAAC,WAAW,IAAI,CAAC,CAAC,CAC1E;AACF;;;;;;;AAQA,SAAS,YAAY,iBAAyB,eAA+B;CAC3E,MAAM,MAAM,OAAO,iBAAiB,0BAA0B;CAC9D,IAAI,IAAI,SAAS,GAAG,OAAO;CAC3B,MAAM,OAAO,OAAO,aAAa;CACjC,IAAI,MAAM;CACV,KAAK,MAAM,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC,GAAG;CACtC,OAAO,MAAM,IAAI;AACnB;AAYA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,GAAG,MAAM;EAC1B,KAAK,EAAE,WAAW,GAAA,CAAI,KAAK,CAAC,CAAC,SAAS,mBAAmB;EAKzD,IAAI,QAAQ,YACV,SACA,GAAG,EAAE,QAAQ,GAAG,GAAG,EAAE,KAAK,IAAI,EAAE,WAAW,GAAA,CAAI,MAAM,GAAG,GAAI,GAC9D;EACA,IAAI,IAAI,YAAY,EAAE,QAAQ,IAAI,SAAS,YAAY,MAAM,EAAE,KAAK,YAAY,GAC9E,QAAQ,KAAK,IAAI,OAAO,CAAC;EAE3B,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY;GACrB;GACA,UAAU,aAAa,EAAE,QAAQ,EAAE,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,CAAC;CACD,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,KAAK,MAAM,KAAK,WAAW;EAEzB,IAAI,EAAE,WAAW,YAAY;EAM7B,MAAM,QAAQ,EAAE,WAAW,GAAA,CAAI,KAAK;EACpC,IAAI,KAAK,SAAS,mBAAmB;EASrC,MAAM,QAAQ,YAAY,SAAS,GAAG,EAAE,MAAM,GAAG,MAAM;EACvD,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY,EAAE;GACvB;GACA,UAAU,sBAAsB,EAAE,MAAM,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS;EACX,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,MAAM,MAAM;EAC7B,MAAM,QAAQ,YAAY,IAAI,OAAO,IAAI;EACzC,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,QAAQ;GACjB;GACA,UAAU,cAAc,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACxE,SAAS;EACX,CAAC;CACH,CAAC;CACD,OAAO;AACT;;;;;;;;AASA,eAAsB,iBACpB,MACA,OACA,kBAC4B;CAC5B,IAAI,KAAK,aAAa,WAAW,GAC/B,MAAM,IAAI,MACR,2BAA2B,KAAK,OAAO,4CACzC;CAMF,MAAM,aAA0B,CAAC;CACjC,KAAK,aAAa,SAAS,KAAK,MAAM;EACpC,MAAM,KAAK,IAAI,eAAe;EAC9B,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,eAAe,OAAO,OAC/B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;CAElE,CAAC;CACD,WAAW,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;CAE3C,MAAM,2BAAW,IAAI,IAAuB;CAC5C,MAAM,4BAAY,IAAI,IAAY;CAClC,KAAK,MAAM,KAAK,YAAY;EAC1B,IAAI,SAAS,IAAI,EAAE,QAAQ,KAAK,UAAU,IAAI,EAAE,OAAO,GAAG;EAC1D,SAAS,IAAI,EAAE,UAAU,CAAC;EAC1B,UAAU,IAAI,EAAE,OAAO;CACzB;CAEA,MAAM,eAAmC,CAAC;CAC1C,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,aAAa,QAAQ,KAAK;EACjD,MAAM,MAAM,KAAK,aAAa;EAC9B,MAAM,QAAQ,SAAS,IAAI,CAAC;EAC5B,MAAM,WAAqB,CAAC;EAC5B,IAAI,UAA0B;EAC9B,IAAI;EAEJ,IAAI,OAAO;GACT,SAAS,KAAK,MAAM,QAAQ;GAC5B,IAAI,MAAM,YAAY,MACpB,IAAI;IACF,MAAM,IAAI,MAAM,iBAAiB,KAAK,MAAM,OAAO;IACnD,UAAU,EAAE;IACZ,SAAS,KAAK,gBAAgB,EAAE,UAAU,SAAS,OAAO,KAAK,EAAE,QAAQ;GAC3E,SAAS,KAAK;IAIZ,mBACE,eAAe,kBACX,6CAA6C,IAAI,IAAI,MAAM,GAAG,GAAG,MACjE,wBAAwB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAC7E,SAAS,KAAK,6BAA6B,kBAAkB;GAC/D;QAEA,SAAS,KAAK,6DAA6D;EAE/E,OAAO;GACL,MAAM,KAAK,IAAI,eAAe;GAC9B,MAAM,OAAO,OAAO,QAAQ,gCAAgC;GAC5D,SAAS,KAAK,eAAe,KAAK,0BAA0B;EAC9D;EAEA,MAAM,sBAAsB,UAAU,KAAA;EACtC,MAAM,aAAa,qBAAqB,KAAA;EACxC,MAAM,YAAY,uBAAuB,CAAC,cAAc,YAAY;EACpE,aAAa,KAAK;GAChB,OAAO,IAAI;GACX,OAAO,IAAI;GACX;GACA;GACA;GACA,GAAI,aAAa;IAAE,YAAY;IAAe;GAAiB,IAAI,CAAC;GACpE;EACF,CAAC;CACH;CAMA,MAAM,cAAc,iBAAiB;CACrC,MAAM,gBAAkD,cACpD;EACE,UAAU,YAAY;EACtB,SAAS,YAAY;EACrB,aAAa,CACX,sHACA,GAAG,YAAY,WACjB;EACA,gBAAgB,4BAA4B;GAAE,QAAQ,KAAK;GAAQ;EAAa,CAAC;CACnF,IACA,KAAA;CAEJ,OAAO,kBAAkB;EACvB,QAAQ,KAAK;EACb;EACA,GAAI,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,cAAc;CACzD,CAAC;AACH;;;;;;;;AAgCA,SAAgB,yBAAyB,KAAmD;CAC1F,MAAM,eAAe,cAAoE;EACvF,IAAI,cAAc,QAAQ,OAAO,cAAc,UAAU,OAAO;EAChE,MAAM,EAAE,SAAS,WAAW;EAC5B,IAAI,OAAO,YAAY,WAAW,OAAO;EACzC,OAAO;GAAE;GAAS,QAAQ,OAAO,WAAW,WAAW,SAAS;EAAG;CACrE;CAEA,MAAM,QAAQ,IAAI,MAAM,aAAa;CACrC,IAAI,OACF,IAAI;EACF,MAAM,SAAS,YAAY,KAAK,MAAM,MAAM,EAAE,CAAC;EAC/C,IAAI,QAAQ,OAAO;CACrB,QAAQ,CAER;CAIF,MAAM,QAAQ,IAAI,QAAQ,GAAG;CAC7B,IAAI,UAAU,IAAI;EAChB,MAAM,YAAY,YAAY,qBAAqB,IAAI,MAAM,KAAK,CAAC,CAAC;EACpE,IAAI,WAAW,OAAO;CACxB;CACA,MAAM,IAAI,gBAAgB,uBAAuB,GAAG;AACtD;;;;;;;AAQA,SAAgB,4BACd,MACA,OAAkC,CAAC,GACf;CACpB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,cAAc,KAAK,eAAe;CACxC,MAAM,aAAa,KAAK,cAAc,IAAI,WAAW;CACrD,MAAM,OAAO,KAAK;CAClB,MAAM,SAAS,OAAO,UAA2C;EAE/D,IAAI;GACF,MAAM,MAAM,OAAO,KAAK;EAC1B,QAAQ,CAER;CACF;CACA,MAAM,UAA8B,OAAO,aAAa,YAAY;EAClE,MAAM,UAAU;GACd;GACA,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IACE,MAAM;IACN,SAAS,gBAAgB,YAAY,MAAM,IACzC,YAAY,WAAW,aAAa,YAAY,SAAS,MAAM,GAChE,wBAAwB,QAAQ,MAAM,GAAG,eAAe;GAC3D,CACF;GACA,aAAa;GACb,WAAW;EACb;EACA,IAAI;EACJ,KAAK,IAAI,UAAU,GAAG,UAAU,aAAa,WAAW;GACtD,MAAM,UAAU,KAAK,IAAI;GACzB,MAAM,OAAO;IACX,SAAS,WAAW;IACpB,UAAU,KAAK;IACf;IACA,UAAU;IACV,SAAS;IACT,cAAc;IACd,WAAW;IACX,WAAW;IACX,aAAa;IACb,gBAAgB,CAAC;GACnB,CAAC;GACD,IAAI;IACF,MAAM,OAAO,MAAM,WAAW,YAAY;KACxC,SAAS;KACT,OAAO,KAAK,aAAa;KACzB,OAAO;KACP;KACA,eACE,KAAK,oBAAoB,KAAA,IACrB,KAAA,IACA,2BAA2B,SAAS,EAAE,iBAAiB,KAAK,gBAAgB,CAAC;KACnF,MAAM;MACJ,GAAG,KAAK;MACR,eAAe,YAAY;MAC3B,SAAS,OAAO,OAAO;KACzB;KACA,QAAQ,KAAK;KACb,UAAU,QAAQ,WAAW,KAAK,KAAK,SAAS;MAAE;MAAQ,gBAAgB;KAAO,CAAC;KAClF,SAAS;KACT,kBAAkB;IACpB,CAAC;IACD,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;IAChC,MAAM,OAAO,KAAK;IAMlB,kBAAkB,OAAO,KAAK,aAAa;KACzC,iBAAiB;KACjB,SAAS,uCAAuC,YAAY;IAC9D,CAAC;IACD,MAAM,MAAM,KAAK;IACjB,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc;KACd,gBAAgB,CAAC;IACnB,CAAC;IACD,OAAO,yBAAyB,GAAG;GACrC,SAAS,KAAK;IACZ,UAAU;IACV,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;KAC7D,gBAAgB,CAAC;IACnB,CAAC;IAGD,IAAI,eAAe,wBAAwB,MAAM;GACnD;EACF;EACA,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;CACtE;CACA,QAAQ,cAAc;EACpB,UAAU;EACV,SAAS;GAAE,MAAM;GAA2B,SAAS;EAAM;EAC3D,aAAa,CACX,uGACF;CACF;CACA,OAAO;AACT;;;ACpmBA,SAAS,aAAa,UAAsC;CAC1D,IAAI,CAAC,UAAU,OAAO;CACtB,IAAI,SAAS,SAAS,MAAM,GAAG,OAAO;CACtC,IAAI,SAAS,WAAW,OAAO,GAAG,OAAO;CACzC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,qBAAqB,QAAoD;CACvF,MAAM,YAAwB,CAAC;CAC/B,MAAM,YAAgC,CAAC;CACvC,MAAM,YAAsB,CAAC;CAC7B,MAAM,4BAAY,IAAI,IAAY;CAElC,KAAK,MAAM,MAAM,QACf,IAAI,GAAG,SAAS,aAAa;EAC3B,MAAM,OAAQ,GAAyB;EACvC,IAAI,QAAQ,CAAC,UAAU,IAAI,IAAI,GAAG;GAChC,UAAU,IAAI,IAAI;GAClB,UAAU,KAAK,IAAI;EACrB;CACF,OAAO,IAAI,GAAG,SAAS,YAAY;EACjC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,MAAM,aAAa,EAAE,QAAQ;GAC7B,MAAM,EAAE,QAAQ,EAAE,OAAO,EAAE;GAC3B,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,OAAO,IAAI,GAAG,SAAS,oBAAoB;EACzC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,IAAI,EAAE;GACN,OAAO,EAAE;GACT,QAAQ,EAAE,UAAU;GACpB,GAAI,EAAE,YAAY,KAAA,IAAY,EAAE,SAAS,EAAE,QAAQ,IAAI,CAAC;EAC1D,CAAC;CACH;CAGF,OAAO;EAAE;EAAW;EAAW;CAAU;AAC3C"}
1
+ {"version":3,"file":"produced-state-7VYDwtkk.js","names":[],"sources":["../src/agent-profile.ts","../src/completion-verifier.ts","../src/produced-state.ts"],"sourcesContent":["import { createHash } from 'node:crypto'\nimport type { AgentProfile } from '@tangle-network/agent-interface'\nimport { type HarnessType, harnessSupportsModel } from '@tangle-network/agent-interface'\nimport { ValidationError } from './errors'\nimport { canonicalString } from './ledger-core/canonical'\n\nexport type { AgentProfile, HarnessType } from '@tangle-network/agent-interface'\n\n/**\n * The agentic coding harnesses an eval sweeps by default — the ones we care about\n * ranking. This is the SINGLE source of that list; consumers import it instead of\n * re-declaring their own (a re-declared list is how the fleet drifts). Pass an\n * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known\n * harness) to widen beyond these.\n */\nexport const CODING_HARNESSES: readonly HarnessType[] = [\n 'opencode',\n 'claude-code',\n 'codex',\n 'kimi-code',\n]\n\nexport interface ProfileAxisSpec {\n /** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the\n * harness and model vary. `model.default` is the fallback model. */\n base: AgentProfile\n /** Harnesses to cross. Default: {@link CODING_HARNESSES}. */\n harnesses?: readonly HarnessType[]\n /** Models to cross. Default: `[base.model.default]` — one model, i.e. today's\n * single-model behaviour, so omitting this never changes an existing run. */\n models?: readonly string[]\n /** Force every (harness, model) pair verbatim, even ones the harness can't run —\n * for deliberately testing failure modes. Default (false): SNAP instead — a\n * vendor-locked harness runs only the swept models in its family, or its native\n * default when it supports none, so no harness is dropped and none gets a\n * guaranteed-failing foreign-model cell. */\n keepIncompatible?: boolean\n}\n\n/** Model sentinel for a vendor-locked harness that supports none of the swept models:\n * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness\n * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi\n * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship\n * table that would rot as router catalogs change. */\nexport const HARNESS_NATIVE_MODEL = 'default'\n\n/**\n * Expand a base profile across the harness × model matrix into the `AgentProfile[]`\n * that `runProfileMatrix` / `selfImprove` score — the ONE place \"which harnesses ×\n * which models do we evaluate\" lives, so no product hand-rolls its own harness list\n * or column→profile mapping (the pattern that let those copies drift and silently\n * break the harness pivot).\n *\n * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,\n * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both\n * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row\n * and results join back by harness/model via {@link harnessAxisOf} with no\n * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or\n * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every\n * requested harness runs; `keepIncompatible` forces every pair verbatim.\n *\n * Omit `harnesses`/`models` to sweep the full default set — the \"turn it on for\n * everything we care about\" switch, identical in shape whether one harness or all.\n */\nexport function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[] {\n const harnesses = spec.harnesses ?? CODING_HARNESSES\n if (harnesses.length === 0) throw new ValidationError('expandProfileAxes: no harnesses to sweep')\n const baseModel = spec.base.model?.default\n const models = spec.models ?? (baseModel ? [baseModel] : [])\n if (models.length === 0) {\n throw new ValidationError(\n 'expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied',\n )\n }\n const out: AgentProfile[] = []\n const seen = new Set<string>()\n for (const harness of harnesses) {\n // A universal (router-backed) harness — opencode/pi/claudish — runs every swept\n // model. A vendor-locked harness — codex/claude-code/kimi-code — runs only the\n // swept models in its own family; when it supports NONE of them it snaps to its\n // native default (the `HARNESS_NATIVE_MODEL` sentinel it resolves at runtime)\n // rather than being dropped, so every requested harness still appears in the\n // sweep on a model it can actually run — e.g. sweeping `deepseek/x` puts opencode\n // on deepseek and kimi-code on its own Kimi model, a real head-to-head.\n // `keepIncompatible` forces every (harness, model) pair verbatim (failure-mode runs).\n const supported = spec.keepIncompatible\n ? models\n : models.filter((model) => harnessSupportsModel(harness, model))\n const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL]\n for (const model of effective) {\n const profile: AgentProfile = {\n ...spec.base,\n harness,\n name: `${spec.base.name ?? 'agent'}/${harness}/${model}`,\n model: { ...spec.base.model, default: model },\n metadata: { ...(spec.base.metadata ?? {}), harness, harnessModel: model },\n }\n const id = agentProfileId(profile)\n if (seen.has(id)) continue\n seen.add(id)\n out.push(profile)\n }\n }\n if (out.length === 0) {\n // Unreachable in normal use — snapping guarantees ≥1 cell per harness — but keep a\n // fail-closed guard so a future refactor can't silently produce an empty sweep.\n throw new ValidationError(\n `expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(', ')}], models=[${models.join(', ')}]).`,\n )\n }\n return out\n}\n\n/**\n * Read the (harness, model) a matrix cell ran under, off a profile or a result row's\n * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile\n * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by\n * this instead of recomputing an id (recomputing the wrong key is what broke the pivot\n * in the hand-rolled copies).\n */\nexport function harnessAxisOf(\n profile: Pick<AgentProfile, 'metadata'>,\n): { harness: HarnessType; model: string } | undefined {\n const m = profile.metadata as Record<string, unknown> | undefined\n const harness = m?.harness\n const model = m?.harnessModel\n if (typeof harness === 'string' && typeof model === 'string') {\n return { harness: harness as HarnessType, model }\n }\n return undefined\n}\n\n/**\n * Collision-resistant, path-safe, human-readable profile id for eval artifacts.\n * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix\n * keys, and directory names where two profiles must not collapse onto one row.\n * The suffix is the first 64 bits of the behaviour hash, enough for ordinary\n * eval matrices while keeping filenames readable.\n */\nexport function agentProfileId(profile: AgentProfile): string {\n const label = pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? 'profile'\n return `${label}-${agentProfileHash(profile).slice(0, 16)}`\n}\n\n/**\n * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete\n * model id because run records reject bare/missing model aliases.\n */\nexport function agentProfileModelId(profile: AgentProfile): string {\n const model = profile.model?.default?.trim()\n if (!model) {\n const label = agentProfileDisplayLabel(profile) ?? 'unnamed profile'\n throw new ValidationError(\n `AgentProfile \"${label}\" has no model.default — cannot record eval run`,\n )\n }\n return model\n}\n\nfunction agentProfileDisplayLabel(profile: AgentProfile): string | undefined {\n return profile.name?.trim() || profile.version?.trim() || undefined\n}\n\nfunction pathSafeProfileLabel(label: string | undefined): string | undefined {\n const safe = label\n ?.trim()\n .replace(/[^A-Za-z0-9._-]+/g, '-')\n .replace(/-+/g, '-')\n .replace(/^-|-$/g, '')\n return safe || undefined\n}\n\nfunction compact<T extends Record<string, unknown>>(input: T): Partial<T> {\n const out: Record<string, unknown> = {}\n for (const [key, value] of Object.entries(input)) {\n if (value !== undefined) out[key] = value\n }\n return out as Partial<T>\n}\n\n/**\n * Deterministic behaviour identity for the canonical\n * `@tangle-network/agent-interface` AgentProfile.\n *\n * `name` and `description` are labels and do not affect the hash. Profile\n * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,\n * and extensions do affect the hash. Resource array order is hash-bearing\n * because mount order can change agent behaviour. Undefined fields are treated\n * as absent; explicit `null` fields remain hash-bearing.\n */\nexport function agentProfileHash(profile: AgentProfile): string {\n const model = agentProfileModelId(profile)\n const behaviour = compact({\n ...profile,\n name: undefined,\n description: undefined,\n tags: profile.tags ? [...profile.tags].sort() : undefined,\n model: compact({ ...profile.model, default: model }),\n })\n return createHash('sha256').update(canonicalString(behaviour)).digest('hex')\n}\n","/**\n * Completion verifier — the task-completion oracle.\n *\n * Answers the only eval question that is not a proxy: did the agent actually\n * COMPLETE the task — produce every required deliverable, persisted and\n * correct — rather than describe what should be done. A fluent transcript\n * that never produces the artifact scores zero here.\n *\n * Per requirement, a two-stage check:\n * 1. Structural — a produced item (vault artifact / approved proposal /\n * tool call) of the right kind is matched against the requirement and\n * carries non-empty content. Deterministic; no LLM.\n * 2. Correctness — only if structurally present AND the matched item\n * carries content, one targeted check decides whether that item\n * actually fulfils the requirement. A hallucinated artifact fails here;\n * an absent one already failed stage 1.\n *\n * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —\n * checker failures — are excluded from the denominator, never scored as\n * zeros). Quality dimensions are meaningless on an incomplete task — callers\n * gate on `fullyComplete` / `completionRate` before scoring quality.\n */\n\nimport { randomUUID } from 'node:crypto'\nimport type { ChatClient, ChatRequest } from './analyst/chat-client'\nimport type { Artifact } from './artifact-validator'\nimport { CostLedger, type CostLedgerHandle } from './cost-ledger'\nimport { assertServedModel, ModelSubstitutionError } from './integrity/served-model'\nimport { recoverTruncatedJson } from './json-recovery'\nimport { JudgeParseError } from './judges'\nimport {\n costReceiptFromLlm,\n costReceiptFromLlmError,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport { packageVersion } from './package-version'\nimport type { RawProviderEvent, RawProviderSink } from './trace/raw-provider-sink'\nimport {\n certificationEvidenceDigest,\n type DefaultVerdict,\n type VerdictCertification,\n} from './verdict'\nimport type { CheckerIdentity, VerificationStrategySource } from './verification-strategy'\n\n/** What kind of produced state can satisfy a requirement structurally. */\nexport type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any'\n\nexport interface CompletionRequirement {\n /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */\n reqId: string\n /** Human-readable description of the required deliverable. */\n title: string\n /** Optional kind/category hint, matched against a produced item's kind. */\n category?: string\n /** What produced state satisfies this requirement. Defaults to 'any'. */\n satisfiedBy?: SatisfiedBy\n}\n\nexport interface TaskGold {\n taskId: string\n requirements: CompletionRequirement[]\n}\n\nexport interface ProducedProposal {\n id: string\n title: string\n status: 'pending' | 'approved' | 'rejected'\n /** Optional persisted body — when present, enables a correctness check. */\n content?: string\n}\n\n/** Everything observable about what a run actually produced. */\nexport interface ProducedState {\n /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */\n artifacts: Artifact[]\n /** Proposals / filings the agent created. */\n proposals: ProducedProposal[]\n /** Names of tools the agent invoked. */\n toolCalls: string[]\n}\n\nexport interface RequirementCheck {\n reqId: string\n title: string\n /** A produced item of the right kind matched the requirement, non-empty. */\n structurallyPresent: boolean\n /**\n * Whether the matched item actually fulfils the requirement. `null` when\n * not structurally present, when the matched item carries no content\n * to assess, or when the correctness check itself failed (`unmeasured`).\n */\n correct: boolean | null\n /** structurallyPresent && !unmeasured && correct !== false. */\n satisfied: boolean\n /**\n * Set when the correctness check itself errored (LLM call failure or an\n * unparseable response after retry). The requirement's fulfilment is\n * UNKNOWN — `correct` stays null, `satisfied` is false, and\n * `completionVerdict` excludes the row from `completionRate`'s\n * denominator. Never folded into a zero: a synthetic zero is\n * indistinguishable from a real failure (see `JudgeParseError`).\n */\n unmeasured?: true\n /** Why the correctness check could not be measured (present iff `unmeasured`). */\n unmeasuredReason?: string\n /** Human-readable evidence for the verdict. */\n evidence: string[]\n}\n\n/** Extends the substrate verdict spine: `valid` = `fullyComplete` and\n * `score` = `completionRate` — derived in `completionVerdict()`, the one\n * place those equalities hold by construction. */\nexport interface CompletionVerdict extends DefaultVerdict {\n taskId: string\n requirements: RequirementCheck[]\n /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */\n completionRate: number\n /** Every measurable requirement satisfied (false when anything is unmeasured). */\n fullyComplete: boolean\n /** Requirements whose correctness check errored — reported, never scored as zero. */\n unmeasuredCount: number\n}\n\n/**\n * Construct a `CompletionVerdict` from the per-requirement checks, deriving\n * `completionRate` / `fullyComplete` and the spine fields (`valid` =\n * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero\n * requirements — a verdict over nothing is a misconfiguration, mirroring\n * `verifyCompletion`'s gold-spec guard.\n */\nexport function completionVerdict(input: {\n taskId: string\n requirements: RequirementCheck[]\n /** What certified the correctness stage, when anything did. Omitted =\n * an uncertified verdict — the honest default for a bare checker. */\n certification?: VerdictCertification\n}): CompletionVerdict {\n if (input.requirements.length === 0) {\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`,\n )\n }\n const measurable = input.requirements.filter((r) => !r.unmeasured)\n const unmeasuredCount = input.requirements.length - measurable.length\n if (measurable.length === 0) {\n // Every check errored: this is an infrastructure failure, not a scored\n // run. A 0-rate verdict here would be a fabricated measurement.\n throw new Error(\n `completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? 'unknown reason'})`,\n )\n }\n const satisfiedCount = measurable.filter((r) => r.satisfied).length\n const completionRate = satisfiedCount / measurable.length\n // A run with unmeasured rows can still report a rate over what WAS\n // measured, but must not claim full completion.\n const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length\n return {\n taskId: input.taskId,\n requirements: input.requirements,\n completionRate,\n fullyComplete,\n unmeasuredCount,\n valid: fullyComplete,\n score: completionRate,\n ...(input.certification === undefined ? {} : { certification: input.certification }),\n }\n}\n\n/**\n * What a correctness checker declares about itself so the completion\n * verdict can carry a certification: which strategy member it discharges,\n * its exact identity, and the steps its answers rest on unverified.\n */\nexport interface CorrectnessCheckerAttestation {\n strategy: VerificationStrategySource\n checker: CheckerIdentity\n assumptions: string[]\n}\n\n/**\n * Decides whether a produced item's content actually fulfils a requirement.\n * Injected so the structural verifier stays pure and unit-testable; the\n * production implementation is `createLlmCorrectnessChecker`.\n *\n * `attestation` is optional metadata on the function value: a checker that\n * carries one yields certified completion verdicts; a bare function yields\n * the same verdict uncertified. A plain arrow function remains a valid\n * checker.\n */\nexport interface CorrectnessChecker {\n (\n requirement: CompletionRequirement,\n content: string,\n ): Promise<{ correct: boolean; reason: string }>\n attestation?: CorrectnessCheckerAttestation\n}\n\nconst STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'of',\n 'for',\n 'and',\n 'or',\n 'to',\n 'in',\n 'on',\n 'with',\n 'by',\n])\n\n// Deliverable-FORM vocabulary — words that name the SHAPE of an output, not its\n// domain content. A correct \"swap-comparison view persisted as a ui/** artifact\"\n// is an OpenUI JSON whose body says nothing about \"artifact\" / \"persisted\" /\n// \"view\"; the discriminative tokens are the domain nouns (swap, comparison).\n// Stripped from the REQUIREMENT side of structural recall so a deliverable is\n// matched on what it IS about, not on the boilerplate describing its form. The\n// correctness checker strips the same class via TITLE_STOPWORDS. Anti-game holds:\n// the distinctive domain tokens remain, so an off-topic item still fails.\nconst REQUIREMENT_FORM_STOPWORDS = new Set([\n 'generated',\n 'generate',\n 'view',\n 'render',\n 'rendered',\n 'persisted',\n 'persist',\n 'artifact',\n 'file',\n 'document',\n 'note',\n 'proposal',\n 'deliverable',\n 'output',\n 'created',\n 'create',\n 'produce',\n 'produced',\n 'flag',\n])\n\nconst MATCH_THRESHOLD = 0.5\nconst MIN_CONTENT_CHARS = 50\n\nfunction tokens(s: string, extraStop?: Set<string>): Set<string> {\n return new Set(\n s\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)),\n )\n}\n\n/**\n * Recall of the requirement's tokens within a candidate's identifying text.\n * Recall, not Jaccard — a candidate's path/id legitimately carries extra\n * tokens the requirement does not name. The requirement side drops\n * deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.\n */\nfunction tokenRecall(requirementText: string, candidateText: string): number {\n const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS)\n if (req.size === 0) return 0\n const cand = tokens(candidateText)\n let hit = 0\n for (const t of req) if (cand.has(t)) hit++\n return hit / req.size\n}\n\ninterface Candidate {\n reqIndex: number\n /** Unique key for a produced item — each item satisfies at most one requirement. */\n itemKey: string\n score: number\n evidence: string\n /** Content to correctness-check, or null when the matched item has none. */\n content: string | null\n}\n\nfunction artifactCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n artifacts: Artifact[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n artifacts.forEach((a, i) => {\n if ((a.content ?? '').trim().length < MIN_CONTENT_CHARS) return\n // Match against the artifact CONTENT too, not just its path + kind — a\n // generated view / note whose path is generic still satisfies a requirement\n // when its body covers it (e.g. an OpenUI comparison grounded in the on-file\n // figures). Bounded slice keeps the recall text cheap; MATCH_THRESHOLD holds.\n let score = tokenRecall(\n reqText,\n `${a.path ?? ''} ${a.kind} ${(a.content ?? '').slice(0, 4000)}`,\n )\n if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) {\n score = Math.max(score, 1)\n }\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `artifact:${i}`,\n score,\n evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,\n content: a.content ?? null,\n })\n })\n return out\n}\n\nfunction proposalCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n proposals: ProducedProposal[],\n): Candidate[] {\n const reqText = `${req.title} ${req.category ?? ''}`\n const out: Candidate[] = []\n for (const p of proposals) {\n // Pending or rejected work is not a completed deliverable.\n if (p.status !== 'approved') continue\n // A proposal needs an assessable BODY to be a deliverable. A bare title is\n // not completion: correctness cannot be judged on it, so a title-only match\n // would auto-pass the oracle (structurallyPresent && correct===null →\n // satisfied) with no verifiable content. Tool calls are the only\n // legitimately content-less deliverable (`toolCallCandidates`).\n const body = (p.content ?? '').trim()\n if (body.length < MIN_CONTENT_CHARS) continue\n // Match against the body as well as the (often short) title — a refusal /\n // flag / analysis proposal whose title is a label still satisfies a\n // descriptively-worded requirement when its content covers it. MATCH_THRESHOLD\n // + the requirement's distinctive tokens keep an off-topic proposal out;\n // correctness (a SEMANTIC checker, NOT this lexical pass) then judges\n // polarity/fulfilment, so a negation that merely contains the tokens fails.\n // Structural and correctness must use different evidence or the two-stage\n // check collapses to one lexical gate.\n const score = tokenRecall(reqText, `${p.title} ${body}`)\n if (score < MATCH_THRESHOLD) continue\n out.push({\n reqIndex,\n itemKey: `proposal:${p.id}`,\n score,\n evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,\n content: body,\n })\n }\n return out\n}\n\nfunction toolCallCandidates(\n req: CompletionRequirement,\n reqIndex: number,\n toolCalls: string[],\n): Candidate[] {\n const out: Candidate[] = []\n toolCalls.forEach((name, i) => {\n const score = tokenRecall(req.title, name)\n if (score < MATCH_THRESHOLD) return\n out.push({\n reqIndex,\n itemKey: `tool:${i}`,\n score,\n evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,\n content: null,\n })\n })\n return out\n}\n\n/**\n * Verify whether a run completed the task. `checkCorrectness` is injected —\n * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.\n *\n * Throws on a gold spec with no requirements: an eval task that requires\n * nothing is a misconfiguration, not a vacuously-complete task.\n */\nexport async function verifyCompletion(\n gold: TaskGold,\n state: ProducedState,\n checkCorrectness: CorrectnessChecker,\n): Promise<CompletionVerdict> {\n if (gold.requirements.length === 0) {\n throw new Error(\n `verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`,\n )\n }\n\n // Collect every above-threshold (requirement, produced-item) candidate, then\n // assign greedily by descending score: each requirement and each produced\n // item is used at most once. One deliverable fulfils one requirement.\n const candidates: Candidate[] = []\n gold.requirements.forEach((req, i) => {\n const by = req.satisfiedBy ?? 'any'\n if (by === 'artifact' || by === 'any') {\n candidates.push(...artifactCandidates(req, i, state.artifacts))\n }\n if (by === 'proposal' || by === 'any') {\n candidates.push(...proposalCandidates(req, i, state.proposals))\n }\n if (by === 'tool-call' || by === 'any') {\n candidates.push(...toolCallCandidates(req, i, state.toolCalls))\n }\n })\n candidates.sort((a, b) => b.score - a.score)\n\n const assigned = new Map<number, Candidate>()\n const itemTaken = new Set<string>()\n for (const c of candidates) {\n if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue\n assigned.set(c.reqIndex, c)\n itemTaken.add(c.itemKey)\n }\n\n const requirements: RequirementCheck[] = []\n for (let i = 0; i < gold.requirements.length; i++) {\n const req = gold.requirements[i]!\n const match = assigned.get(i)\n const evidence: string[] = []\n let correct: boolean | null = null\n let unmeasuredReason: string | undefined\n\n if (match) {\n evidence.push(match.evidence)\n if (match.content !== null) {\n try {\n const r = await checkCorrectness(req, match.content)\n correct = r.correct\n evidence.push(`correctness: ${r.correct ? 'pass' : 'fail'} — ${r.reason}`)\n } catch (err) {\n // The CHECKER failed, not the requirement. Recording this as a\n // zero would fabricate a model failure out of an infrastructure\n // one; the requirement is unmeasured and leaves the denominator.\n unmeasuredReason =\n err instanceof JudgeParseError\n ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}`\n : `checker call failed: ${err instanceof Error ? err.message : String(err)}`\n evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`)\n }\n } else {\n evidence.push('correctness: not assessed — matched item carries no content')\n }\n } else {\n const by = req.satisfiedBy ?? 'any'\n const kind = by === 'any' ? 'artifact/proposal/tool-call' : by\n evidence.push(`no produced ${kind} matched this requirement`)\n }\n\n const structurallyPresent = match !== undefined\n const unmeasured = unmeasuredReason !== undefined\n const satisfied = structurallyPresent && !unmeasured && correct !== false\n requirements.push({\n reqId: req.reqId,\n title: req.title,\n structurallyPresent,\n correct,\n satisfied,\n ...(unmeasured ? { unmeasured: true as const, unmeasuredReason } : {}),\n evidence,\n })\n }\n\n // The verdict blends the deterministic structural stage with the injected\n // correctness stage, so the certification names the checker's own member\n // and carries the structural stage's lexical nature as an assumption. A\n // checker without an attestation yields an uncertified verdict.\n const attestation = checkCorrectness.attestation\n const certification: VerdictCertification | undefined = attestation\n ? {\n strategy: attestation.strategy,\n checker: attestation.checker,\n assumptions: [\n 'structural matching is lexical token recall over produced items — the correctness stage only sees items it matched',\n ...attestation.assumptions,\n ],\n evidenceDigest: certificationEvidenceDigest({ taskId: gold.taskId, requirements }),\n }\n : undefined\n\n return completionVerdict({\n taskId: gold.taskId,\n requirements,\n ...(certification === undefined ? {} : { certification }),\n })\n}\n\nexport interface LlmCorrectnessCheckerOpts {\n model?: string\n /** Optional ledger for direct use. */\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /** Max chars of artifact content sent to the checker. */\n maxContentChars?: number\n /**\n * Checker LLM calls per requirement before giving up (parse failures and\n * call errors both consume attempts). The failure then surfaces as an\n * `unmeasured` requirement, never a zero.\n */\n maxAttempts?: number\n /**\n * Forensic capture of every checker request/response/error — without it a\n * checker failure is unauditable (the agent-turn raws never contain the\n * checker's own calls). Same sink contract as `LlmClient`.\n */\n rawSink?: RawProviderSink\n}\n\n/**\n * Parse the correctness checker's model response. Tolerates a response\n * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the\n * verdict boolean usually lands in the first few tokens, so a recovered\n * prefix with a boolean `correct` is a real measurement, not a guess.\n * Fails loud (JudgeParseError) when no boolean verdict is recoverable.\n */\nexport function parseCorrectnessResponse(raw: string): { correct: boolean; reason: string } {\n const readVerdict = (candidate: unknown): { correct: boolean; reason: string } | null => {\n if (candidate === null || typeof candidate !== 'object') return null\n const { correct, reason } = candidate as { correct?: unknown; reason?: unknown }\n if (typeof correct !== 'boolean') return null\n return { correct, reason: typeof reason === 'string' ? reason : '' }\n }\n\n const match = raw.match(/\\{[\\s\\S]*\\}/)\n if (match) {\n try {\n const strict = readVerdict(JSON.parse(match[0]))\n if (strict) return strict\n } catch {\n // fall through to truncation recovery\n }\n }\n // The strict path needs a closing `}`; a cap-hit response has none. Take\n // everything from the first `{` and auto-close it.\n const start = raw.indexOf('{')\n if (start !== -1) {\n const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)))\n if (recovered) return recovered\n }\n throw new JudgeParseError('correctness-checker', raw)\n}\n\n/**\n * Production `CorrectnessChecker` — one LLM call per matched artifact,\n * deterministic (temperature 0), structured JSON out. Judges fulfilment\n * only: a plan, a gesture, or a description of what should be done does not\n * fulfil a requirement — the artifact must BE the deliverable.\n */\nexport function createLlmCorrectnessChecker(\n chat: ChatClient,\n opts: LlmCorrectnessCheckerOpts = {},\n): CorrectnessChecker {\n const model = opts.model ?? 'claude-sonnet-4-6'\n const maxContentChars = opts.maxContentChars ?? 8000\n const maxAttempts = opts.maxAttempts ?? 2\n const costLedger = opts.costLedger ?? new CostLedger()\n const sink = opts.rawSink\n const record = async (event: RawProviderEvent): Promise<void> => {\n // Forensic capture is best-effort; the verdict is the system of record.\n try {\n await sink?.record(event)\n } catch {\n // Intentionally swallowed.\n }\n }\n const checker: CorrectnessChecker = async (requirement, content) => {\n const request = {\n model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}.',\n },\n {\n role: 'user' as const,\n content: `Requirement: ${requirement.title}\\n${\n requirement.category ? `Category: ${requirement.category}\\n` : ''\n }\\nProduced artifact:\\n${content.slice(0, maxContentChars)}`,\n },\n ],\n temperature: 0,\n maxTokens: 200,\n } satisfies ChatRequest\n let lastErr: unknown\n for (let attempt = 0; attempt < maxAttempts; attempt++) {\n const started = Date.now()\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'request',\n timestamp: started,\n requestBody: request,\n redactedFields: [],\n })\n try {\n const paid = await costLedger.runPaidCall({\n channel: 'verifier',\n phase: opts.costPhase ?? 'completion.correctness',\n actor: 'correctness-checker',\n model,\n maximumCharge:\n chat.maximumAttempts === undefined\n ? undefined\n : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),\n tags: {\n ...opts.costTags,\n requirementId: requirement.reqId,\n attempt: String(attempt),\n },\n signal: opts.signal,\n execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),\n receipt: costReceiptFromLlm,\n receiptFromError: costReceiptFromLlmError,\n })\n if (!paid.succeeded) throw paid.error\n const resp = paid.value\n // Hold the transport to its own word: a verdict produced by a model\n // other than the one requested is not this model's verdict. A\n // transport that echoes no id cannot be made to prove identity here —\n // callers needing that proof enable it at the client\n // (`LlmClientOptions.assertServedModel`) or gate on `assertModelsServed`.\n assertServedModel(model, resp.servedModel, {\n allowUnreported: true,\n context: `correctness checker for requirement ${requirement.reqId}`,\n })\n const raw = resp.content\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'response',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n responseBody: resp,\n redactedFields: [],\n })\n return parseCorrectnessResponse(raw)\n } catch (err) {\n lastErr = err\n await record({\n eventId: randomUUID(),\n provider: chat.transport,\n model,\n endpoint: '/chat',\n baseUrl: '',\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n // Substitution is a verdict about identity, not a transient fault:\n // another attempt reaches the same wrong model and spends more tokens.\n if (err instanceof ModelSubstitutionError) throw err\n }\n }\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n }\n checker.attestation = {\n strategy: 'judge',\n checker: { name: 'llm-correctness-checker', version: model },\n assumptions: [\n 'served-model identity is accepted when the transport reports none (assertServedModel allowUnreported)',\n ],\n }\n return checker\n}\n\n/** Stopwords for requirement-title tokenization — drops the imperative verbs\n * ('review', 'update', …) common to deliverable titles so recall keys on the\n * substantive nouns, not the boilerplate ask. */\nconst TITLE_STOPWORDS = new Set([\n 'the',\n 'a',\n 'an',\n 'and',\n 'or',\n 'for',\n 'to',\n 'of',\n 'in',\n 'on',\n 'with',\n 'review',\n 'update',\n 'new',\n 'proposed',\n])\n\n/**\n * Deterministic `CorrectnessChecker` — the no-LLM counterpart to\n * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its\n * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`\n * of the requirement title's significant tokens. No network.\n *\n * Polarity-blind: token recall credits a negation that contains the\n * requirement's tokens (\"I will NOT produce the comparison\" recalls every token\n * of \"produce the comparison\"). The structural match stage is ALSO lexical, so\n * pairing the two collapses to a single gameable gate. Use this only as an\n * opt-in structural pre-filter or for tasks whose requirements have no polarity\n * to invert; for produced-state grading the correctness checker MUST be semantic\n * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.\n */\nexport function createTokenRecallChecker(\n opts: { minRecall?: number; minContentLength?: number } = {},\n): CorrectnessChecker {\n const minRecall = opts.minRecall ?? 0.5\n const minLen = opts.minContentLength ?? 120\n const checker: CorrectnessChecker = async (requirement, content) => {\n const body = content.trim()\n if (body.length < minLen)\n return {\n correct: false,\n reason: `content too thin (${body.length} chars) to be the deliverable`,\n }\n const titleTokens = requirement.title\n .toLowerCase()\n .split(/[^a-z0-9]+/)\n .filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t))\n if (titleTokens.length === 0)\n return {\n correct: true,\n reason: 'requirement title has no significant tokens — structural match accepted',\n }\n const lower = body.toLowerCase()\n const hits = titleTokens.filter((t) => lower.includes(t)).length\n const recall = hits / titleTokens.length\n return recall >= minRecall\n ? {\n correct: true,\n reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`,\n }\n : {\n correct: false,\n reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`,\n }\n }\n // 'schema': token recall checks shape, not meaning — the member whose\n // documented failure mode (\"a well-formed wrong answer passes\") is\n // exactly this checker's polarity blindness.\n checker.attestation = {\n strategy: 'schema',\n checker: { name: 'token-recall-checker', version: packageVersion() },\n assumptions: ['polarity-blind: a negation that recalls the requirement tokens passes'],\n }\n return checker\n}\n","/**\n * Produced-state extraction — normalize a run's runtime event stream into the\n * typed `ProducedState` the completion oracle consumes.\n *\n * `ProducedState` answers \"what did the agent actually produce\" — vault\n * artifacts, proposals, tool calls. The runtime emits these as a stream of\n * events; this module is the single normalization point from that stream to\n * the shape `verifyCompletion` expects.\n *\n * Input is structurally typed (`RuntimeEventLike`) so this module does not\n * depend on agent-runtime — agent-runtime's `RuntimeStreamEvent` satisfies it\n * structurally. The `content` on `ArtifactEventLike` and the whole\n * `proposal_created` variant are the runtime-side enrichments this contract\n * requires; the runtime emits them, this module consumes them.\n */\n\nimport type { Artifact } from './artifact-validator'\nimport type { ProducedProposal, ProducedState } from './completion-verifier'\n\n/** A tool the agent invoked. */\nexport interface ToolCallEventLike {\n type: 'tool_call'\n toolName: string\n}\n\n/**\n * An artifact the agent produced. `content` is the enriched field — the\n * runtime's base `artifact` event carries only metadata; the completion\n * oracle needs the body to verify the deliverable, so the runtime emits it.\n */\nexport interface ArtifactEventLike {\n type: 'artifact'\n artifactId: string\n name?: string\n mimeType?: string\n uri?: string\n content?: string\n}\n\n/** A proposal / filing the agent created. */\nexport interface ProposalEventLike {\n type: 'proposal_created'\n proposalId: string\n title: string\n status?: 'pending' | 'approved' | 'rejected'\n // body of the proposal (e.g. a submit_proposal `description`). When present,\n // the completion oracle correctness-checks it like artifact content; absent,\n // the proposal is graded presence-only.\n content?: string\n}\n\n/**\n * The subset of runtime stream events `extractProducedState` consumes.\n * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;\n * the `{ type: string }` catch-all keeps the input permissive so callers can\n * pass the whole unfiltered telemetry stream — unrecognized events are skipped.\n */\nexport type RuntimeEventLike =\n | ToolCallEventLike\n | ArtifactEventLike\n | ProposalEventLike\n | { type: string }\n\nfunction artifactKind(mimeType: string | undefined): string {\n if (!mimeType) return 'file'\n if (mimeType.includes('json')) return 'json'\n if (mimeType.startsWith('text/')) return 'text'\n return 'file'\n}\n\n/**\n * Normalize a run's runtime event stream into `ProducedState`.\n *\n * Pure and total — unrecognized event types are skipped. `toolCalls` is\n * deduplicated by name in first-seen order (completion cares about a tool's\n * presence, not its call count). An artifact with neither a name nor a uri\n * still yields an entry keyed by its `artifactId` so it is never silently\n * dropped; an artifact with no `content` yields empty content, which the\n * completion oracle's structural check then rejects on its own.\n */\nexport function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState {\n const artifacts: Artifact[] = []\n const proposals: ProducedProposal[] = []\n const toolCalls: string[] = []\n const seenTools = new Set<string>()\n\n for (const ev of events) {\n if (ev.type === 'tool_call') {\n const name = (ev as ToolCallEventLike).toolName\n if (name && !seenTools.has(name)) {\n seenTools.add(name)\n toolCalls.push(name)\n }\n } else if (ev.type === 'artifact') {\n const a = ev as ArtifactEventLike\n artifacts.push({\n kind: artifactKind(a.mimeType),\n path: a.name ?? a.uri ?? a.artifactId,\n content: a.content ?? '',\n })\n } else if (ev.type === 'proposal_created') {\n const p = ev as ProposalEventLike\n proposals.push({\n id: p.proposalId,\n title: p.title,\n status: p.status ?? 'pending',\n ...(p.content !== undefined ? { content: p.content } : {}),\n })\n }\n }\n\n return { artifacts, proposals, toolCalls }\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAa,mBAA2C;CACtD;CACA;CACA;CACA;AACF;;;;;;AAwBA,MAAa,uBAAuB;;;;;;;;;;;;;;;;;;;AAoBpC,SAAgB,kBAAkB,MAAuC;CACvE,MAAM,YAAY,KAAK,aAAa;CACpC,IAAI,UAAU,WAAW,GAAG,MAAM,IAAI,gBAAgB,0CAA0C;CAChG,MAAM,YAAY,KAAK,KAAK,OAAO;CACnC,MAAM,SAAS,KAAK,WAAW,YAAY,CAAC,SAAS,IAAI,CAAC;CAC1D,IAAI,OAAO,WAAW,GACpB,MAAM,IAAI,gBACR,kGACF;CAEF,MAAM,MAAsB,CAAC;CAC7B,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,WAAW,WAAW;EAS/B,MAAM,YAAY,KAAK,mBACnB,SACA,OAAO,QAAQ,UAAU,qBAAqB,SAAS,KAAK,CAAC;EACjE,MAAM,YAAY,UAAU,SAAS,IAAI,YAAY,CAAC,oBAAoB;EAC1E,KAAK,MAAM,SAAS,WAAW;GAC7B,MAAM,UAAwB;IAC5B,GAAG,KAAK;IACR;IACA,MAAM,GAAG,KAAK,KAAK,QAAQ,QAAQ,GAAG,QAAQ,GAAG;IACjD,OAAO;KAAE,GAAG,KAAK,KAAK;KAAO,SAAS;IAAM;IAC5C,UAAU;KAAE,GAAI,KAAK,KAAK,YAAY,CAAC;KAAI;KAAS,cAAc;IAAM;GAC1E;GACA,MAAM,KAAK,eAAe,OAAO;GACjC,IAAI,KAAK,IAAI,EAAE,GAAG;GAClB,KAAK,IAAI,EAAE;GACX,IAAI,KAAK,OAAO;EAClB;CACF;CACA,IAAI,IAAI,WAAW,GAGjB,MAAM,IAAI,gBACR,uDAAuD,UAAU,KAAK,IAAI,EAAE,aAAa,OAAO,KAAK,IAAI,EAAE,IAC7G;CAEF,OAAO;AACT;;;;;;;;AASA,SAAgB,cACd,SACqD;CACrD,MAAM,IAAI,QAAQ;CAClB,MAAM,UAAU,GAAG;CACnB,MAAM,QAAQ,GAAG;CACjB,IAAI,OAAO,YAAY,YAAY,OAAO,UAAU,UAClD,OAAO;EAAW;EAAwB;CAAM;AAGpD;;;;;;;;AASA,SAAgB,eAAe,SAA+B;CAE5D,OAAO,GADO,qBAAqB,yBAAyB,OAAO,CAAC,KAAK,UACzD,GAAG,iBAAiB,OAAO,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1D;;;;;AAMA,SAAgB,oBAAoB,SAA+B;CACjE,MAAM,QAAQ,QAAQ,OAAO,SAAS,KAAK;CAC3C,IAAI,CAAC,OAEH,MAAM,IAAI,gBACR,iBAFY,yBAAyB,OAAO,KAAK,kBAE1B,gDACzB;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,SAA2C;CAC3E,OAAO,QAAQ,MAAM,KAAK,KAAK,QAAQ,SAAS,KAAK,KAAK,KAAA;AAC5D;AAEA,SAAS,qBAAqB,OAA+C;CAM3E,OALa,OACT,KAAK,CAAC,CACP,QAAQ,qBAAqB,GAAG,CAAC,CACjC,QAAQ,OAAO,GAAG,CAAC,CACnB,QAAQ,UAAU,EAAE,KACR,KAAA;AACjB;AAEA,SAAS,QAA2C,OAAsB;CACxE,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,KAAK,GAC7C,IAAI,UAAU,KAAA,GAAW,IAAI,OAAO;CAEtC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,iBAAiB,SAA+B;CAC9D,MAAM,QAAQ,oBAAoB,OAAO;CACzC,MAAM,YAAY,QAAQ;EACxB,GAAG;EACH,MAAM,KAAA;EACN,aAAa,KAAA;EACb,MAAM,QAAQ,OAAO,CAAC,GAAG,QAAQ,IAAI,CAAC,CAAC,KAAK,IAAI,KAAA;EAChD,OAAO,QAAQ;GAAE,GAAG,QAAQ;GAAO,SAAS;EAAM,CAAC;CACrD,CAAC;CACD,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,gBAAgB,SAAS,CAAC,CAAC,CAAC,OAAO,KAAK;AAC7E;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACtEA,SAAgB,kBAAkB,OAMZ;CACpB,IAAI,MAAM,aAAa,WAAW,GAChC,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,+DAC3C;CAEF,MAAM,aAAa,MAAM,aAAa,QAAQ,MAAM,CAAC,EAAE,UAAU;CACjE,MAAM,kBAAkB,MAAM,aAAa,SAAS,WAAW;CAC/D,IAAI,WAAW,WAAW,GAGxB,MAAM,IAAI,MACR,4BAA4B,MAAM,OAAO,yCAAyC,MAAM,aAAa,OAAO,8BAA8B,MAAM,aAAa,EAAE,EAAE,oBAAoB,iBAAiB,EACxM;CAEF,MAAM,iBAAiB,WAAW,QAAQ,MAAM,EAAE,SAAS,CAAC,CAAC;CAC7D,MAAM,iBAAiB,iBAAiB,WAAW;CAGnD,MAAM,gBAAgB,oBAAoB,KAAK,mBAAmB,WAAW;CAC7E,OAAO;EACL,QAAQ,MAAM;EACd,cAAc,MAAM;EACpB;EACA;EACA;EACA,OAAO;EACP,OAAO;EACP,GAAI,MAAM,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,eAAe,MAAM,cAAc;CACpF;AACF;AA+BA,MAAM,4BAAY,IAAI,IAAI;CACxB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAUD,MAAM,6CAA6B,IAAI,IAAI;CACzC;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAM,kBAAkB;AACxB,MAAM,oBAAoB;AAE1B,SAAS,OAAO,GAAW,WAAsC;CAC/D,OAAO,IAAI,IACT,EACG,YAAY,CAAC,CACb,MAAM,YAAY,CAAC,CACnB,QAAQ,MAAM,EAAE,SAAS,KAAK,CAAC,UAAU,IAAI,CAAC,KAAK,CAAC,WAAW,IAAI,CAAC,CAAC,CAC1E;AACF;;;;;;;AAQA,SAAS,YAAY,iBAAyB,eAA+B;CAC3E,MAAM,MAAM,OAAO,iBAAiB,0BAA0B;CAC9D,IAAI,IAAI,SAAS,GAAG,OAAO;CAC3B,MAAM,OAAO,OAAO,aAAa;CACjC,IAAI,MAAM;CACV,KAAK,MAAM,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC,GAAG;CACtC,OAAO,MAAM,IAAI;AACnB;AAYA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,GAAG,MAAM;EAC1B,KAAK,EAAE,WAAW,GAAA,CAAI,KAAK,CAAC,CAAC,SAAS,mBAAmB;EAKzD,IAAI,QAAQ,YACV,SACA,GAAG,EAAE,QAAQ,GAAG,GAAG,EAAE,KAAK,IAAI,EAAE,WAAW,GAAA,CAAI,MAAM,GAAG,GAAI,GAC9D;EACA,IAAI,IAAI,YAAY,EAAE,QAAQ,IAAI,SAAS,YAAY,MAAM,EAAE,KAAK,YAAY,GAC9E,QAAQ,KAAK,IAAI,OAAO,CAAC;EAE3B,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY;GACrB;GACA,UAAU,aAAa,EAAE,QAAQ,EAAE,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,CAAC;CACD,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,UAAU,GAAG,IAAI,MAAM,GAAG,IAAI,YAAY;CAChD,MAAM,MAAmB,CAAC;CAC1B,KAAK,MAAM,KAAK,WAAW;EAEzB,IAAI,EAAE,WAAW,YAAY;EAM7B,MAAM,QAAQ,EAAE,WAAW,GAAA,CAAI,KAAK;EACpC,IAAI,KAAK,SAAS,mBAAmB;EASrC,MAAM,QAAQ,YAAY,SAAS,GAAG,EAAE,MAAM,GAAG,MAAM;EACvD,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,YAAY,EAAE;GACvB;GACA,UAAU,sBAAsB,EAAE,MAAM,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACnF,SAAS;EACX,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAS,mBACP,KACA,UACA,WACa;CACb,MAAM,MAAmB,CAAC;CAC1B,UAAU,SAAS,MAAM,MAAM;EAC7B,MAAM,QAAQ,YAAY,IAAI,OAAO,IAAI;EACzC,IAAI,QAAQ,iBAAiB;EAC7B,IAAI,KAAK;GACP;GACA,SAAS,QAAQ;GACjB;GACA,UAAU,cAAc,KAAK,0BAA0B,MAAM,QAAQ,CAAC,EAAE;GACxE,SAAS;EACX,CAAC;CACH,CAAC;CACD,OAAO;AACT;;;;;;;;AASA,eAAsB,iBACpB,MACA,OACA,kBAC4B;CAC5B,IAAI,KAAK,aAAa,WAAW,GAC/B,MAAM,IAAI,MACR,2BAA2B,KAAK,OAAO,4CACzC;CAMF,MAAM,aAA0B,CAAC;CACjC,KAAK,aAAa,SAAS,KAAK,MAAM;EACpC,MAAM,KAAK,IAAI,eAAe;EAC9B,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,cAAc,OAAO,OAC9B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;EAEhE,IAAI,OAAO,eAAe,OAAO,OAC/B,WAAW,KAAK,GAAG,mBAAmB,KAAK,GAAG,MAAM,SAAS,CAAC;CAElE,CAAC;CACD,WAAW,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;CAE3C,MAAM,2BAAW,IAAI,IAAuB;CAC5C,MAAM,4BAAY,IAAI,IAAY;CAClC,KAAK,MAAM,KAAK,YAAY;EAC1B,IAAI,SAAS,IAAI,EAAE,QAAQ,KAAK,UAAU,IAAI,EAAE,OAAO,GAAG;EAC1D,SAAS,IAAI,EAAE,UAAU,CAAC;EAC1B,UAAU,IAAI,EAAE,OAAO;CACzB;CAEA,MAAM,eAAmC,CAAC;CAC1C,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,aAAa,QAAQ,KAAK;EACjD,MAAM,MAAM,KAAK,aAAa;EAC9B,MAAM,QAAQ,SAAS,IAAI,CAAC;EAC5B,MAAM,WAAqB,CAAC;EAC5B,IAAI,UAA0B;EAC9B,IAAI;EAEJ,IAAI,OAAO;GACT,SAAS,KAAK,MAAM,QAAQ;GAC5B,IAAI,MAAM,YAAY,MACpB,IAAI;IACF,MAAM,IAAI,MAAM,iBAAiB,KAAK,MAAM,OAAO;IACnD,UAAU,EAAE;IACZ,SAAS,KAAK,gBAAgB,EAAE,UAAU,SAAS,OAAO,KAAK,EAAE,QAAQ;GAC3E,SAAS,KAAK;IAIZ,mBACE,eAAe,kBACX,6CAA6C,IAAI,IAAI,MAAM,GAAG,GAAG,MACjE,wBAAwB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAC7E,SAAS,KAAK,6BAA6B,kBAAkB;GAC/D;QAEA,SAAS,KAAK,6DAA6D;EAE/E,OAAO;GACL,MAAM,KAAK,IAAI,eAAe;GAC9B,MAAM,OAAO,OAAO,QAAQ,gCAAgC;GAC5D,SAAS,KAAK,eAAe,KAAK,0BAA0B;EAC9D;EAEA,MAAM,sBAAsB,UAAU,KAAA;EACtC,MAAM,aAAa,qBAAqB,KAAA;EACxC,MAAM,YAAY,uBAAuB,CAAC,cAAc,YAAY;EACpE,aAAa,KAAK;GAChB,OAAO,IAAI;GACX,OAAO,IAAI;GACX;GACA;GACA;GACA,GAAI,aAAa;IAAE,YAAY;IAAe;GAAiB,IAAI,CAAC;GACpE;EACF,CAAC;CACH;CAMA,MAAM,cAAc,iBAAiB;CACrC,MAAM,gBAAkD,cACpD;EACE,UAAU,YAAY;EACtB,SAAS,YAAY;EACrB,aAAa,CACX,sHACA,GAAG,YAAY,WACjB;EACA,gBAAgB,4BAA4B;GAAE,QAAQ,KAAK;GAAQ;EAAa,CAAC;CACnF,IACA,KAAA;CAEJ,OAAO,kBAAkB;EACvB,QAAQ,KAAK;EACb;EACA,GAAI,kBAAkB,KAAA,IAAY,CAAC,IAAI,EAAE,cAAc;CACzD,CAAC;AACH;;;;;;;;AAgCA,SAAgB,yBAAyB,KAAmD;CAC1F,MAAM,eAAe,cAAoE;EACvF,IAAI,cAAc,QAAQ,OAAO,cAAc,UAAU,OAAO;EAChE,MAAM,EAAE,SAAS,WAAW;EAC5B,IAAI,OAAO,YAAY,WAAW,OAAO;EACzC,OAAO;GAAE;GAAS,QAAQ,OAAO,WAAW,WAAW,SAAS;EAAG;CACrE;CAEA,MAAM,QAAQ,IAAI,MAAM,aAAa;CACrC,IAAI,OACF,IAAI;EACF,MAAM,SAAS,YAAY,KAAK,MAAM,MAAM,EAAE,CAAC;EAC/C,IAAI,QAAQ,OAAO;CACrB,QAAQ,CAER;CAIF,MAAM,QAAQ,IAAI,QAAQ,GAAG;CAC7B,IAAI,UAAU,IAAI;EAChB,MAAM,YAAY,YAAY,qBAAqB,IAAI,MAAM,KAAK,CAAC,CAAC;EACpE,IAAI,WAAW,OAAO;CACxB;CACA,MAAM,IAAI,gBAAgB,uBAAuB,GAAG;AACtD;;;;;;;AAQA,SAAgB,4BACd,MACA,OAAkC,CAAC,GACf;CACpB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,kBAAkB,KAAK,mBAAmB;CAChD,MAAM,cAAc,KAAK,eAAe;CACxC,MAAM,aAAa,KAAK,cAAc,IAAI,WAAW;CACrD,MAAM,OAAO,KAAK;CAClB,MAAM,SAAS,OAAO,UAA2C;EAE/D,IAAI;GACF,MAAM,MAAM,OAAO,KAAK;EAC1B,QAAQ,CAER;CACF;CACA,MAAM,UAA8B,OAAO,aAAa,YAAY;EAClE,MAAM,UAAU;GACd;GACA,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IACE,MAAM;IACN,SAAS,gBAAgB,YAAY,MAAM,IACzC,YAAY,WAAW,aAAa,YAAY,SAAS,MAAM,GAChE,wBAAwB,QAAQ,MAAM,GAAG,eAAe;GAC3D,CACF;GACA,aAAa;GACb,WAAW;EACb;EACA,IAAI;EACJ,KAAK,IAAI,UAAU,GAAG,UAAU,aAAa,WAAW;GACtD,MAAM,UAAU,KAAK,IAAI;GACzB,MAAM,OAAO;IACX,SAAS,WAAW;IACpB,UAAU,KAAK;IACf;IACA,UAAU;IACV,SAAS;IACT,cAAc;IACd,WAAW;IACX,WAAW;IACX,aAAa;IACb,gBAAgB,CAAC;GACnB,CAAC;GACD,IAAI;IACF,MAAM,OAAO,MAAM,WAAW,YAAY;KACxC,SAAS;KACT,OAAO,KAAK,aAAa;KACzB,OAAO;KACP;KACA,eACE,KAAK,oBAAoB,KAAA,IACrB,KAAA,IACA,2BAA2B,SAAS,EAAE,iBAAiB,KAAK,gBAAgB,CAAC;KACnF,MAAM;MACJ,GAAG,KAAK;MACR,eAAe,YAAY;MAC3B,SAAS,OAAO,OAAO;KACzB;KACA,QAAQ,KAAK;KACb,UAAU,QAAQ,WAAW,KAAK,KAAK,SAAS;MAAE;MAAQ,gBAAgB;KAAO,CAAC;KAClF,SAAS;KACT,kBAAkB;IACpB,CAAC;IACD,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;IAChC,MAAM,OAAO,KAAK;IAMlB,kBAAkB,OAAO,KAAK,aAAa;KACzC,iBAAiB;KACjB,SAAS,uCAAuC,YAAY;IAC9D,CAAC;IACD,MAAM,MAAM,KAAK;IACjB,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc;KACd,gBAAgB,CAAC;IACnB,CAAC;IACD,OAAO,yBAAyB,GAAG;GACrC,SAAS,KAAK;IACZ,UAAU;IACV,MAAM,OAAO;KACX,SAAS,WAAW;KACpB,UAAU,KAAK;KACf;KACA,UAAU;KACV,SAAS;KACT,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;KAC7D,gBAAgB,CAAC;IACnB,CAAC;IAGD,IAAI,eAAe,wBAAwB,MAAM;GACnD;EACF;EACA,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;CACtE;CACA,QAAQ,cAAc;EACpB,UAAU;EACV,SAAS;GAAE,MAAM;GAA2B,SAAS;EAAM;EAC3D,aAAa,CACX,uGACF;CACF;CACA,OAAO;AACT;;;ACpmBA,SAAS,aAAa,UAAsC;CAC1D,IAAI,CAAC,UAAU,OAAO;CACtB,IAAI,SAAS,SAAS,MAAM,GAAG,OAAO;CACtC,IAAI,SAAS,WAAW,OAAO,GAAG,OAAO;CACzC,OAAO;AACT;;;;;;;;;;;AAYA,SAAgB,qBAAqB,QAAoD;CACvF,MAAM,YAAwB,CAAC;CAC/B,MAAM,YAAgC,CAAC;CACvC,MAAM,YAAsB,CAAC;CAC7B,MAAM,4BAAY,IAAI,IAAY;CAElC,KAAK,MAAM,MAAM,QACf,IAAI,GAAG,SAAS,aAAa;EAC3B,MAAM,OAAQ,GAAyB;EACvC,IAAI,QAAQ,CAAC,UAAU,IAAI,IAAI,GAAG;GAChC,UAAU,IAAI,IAAI;GAClB,UAAU,KAAK,IAAI;EACrB;CACF,OAAO,IAAI,GAAG,SAAS,YAAY;EACjC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,MAAM,aAAa,EAAE,QAAQ;GAC7B,MAAM,EAAE,QAAQ,EAAE,OAAO,EAAE;GAC3B,SAAS,EAAE,WAAW;EACxB,CAAC;CACH,OAAO,IAAI,GAAG,SAAS,oBAAoB;EACzC,MAAM,IAAI;EACV,UAAU,KAAK;GACb,IAAI,EAAE;GACN,OAAO,EAAE;GACT,QAAQ,EAAE,UAAU;GACpB,GAAI,EAAE,YAAY,KAAA,IAAY,EAAE,SAAS,EAAE,QAAQ,IAAI,CAAC;EAC1D,CAAC;CACH;CAGF,OAAO;EAAE;EAAW;EAAW;CAAU;AAC3C"}
@@ -1,5 +1,5 @@
1
1
  import { d as PairedBootstrapResult, i as PairedMcNemarEvidence, r as PairedDecisionStatistic, t as PairedDecisionMethod } from "./paired-promotion-decision-CGzg0cI_.js";
2
- import { R as Scenario, h as GateContext, p as Gate, v as GateResult } from "./types-nokrtr7M.js";
2
+ import { R as Scenario, h as GateContext, p as Gate, v as GateResult } from "./types-Ba5UQyVD.js";
3
3
  import { i as Direction } from "./power-preflight-Ptse_Kq7.js";
4
4
  //#region src/campaign/gates/promotion-policy.d.ts
5
5
  /** Where an objective's per-cell scalar comes from. `composite` reads the
@@ -131,4 +131,4 @@ interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
131
131
  declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
132
132
  //#endregion
133
133
  export { ObjectiveSource as a, PromotionPolicy as c, paretoSignificanceGate as d, EvidenceVector as i, buildEvidenceVector as l, AxisVerdict as n, ParetoSignificanceGateOptions as o, BuildEvidenceVectorOptions as r, PromotionObjective as s, AxisEvidence as t, paretoPolicy as u };
134
- //# sourceMappingURL=promotion-policy-BBBcz5_3.d.ts.map
134
+ //# sourceMappingURL=promotion-policy-CvMda3kU.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"promotion-policy-BBBcz5_3.d.ts","names":[],"sources":["../src/campaign/gates/promotion-policy.ts"],"mappings":";;;;;;KA0CY;EAAoB;;EAAwB;EAAmB;;UAE1D;;EAEf;EACA,QAAQ;;;EAGR,WAAW;;;;EAIX;;;;;EAKA;;;KAIU;UAEK;EACf;EACA,QAAQ;EACR,WAAW;;;;;;;;EAQX,WAAW;;;;;EAKX;;;EAGA;IAAM;IAAa;;;EAEnB,mBAAmB;;EAEnB,SAAS;;;EAGT;;EAEA;EACA;EACA,gBAAgB;EAChB;EACA;EACA,SAAS;;UAGM;;EAEf,MAAM;;;EAGN;;;EAGA;IAAQ;IAAmB;;;;;;KAMjB,mBAAmB,IAAI,mBAAmB;UAErC;;;;EAIf;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;;iBASc,oBAAoB,WAAW,kBAAkB,UAC/D,KAAK,YAAY,WAAW,YAC5B,YAAY,sBACZ,OAAM,6BACL;;;;;;;;cAwIU,cAAc;UAiFV,sCAAsC;;EAErD,YAAY;;;EAGZ,SAAS;;EAET;;;;;;;iBAQc,uBAAuB,qBAAqB,kBAAkB,WAAW,UACvF,SAAS,gCACR,KAAK,WAAW"}
1
+ {"version":3,"file":"promotion-policy-CvMda3kU.d.ts","names":[],"sources":["../src/campaign/gates/promotion-policy.ts"],"mappings":";;;;;;KA0CY;EAAoB;;EAAwB;EAAmB;;UAE1D;;EAEf;EACA,QAAQ;;;EAGR,WAAW;;;;EAIX;;;;;EAKA;;;KAIU;UAEK;EACf;EACA,QAAQ;EACR,WAAW;;;;;;;;EAQX,WAAW;;;;;EAKX;;;EAGA;IAAM;IAAa;;;EAEnB,mBAAmB;;EAEnB,SAAS;;;EAGT;;EAEA;EACA;EACA,gBAAgB;EAChB;EACA;EACA,SAAS;;UAGM;;EAEf,MAAM;;;EAGN;;;EAGA;IAAQ;IAAmB;;;;;;KAMjB,mBAAmB,IAAI,mBAAmB;UAErC;;;;EAIf;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;;iBASc,oBAAoB,WAAW,kBAAkB,UAC/D,KAAK,YAAY,WAAW,YAC5B,YAAY,sBACZ,OAAM,6BACL;;;;;;;;cAwIU,cAAc;UAiFV,sCAAsC;;EAErD,YAAY;;;EAGZ,SAAS;;EAET;;;;;;;iBAQc,uBAAuB,qBAAqB,kBAAkB,WAAW,UACvF,SAAS,gCACR,KAAK,WAAW"}
@@ -1,170 +1,14 @@
1
- import { t as AgentEvalError } from "./errors-DEE6u6ot.js";
2
1
  import { c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt, o as CostLedger, p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
3
- import { a as RunRecord } from "./run-record-DQjRcYwA.js";
4
- import { _ as ProposalFinding } from "./types-DMoNFDWi.js";
5
- import { p as ChatClient } from "./types-Bfk0uxRj.js";
6
- import { B as ScoredSurfaceOutcome, C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-nokrtr7M.js";
2
+ import { a as RunRecord } from "./run-record-DTv1MdjK.js";
3
+ import { _ as ProposalFinding } from "./types-DN2WdT5S.js";
4
+ import { p as ChatClient } from "./types-gvRsyJLh.js";
5
+ import { B as ScoredSurfaceOutcome, C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-Ba5UQyVD.js";
7
6
  import { m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-D-UdhAmg.js";
8
7
  import { r as LedgerHash } from "./canonical-CFpojCN5.js";
9
- import { _ as ExternalTextCandidate, g as ExternalOptimizerWireCounts, o as ExternalOptimizerEvaluationObservation } from "./external-optimizer-contracts-szBJ_1vh.js";
8
+ import { _ as ExternalTextCandidate, g as ExternalOptimizerWireCounts, o as ExternalOptimizerEvaluationObservation } from "./external-optimizer-contracts-CQCpyrIL.js";
10
9
  import { r as DatasetScenario, t as Dataset } from "./dataset-DQqhOCPt.js";
11
- import { g as TraceSpanEvent, t as HostedClient } from "./client-Df7wdslk.js";
10
+ import { g as TraceSpanEvent, t as HostedClient } from "./client-DlqdbM7n.js";
12
11
  import { z } from "zod";
13
- //#region src/judge-families.d.ts
14
- /**
15
- * Judge model-family classification + cross-family enforcement.
16
- *
17
- * A judge ensemble built entirely from one provider family shares that
18
- * family's blind spots and self-preference — its "agreement" is correlated
19
- * bias, not independent signal. `assertCrossFamily` makes the consumer prove
20
- * the ensemble spans ≥2 families; `judgeFamily` is the single regex map that
21
- * replaces the per-consumer copies (tax/legal/creative/gtm each ship one).
22
- */
23
- /** Provider family a model belongs to. `unknown` when no rule matches. */
24
- type JudgeFamily = 'anthropic' | 'openai' | 'google' | 'meta' | 'mistral' | 'deepseek' | 'xai' | 'qwen' | 'cohere' | 'amazon' | 'moonshot' | 'zhipu' | 'unknown';
25
- /**
26
- * Classify a model id into its provider family. Strips a `@snapshot` suffix
27
- * and prefers an explicit `provider/...` prefix; otherwise matches the model
28
- * name. Returns `unknown` when nothing matches (callers decide whether that's
29
- * acceptable — `assertCrossFamily` counts it as its own family).
30
- */
31
- declare function judgeFamily(modelId: string): JudgeFamily;
32
- interface AssertCrossFamilyOptions {
33
- /** Minimum number of distinct families the ensemble must span. Default 2. */
34
- minFamilies?: number;
35
- /** When false (default), `unknown`-family models do NOT count toward the
36
- * family total — an ensemble of all-unclassifiable models is not provably
37
- * cross-family. Set true to count `unknown` as one shared family. */
38
- allowUnknown?: boolean;
39
- }
40
- declare class CrossFamilyError extends Error {
41
- readonly families: JudgeFamily[];
42
- readonly models: string[];
43
- constructor(message: string, families: JudgeFamily[], models: string[]);
44
- }
45
- /**
46
- * Throw unless the judge models span at least `minFamilies` distinct provider
47
- * families. Pass the model ids backing your judge ensemble. Fail-loud by
48
- * design — a correlated single-family ensemble silently inflates agreement.
49
- *
50
- * Scope: this reads the ids you REQUEST. It proves the panel was configured
51
- * across families; it cannot prove the panel RAN across families, because a
52
- * routing gateway may answer several different ids from one provider. Where
53
- * the diversity claim is load-bearing (a published leaderboard, a
54
- * certification, a non-self-judging exclusion), assert on the ids the
55
- * provider echoed instead: `assertCrossFamilyServed` in
56
- * ./integrity/served-model.
57
- */
58
- declare function assertCrossFamily(models: string[], opts?: AssertCrossFamilyOptions): JudgeFamily[];
59
- //#endregion
60
- //#region src/integrity/served-model.d.ts
61
- /** How a served id relates to the id that was requested. */
62
- type ServedModelVerdict =
63
- /** Byte-identical after normalisation — the requested model answered. */
64
- 'exact' |
65
- /** Same model, different spelling (provider prefix, snapshot, tier suffix). */
66
- 'alias' |
67
- /** A different model of the SAME provider family answered. */
68
- 'substituted-within-family' |
69
- /** A different provider's model answered. */
70
- 'substituted-cross-family' |
71
- /** The response carried no model id — identity is unproven either way. */
72
- 'unreported';
73
- interface ServedModelCheck {
74
- /** The id the caller asked for. */
75
- requested: string;
76
- /** The id echoed on the response; `null` when the response omitted it. */
77
- served: string | null;
78
- requestedFamily: JudgeFamily;
79
- /** `null` when `served` is null. */
80
- servedFamily: JudgeFamily | null;
81
- verdict: ServedModelVerdict;
82
- /** True for every verdict except `exact` and `alias`. */
83
- substituted: boolean;
84
- }
85
- /**
86
- * Classify one requested/served pair. Pure — no I/O — so it is safe inside
87
- * response handlers, reducers, and CI gates.
88
- *
89
- * `served` is the id echoed by the provider (OpenAI-compatible bodies put it
90
- * at `model`). `null`/`undefined` means the body omitted it; that is
91
- * `unreported`, NOT a pass — a provider that does not name what answered has
92
- * not proven identity, and a transport that drops the field must not read as
93
- * agreement.
94
- */
95
- declare function checkServedModel(requested: string, served: string | null | undefined): ServedModelCheck;
96
- declare class ModelSubstitutionError extends AgentEvalError {
97
- readonly checks: ReadonlyArray<ServedModelCheck>;
98
- constructor(message: string, checks: ReadonlyArray<ServedModelCheck>);
99
- }
100
- /**
101
- * Consumer-facing name for the substitution policy a metered surface applies.
102
- * `'exact'` rejects every substitution. `'allow-within-family'` accepts a
103
- * different model of the same provider family; it keeps family-level claims
104
- * valid and forfeits per-model claims. Maps to
105
- * `AssertServedModelOptions.allowWithinFamily`.
106
- */
107
- type ServedModelPolicy = 'exact' | 'allow-within-family';
108
- interface AssertServedModelOptions {
109
- /**
110
- * Accept a different model of the same provider family (e.g. requested
111
- * `deepseek-v3.2`, served `deepseek-v4-flash`). Default false. Setting this
112
- * keeps family-level claims valid and forfeits per-model claims.
113
- */
114
- allowWithinFamily?: boolean;
115
- /**
116
- * Accept a response that carried no model id. Default false — an
117
- * unidentified response cannot support a per-model or per-family claim.
118
- */
119
- allowUnreported?: boolean;
120
- /** Prefixed to the thrown message, e.g. the judge or campaign cell name. */
121
- context?: string;
122
- }
123
- /**
124
- * The one place the accept/reject policy lives, so a caller that reports
125
- * substitution (a preflight table, a run record) and a caller that throws on it
126
- * can never drift apart. A cross-family substitution is never acceptable.
127
- */
128
- declare function servedModelAcceptable(check: ServedModelCheck, opts?: AssertServedModelOptions): boolean;
129
- /**
130
- * Throw `ModelSubstitutionError` unless the served id is the requested model.
131
- * Returns the check on success so callers can record the served id alongside
132
- * the result.
133
- */
134
- declare function assertServedModel(requested: string, served: string | null | undefined, opts?: AssertServedModelOptions): ServedModelCheck;
135
- /**
136
- * Batch form: check every pair and throw naming EVERY substitution, so one
137
- * failure does not hide the rest. Returns all checks on success.
138
- */
139
- declare function assertServedModels(pairs: ReadonlyArray<{
140
- requested: string;
141
- served: string | null | undefined;
142
- }>, opts?: AssertServedModelOptions): ServedModelCheck[];
143
- interface AssertCrossFamilyServedOptions extends AssertServedModelOptions {
144
- /** Minimum distinct SERVED families required. Default 2. */
145
- minFamilies?: number;
146
- /** Count `unknown`-family served ids toward the total. Default false. */
147
- allowUnknown?: boolean;
148
- }
149
- declare class ServedCrossFamilyError extends AgentEvalError {
150
- readonly families: JudgeFamily[];
151
- readonly checks: ReadonlyArray<ServedModelCheck>;
152
- constructor(message: string, families: JudgeFamily[], checks: ReadonlyArray<ServedModelCheck>);
153
- }
154
- /**
155
- * Family-diversity rule over the models that actually ANSWERED.
156
- *
157
- * `assertCrossFamily` (../judge-families) reads the requested ids and so
158
- * cannot see a gateway that answers three "different" requests from one
159
- * provider. This one asserts no substitution first, then counts families from
160
- * the served ids — a panel that collapsed to one family under the hood fails
161
- * here even though its request list looked diverse.
162
- */
163
- declare function assertCrossFamilyServed(pairs: ReadonlyArray<{
164
- requested: string;
165
- served: string | null | undefined;
166
- }>, opts?: AssertCrossFamilyServedOptions): JudgeFamily[];
167
- //#endregion
168
12
  //#region src/campaign/storage.d.ts
169
13
  /**
170
14
  * `CampaignStorage` — the filesystem seam `runCampaign` writes through
@@ -1395,7 +1239,35 @@ interface TransientFailureOptions {
1395
1239
  readonly retryFullDurationTimeouts?: boolean;
1396
1240
  /** Additional caller-specific transient patterns. */
1397
1241
  readonly extraPatterns?: readonly RegExp[];
1242
+ /**
1243
+ * The instant a dated quota refusal is measured against. Defaults to `Date.now()`; inject it
1244
+ * to replay a past classification, which is what a retry audit needs.
1245
+ */
1246
+ readonly now?: number;
1398
1247
  }
1248
+ /**
1249
+ * The instant a provider says a spent quota works again, or null when the text states none.
1250
+ *
1251
+ * MEASURED (2026-09-01, discovery lab). The codex/ChatGPT backend answered
1252
+ * `You've hit your usage limit. Visit https://chatgpt.com/codex/settings/usage to purchase more
1253
+ * credits or try again at Sep 6th, 2026 8:29 PM.` — a refusal SIX DAYS out. 25 supervised runs met
1254
+ * it. 21 of them retried it 12 times over about 31 minutes and settled with zero children, zero
1255
+ * tokens and zero claims: about 11 hours of one subscription's capacity spent on a wall that had
1256
+ * already told the caller when it would come down.
1257
+ *
1258
+ * The distinction this draws is not "quota" versus "not quota". It is "the provider named a
1259
+ * release time" versus "it did not". A bare 429, or z.ai's `您的账户已达到速率限制`, recovers on
1260
+ * its own in seconds and SHOULD be retried; those return null here and keep their existing
1261
+ * treatment. Only a stated release is terminal, and only until that instant.
1262
+ *
1263
+ * A date with no zone is read in the host's zone, because a CLI renders it in the host's zone.
1264
+ * An unparseable date returns null rather than a guess: a caller that stops dispatching must
1265
+ * never do so on a misread string.
1266
+ *
1267
+ * @param message the provider's error text
1268
+ * @returns the stated release time, or null when the text names none
1269
+ */
1270
+ declare function quotaExhaustedUntil(message: string | null | undefined): Date | null;
1399
1271
  /**
1400
1272
  * True when the error text describes an infrastructure hiccup that should be
1401
1273
  * retried rather than scored. Empty/undefined input is not transient.
@@ -1408,7 +1280,9 @@ declare function isTransientTransportFailure(message: string | null | undefined,
1408
1280
  * dispatch already produced an artifact, so re-dispatching would score a
1409
1281
  * different sample. A per-cell dispatch deadline ("dispatch exceeded <N>ms")
1410
1282
  * is not transient by default; opt in via `extraPatterns` or
1411
- * `retryFullDurationTimeouts` when queue starvation eats the clock.
1283
+ * `retryFullDurationTimeouts` when queue starvation eats the clock. A provider refusal that
1284
+ * states its own release time is never retried while that time is in the future
1285
+ * (`quotaExhaustedUntil`).
1412
1286
  */
1413
1287
  declare function transientDispatchFailure(opts?: TransientFailureOptions): (failure: CampaignCellFailureReceipt['failure']) => boolean;
1414
1288
  //#endregion
@@ -2117,5 +1991,5 @@ interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends
2117
1991
  */
2118
1992
  declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
2119
1993
  //#endregion
2120
- export { readExternalOptimizerObservationArtifact as $, ServedModelCheck as $n, SearchLedgerAppendResult as $t, CanaryOptions as A, runEval as An, SearchHistoryPolicy as At, OptimizationMethodResult as B, CacheRead as Bn, SearchAccountingAudit as Bt, RedTeamReport as C, ParentSelectionContext as Cn, GepaCandidatePopulationSummary as Ct, CanaryAlert as D, OpenAutoPrResult as Dn, SearchHistoryAuditSummary as Dt, scoreRedTeamOutput as E, OpenAutoPrOptions as En, CreateSearchHistoryReceiptInput as Et, OptimizationMethod as F, CampaignRunPlan as Fn, createSearchHistoryReceipt as Ft, combineComparisonCosts as G, CampaignStorage as Gn, SearchCandidateRegisteredEvent as Gt, OptimizationMethodScore as H, CellScheduleSlot as Hn, SearchAttemptAccounting as Ht, OptimizationMethodComparison as I, CampaignRunPlanCell as In, searchHistoryCoverageRow as It, optimizationTokenUsageFromSummary as J, inMemoryCampaignStorage as Jn, SearchCandidateSurface as Jt, compareOptimizationMethods as K, createRunCostLedger as Kn, SearchCandidateSlot as Kt, OptimizationMethodInput as L, PlanCampaignRunOptions as Ln, verifySearchHistoryReceipt as Lt, runCanaries as M, CampaignCellRetryPolicy as Mn, SearchHistoryRequiredError as Mt, CompareOptimizationMethodsOptions as N, RunCampaignOptions as Nn, assertCompleteSearchHistory as Nt, CanaryEvaluation as O, openAutoPr as On, SearchHistoryCoverage as Ot, ComparisonCost as P, runCampaign as Pn, assertSearchHistoryMatchesReplay as Pt, ExternalOptimizerSubmittedCandidate as Q, ServedCrossFamilyError as Qn, SearchLedger as Qt, OptimizationMethodPairwise as R, planCampaignRun as Rn, FileSearchLedger as Rt, RedTeamFinding as S, CrowdedFrontierParentOptions as Sn, GepaCandidatePopulationCandidate as St, redTeamReport as T, crowdedFrontierParent as Tn, readGepaCandidatePopulationArtifact as Tt, OptimizationPackageSource as U, buildCellSchedule as Un, SearchCandidateDecidedEvent as Ut, OptimizationMethodRunOptions as V, readCachedCell as Vn, SearchArtifactRef as Vt, OptimizationTokenUsage as W, cellCachePath as Wn, SearchCandidateLineage as Wt, ExternalOptimizerObservationArtifact as X, AssertServedModelOptions as Xn, SearchCostAccounting as Xt, ExternalOptimizerExecutionSummary as Y, AssertCrossFamilyServedOptions as Yn, SearchCompletedEvent as Yt, ExternalOptimizerObservationSummary as Z, ModelSubstitutionError as Zn, SearchFailureReason as Zt, provenanceSpansPath as _, SearchTaskAttemptedEvent as _n, SearchRecorder as _t, LoopProvenanceBackend as a, SearchModelIdentity as an, checkServedModel as ar, transientDispatchFailure as at, RedTeamCase as b, openSearchLedger as bn, recordCandidatePopulationSearch as bt, LoopProvenanceOptimizationMethod as c, SearchPlan as cn, CrossFamilyError as cr, runImprovementLoop as ct, campaignMeasurementDigest as d, SearchPlannedOperation as dn, judgeFamily as dr, RunOptimizationResult as dt, SearchLedgerEntry as en, ServedModelPolicy as er, LlmJudgeDimension as et, canonicalDigest as f, SearchPlannedTask as fn, runOptimization as ft, provenanceRecordPath as g, SearchSurfaceKind as gn, SearchLedgerBinding as gt, loopProvenanceSpans as h, SearchSurfaceEvidence as hn, SearchExecutionIdentity as ht, LoopProvenanceArgsFromResult as i, SearchLedgerTrustedHeadMode as in, assertServedModels as ir, isTransientTransportFailure as it, CanaryReport as j, CampaignCellFailureReceipt as jn, SearchHistoryReceipt as jt, CanaryKind as k, RunEvalOptions as kn, SearchHistoryCoverageRow as kt, LoopProvenanceRecord as l, SearchPlanExtendedEvent as ln, JudgeFamily as lr, PremeasuredOptimizationBaseline as lt, loopProvenanceArgsFromResult as m, SearchSurfaceEffect as mn, ProposedSearchCandidate as mt, EmitLoopProvenanceArgs as n, SearchLedgerHash as nn, assertCrossFamilyServed as nr, llmJudge as nt, LoopProvenanceCandidate as o, SearchOperationKind as on, servedModelAcceptable as or, RunImprovementLoopOptions as ot, emitLoopProvenance as p, SearchSourceRef as pn, MeasuredSearchCandidate as pt, costFromLedgerSummary as q, fsCampaignStorage as qn, SearchCandidateSlotClosedEvent as qt, EmitLoopProvenanceResult as r, SearchLedgerReplay as rn, assertServedModel as rr, TransientFailureOptions as rt, LoopProvenanceEvidence as s, SearchOperationRecordedEvent as sn, AssertCrossFamilyOptions as sr, RunImprovementLoopResult as st, BuildLoopProvenanceArgs as t, SearchLedgerEvent as tn, ServedModelVerdict as tr, LlmJudgeOptions as tt, buildLoopProvenanceRecord as u, SearchPlannedEvent as un, assertCrossFamily as ur, RunOptimizationOptions as ut, verifyLoopProvenanceRecord as v, SearchTaskOutcome as vn, SearchRecorderOptions as vt, redTeamDataset as w, ParentSelector as wn, GepaCandidateSelectionScore as wt, RedTeamCategory as x, validateSearchLedgerEvent as xn, GepaCandidatePopulationArtifact as xt, DEFAULT_RED_TEAM_CORPUS as y, SearchTokenAccounting as yn, SearchRunIdentity as yt, OptimizationMethodProvenance as z, CacheIssueReason as zn, OpenSearchLedgerOptions as zt };
2121
- //# sourceMappingURL=provenance-Dp-vvyrU.d.ts.map
1994
+ export { readExternalOptimizerObservationArtifact as $, SearchLedger as $t, CanaryOptions as A, RunEvalOptions as An, SearchHistoryCoverageRow as At, OptimizationMethodResult as B, CacheIssueReason as Bn, OpenSearchLedgerOptions as Bt, RedTeamReport as C, CrowdedFrontierParentOptions as Cn, GepaCandidatePopulationCandidate as Ct, CanaryAlert as D, OpenAutoPrOptions as Dn, CreateSearchHistoryReceiptInput as Dt, scoreRedTeamOutput as E, crowdedFrontierParent as En, readGepaCandidatePopulationArtifact as Et, OptimizationMethod as F, runCampaign as Fn, assertSearchHistoryMatchesReplay as Ft, combineComparisonCosts as G, cellCachePath as Gn, SearchCandidateLineage as Gt, OptimizationMethodScore as H, readCachedCell as Hn, SearchArtifactRef as Ht, OptimizationMethodComparison as I, CampaignRunPlan as In, createSearchHistoryReceipt as It, optimizationTokenUsageFromSummary as J, fsCampaignStorage as Jn, SearchCandidateSlotClosedEvent as Jt, compareOptimizationMethods as K, CampaignStorage as Kn, SearchCandidateRegisteredEvent as Kt, OptimizationMethodInput as L, CampaignRunPlanCell as Ln, searchHistoryCoverageRow as Lt, runCanaries as M, CampaignCellFailureReceipt as Mn, SearchHistoryReceipt as Mt, CompareOptimizationMethodsOptions as N, CampaignCellRetryPolicy as Nn, SearchHistoryRequiredError as Nt, CanaryEvaluation as O, OpenAutoPrResult as On, SearchHistoryAuditSummary as Ot, ComparisonCost as P, RunCampaignOptions as Pn, assertCompleteSearchHistory as Pt, ExternalOptimizerSubmittedCandidate as Q, SearchFailureReason as Qt, OptimizationMethodPairwise as R, PlanCampaignRunOptions as Rn, verifySearchHistoryReceipt as Rt, RedTeamFinding as S, validateSearchLedgerEvent as Sn, GepaCandidatePopulationArtifact as St, redTeamReport as T, ParentSelector as Tn, GepaCandidateSelectionScore as Tt, OptimizationPackageSource as U, CellScheduleSlot as Un, SearchAttemptAccounting as Ut, OptimizationMethodRunOptions as V, CacheRead as Vn, SearchAccountingAudit as Vt, OptimizationTokenUsage as W, buildCellSchedule as Wn, SearchCandidateDecidedEvent as Wt, ExternalOptimizerObservationArtifact as X, SearchCompletedEvent as Xt, ExternalOptimizerExecutionSummary as Y, inMemoryCampaignStorage as Yn, SearchCandidateSurface as Yt, ExternalOptimizerObservationSummary as Z, SearchCostAccounting as Zt, provenanceSpansPath as _, SearchSurfaceKind as _n, SearchLedgerBinding as _t, LoopProvenanceBackend as a, SearchLedgerTrustedHeadMode as an, quotaExhaustedUntil as at, RedTeamCase as b, SearchTokenAccounting as bn, SearchRunIdentity as bt, LoopProvenanceOptimizationMethod as c, SearchOperationRecordedEvent as cn, RunImprovementLoopResult as ct, campaignMeasurementDigest as d, SearchPlannedEvent as dn, RunOptimizationOptions as dt, SearchLedgerAppendResult as en, LlmJudgeDimension as et, canonicalDigest as f, SearchPlannedOperation as fn, RunOptimizationResult as ft, provenanceRecordPath as g, SearchSurfaceEvidence as gn, SearchExecutionIdentity as gt, loopProvenanceSpans as h, SearchSurfaceEffect as hn, ProposedSearchCandidate as ht, LoopProvenanceArgsFromResult as i, SearchLedgerReplay as in, isTransientTransportFailure as it, CanaryReport as j, runEval as jn, SearchHistoryPolicy as jt, CanaryKind as k, openAutoPr as kn, SearchHistoryCoverage as kt, LoopProvenanceRecord as l, SearchPlan as ln, runImprovementLoop as lt, loopProvenanceArgsFromResult as m, SearchSourceRef as mn, MeasuredSearchCandidate as mt, EmitLoopProvenanceArgs as n, SearchLedgerEvent as nn, llmJudge as nt, LoopProvenanceCandidate as o, SearchModelIdentity as on, transientDispatchFailure as ot, emitLoopProvenance as p, SearchPlannedTask as pn, runOptimization as pt, costFromLedgerSummary as q, createRunCostLedger as qn, SearchCandidateSlot as qt, EmitLoopProvenanceResult as r, SearchLedgerHash as rn, TransientFailureOptions as rt, LoopProvenanceEvidence as s, SearchOperationKind as sn, RunImprovementLoopOptions as st, BuildLoopProvenanceArgs as t, SearchLedgerEntry as tn, LlmJudgeOptions as tt, buildLoopProvenanceRecord as u, SearchPlanExtendedEvent as un, PremeasuredOptimizationBaseline as ut, verifyLoopProvenanceRecord as v, SearchTaskAttemptedEvent as vn, SearchRecorder as vt, redTeamDataset as w, ParentSelectionContext as wn, GepaCandidatePopulationSummary as wt, RedTeamCategory as x, openSearchLedger as xn, recordCandidatePopulationSearch as xt, DEFAULT_RED_TEAM_CORPUS as y, SearchTaskOutcome as yn, SearchRecorderOptions as yt, OptimizationMethodProvenance as z, planCampaignRun as zn, FileSearchLedger as zt };
1995
+ //# sourceMappingURL=provenance-CRY67X50.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"provenance-CRY67X50.d.ts","names":[],"sources":["../src/campaign/storage.ts","../src/campaign/cell-schedule.ts","../src/campaign/cell-cache.ts","../src/campaign/plan-campaign-run.ts","../src/campaign/run-campaign.ts","../src/campaign/presets/run-eval.ts","../src/campaign/auto-pr.ts","../src/campaign/parent-selection.ts","../src/campaign/search-ledger.ts","../src/campaign/search-history-receipt.ts","../src/campaign/gepa-candidate-population.ts","../src/campaign/search-ledger-recording.ts","../src/campaign/presets/run-optimization.ts","../src/campaign/presets/run-improvement-loop.ts","../src/campaign/transient-failure.ts","../src/llm-judge.ts","../src/campaign/external-optimizer-observations.ts","../src/campaign/presets/compare-optimization-methods.ts","../src/canary.ts","../src/red-team.ts","../src/campaign/provenance.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;UAoBiB;;EAEf,UAAU;;EAEV,OAAO;;EAEP,KAAK;;EAEL,MAAM,cAAc,kBAAkB;;;EAGtC,OAAO,cAAc,iBAAiB;;;;;;;;;iBAUxB,qBAAqB;;;;iBA0CrB,2BAA2B;;iBAmC3B,oBAAoB;EAClC,SAAS;EACT;EACA;;EAEA;IACE;;;iBClHY,kBAAkB,kBAAkB,UAClD,WAAW,aACX,cACA,eACC;EAAQ,UAAU;EAAW;EAAa;EAAgB;;KA4BjD,iBAAiB,kBAAkB,YAAY,kBAClD,kBAAkB;iBAOX,cAAc,gBAAgB;;;KCkElC;KASA,UAAU;EAChB;EAAe,MAAM,mBAAmB;;EACxC;EAAgB,QAAQ;;iBAEd,eAAe,WAAW;EACxC,SAAS;EACT;EACA;EACA;IACE,UAAU;;;UCxHG;EACf;EACA;EACA;EACA;EACA;EACA;EACA,SAAS;;UAGM;EACf;EACA;EACA;EACA;EACA;EACA;EACA,OAAO;;UAGQ,uBAAuB,kBAAkB,UAAU;EAClE,WAAW;EACX,WAAW,WAAW,WAAW;EACjC;EACA,SAAS,YAAY,WAAW;EAChC;EACA;EACA;;EAEA;EACA;;EAEA;EACA,UAAU;;EAEV,aAAa;;EAEb,WAAW,SAAS;;;;;;iBAON,gBAAgB,kBAAkB,UAAU,WAC1D,MAAM,uBAAuB,WAAW,aACvC;;;UCjBc,mBAAmB,kBAAkB,UAAU;EAC9D,WAAW;EACX,UAAU,WAAW,WAAW;;EAEhC,SAAS;;;;;;EAMT;EACA,SAAS,YAAY,WAAW;;EAEhC;;;EAGA;;;EAGA;;;;;;EAMA,cAAc;IAAS,UAAU;IAAW;;;;;;EAK5C;;;;;;EAMA;;;;EAIA,eAAe;EACf;EACA;;EAEA;;;EAGA,aAAa;;EAEb;;EAEA,WAAW,SAAS;;EAEpB;;;;;;;;;;EAUA;;;;;;;;EAQA,YAAY;;;;;;;;;;EAUZ;;;;;EAKA;;;;EAIA;;;EAGA;;;;EAIA;;;;;;;;;;;EAWA;;EAEA,YAAY;;EAEZ,oBAAoB,gBAAgB,gBAAgB;;;;;;EAMpD,UAAU;;;;;;;;;;;;EAYV,iBAAiB;IACf,UAAU;IACV;IACA;;;;;;;UAQa,2BAA2B;EAC1C;EACA;EACA;EACA;IACE;IACA;IACA;MACE;MACA;MACA;;;EAGJ,MAAM,mBAAmB;EACzB,MAAM;;;;;;;;;;;;UAaS;;EAEf;;;;;EAKA,YAAY,SAAS;;;;;iBAMD,YAAY,kBAAkB,UAAU,WAC5D,MAAM,mBAAmB,WAAW,aACnC,QAAQ,eAAe,WAAW;;;UCvNpB,eAAe,kBAAkB,UAAU,mBAClD,KAAK,mBAAmB,WAAW;EAC3C;;;;;iBAMoB,QAAQ,kBAAkB,UAAU,WACxD,MAAM,eAAe,WAAW,aAC/B,QAAQ,eAAe,WAAW;;;UCHpB,kBAAkB,WAAW,kBAAkB;;EAE9D,QAAQ,eAAe,WAAW;;;EAGlC,MAAM;;;EAGN;;EAEA;EACA;;EAEA;;EAEA;;;EAGA;;EAEA,UAAU;IAAqB;IAAgB;IAAgB;;;UAGhD;EACf;EACA;EACA;EACA;;;;;iBAMc,WAAW,WAAW,kBAAkB,UACtD,SAAS,kBAAkB,WAAW,aACrC;;;;UCtCc;;;WAGN,UAAU,cAAc;;;WAGxB,WAAW;;WAEX,SAAS,cAAc;;WAEvB;;;;;KAMC,kBAAkB,KAAK,2BAA2B;UAE7C;;;EAGf;;;;;;;;;;;iBAYc,sBAAsB,SAAS,+BAA+B;;;cCPxE;KAEM,mBAAmB;KAEnB;;;UAYK;EACf;EACA;EACA,QAAQ;EACR;;;;UAKe;EACf;EACA;;UAGe;EACf;EACA;;UAGe;EACf;EACA,MAAM;EACN,UAAU;;UAGK;;EAEf;EACA;EACA;EACA;EACA,gBAAgB;;KAGN;UAOK;EACf;EACA,QAAQ;EACR,WAAW;;;EAGX;;UAGe;EACf;EACA,MAAM;;UAGS;EACf;;;EAGA;;UAGe;;EAEf,gBAAgB;;EAEhB,OAAO;;EAEP,YAAY;;KAGF;EAEN;EACA;EACA;EACA;;EAGA;EACA;;KAGM;EAEN;EACA;EACA;;EAGA;;EAEA;EACA;;UAGW;EACf,QAAQ;EACR,MAAM;;UAGS;EACf;EACA;;KAGU;EAEN;EACA;EACA,SAAS;;EAGT;EACA;EACA,SAAS;EACT,SAAS;;EAGT;EACA,SAAS;EACT,OAAO;IAAwB;;;KAGzB;EAEN;EACA;EACA;EACA;EACA;;EAGA;EACA;;;;UAKW;EACf;EACA;EACA;EACA,QAAQ;EACR,UAAU;;UAGF;EACR;EACA;EACA,WAAW;;UAGI,2BAA2B;EAC1C;EACA,MAAM;;;;;;UAOS,gCAAgC;EAC/C;EACA;IACE,gBAAgB;IAChB,YAAY;;;UAIC,uCAAuC;EACtD;EACA;EACA;EACA;EACA,SAAS;EACT,UAAU;;UAGK,uCAAuC;EACtD;EACA;EACA;EACA,QAAQ;;UAGO,iCAAiC;EAChD;EACA;EACA;EACA;EACA;IACE;IACA,QAAQ;;EAEV;IACE,OAAO;IACP,OAAO;IACP,WAAW;;EAEb,SAAS;EACT,YAAY;EACZ,iBAAiB;;UAGF,qCAAqC;EACpD;EACA;EACA,eAAe;EACf;IAEM;IACA,OAAO;IACP,QAAQ;;IAGR;IACA,QAAQ;;EAEd;IACM;;IACA;IAAmB,SAAS;;IAC5B;IAAkB,SAAS;;EACjC,YAAY;;UAGG,oCAAoC;EACnD;EACA;EACA;IACM;;IAEA;IACA,QAAQ;;;UAIC,6BAA6B;EAC5C;EACA;IAEM;IACA;;IAGA;IACA,QAAQ;;;KAIJ,oBACR,qBACA,0BACA,iCACA,iCACA,2BACA,+BACA,8BACA;UAEa;EACf,eAAe;EACf;EACA;EACA,cAAc;EACd,OAAO;EACP,WAAW;;KAGD;EAEN;EACA;EACA;EACA;EACA;;EAGA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGW;EACf;EACA;EACA;EACA;EACA;EACA;EACA;IAAY;IAAgB;IAAgB;;EAC5C;IAAqB;IAAmB;IAAiB;;EACzD;IAAa;IAAkB;IAAkB;;EACjD;IACE;IACA;IACA;IACA;IACA;IACA;;EAEF;EACA;EACA,YAAY;EACZ,UAAU;;UAGK;EACf,SAAS;EACT,MAAM;;;EAGN,gBAAgB;EAChB,YAAY;EACZ,sBAAsB;EACtB,UAAU;EACV,YAAY;EACZ,WAAW;EACX,YAAY;EACZ,OAAO;;UAGQ;EACf,OAAO;;EAEP;EACA,QAAQ;;;;iBA8aM,0BAA0B,iBAAiB;;;;;;;;;;;;;;;KAsB/C;UAEK;EACf;EACA;EACA,cAAc;;UAGC;WACN;WACA;;WAEA;EACT,OAAO,OAAO,oBAAoB,QAAQ;EAC1C,UAAU,QAAQ;;EAElB,eAAe,QAAQ;;;EAGvB,kBAAkB,QAAQ;;;;;;;EAO1B,oBAAoB,QAAQ;;;;iBAKd,iBAAiB,SAAS,0BAA0B;;cA8BvD,4BAA4B;WAC9B;WACA;WACA;mBACQ;mBACA;EAMjB,YAAY,cAAc,oBAAoB,cAAa;EAerD,UAAU,QAAQ;EAIlB,OAAO,OAAO,oBAAoB,QAAQ;EAU1C,eAAe,QAAQ;EAIvB,kBAAkB,QAAQ;EAI1B,oBAAoB,QAAQ;;;;cC56B9B;cACA;;UAGW;WACN;WACA,UAAU;WACV,QAAQ;WACR;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;;;;;;;;;UAWM;WACN,sBAAsB;WACtB;WACA,wBAAwB;WACxB,eAAe;;WAEf;;WAEA;WACA,QAAQ;;WAER,aAAa;WACb,SAAS;WACT;WACA;;UAGM;WACN;WACA;;WAEA,QAAQ;;WAER,QAAQ;;KAGP;UAEK;WACN;WACA;WACA;WACA,UAAU;;UAGJ;WACN,QAAQ;WACR;WACA,oBAAoB;;cAGlB,mCAAmC;WACrC;WACA;EAET,YAAY,oBAAoB;;;iBAUlB,2BACd,OAAO,kCACN;;iBA2Ba,2BAA2B,SAAS,uBAAuB;;iBA6C3D,iCACd,SAAS,sBACT,QAAQ;;iBAeM,4BACd,oBACA,SAAS,2CACA,WAAW;;iBAcN,yBACd,oBACA,SAAS,mCACR;;;UCrMc;WACN;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;UAGM;WACN;WACA;;UAGM;;WAEN;WACA,WAAW;WACX;WACA;;WAEA;;WAEA;WACA,0BAA0B;WAC1B;;UAGM;WACN,SAAS;WACT;WACA;WACA,qBAAqB;;;;;;;;;;iBAWhB,oCAAoC;EAClD,SAAS;EACT,UAAU;IACR;;;;KCpBQ,0BAA0B;;UAGrB;;EAEf,OAAO;;EAEP,UAAU;;EAEV,QAAQ;;EAER,OAAO;;UAGQ;EACf,QAAQ;EACR,UAAU;;;UAIK;EACf,SAAS;EACT;EACA;;;UAIe,wBAAwB;EACvC,SAAS;EACT;EACA,OAAO,cAAc,mBAAmB;EACxC;;EAEA;;UAGe,sBAAsB,kBAAkB;EACvD,SAAS;EACT,SAAS;EACT;EACA,WAAW,cAAc;EACzB;EACA;EACA;;EAEA;;EAEA;EACA,YAAY;;;;;;;;;;cAmBD,eAAe,kBAAkB,UAAU;mBACrC;mBACA;mBACA;mBACA;mBACA;mBACA;mBACA;UACT;UACA;UAED;;;SAOM,KAAK,kBAAkB,UAAU,WAC5C,MAAM,sBAAsB,aAC3B,QAAQ,eAAe,WAAW;;;;;;EAY/B,iBAAiB;IACrB;IACA;IACA,YAAY,cAAc;MACxB;;EAgEE,cACJ,YAAY,cAAc,wBAAwB,cACjD;;;;;;;;;;EAmEG,OAAO;IACX;IACA;IACA;MACE,QAAQ;;EA0FN,QAAQ,gBAAgB,QAAQ;;UAkBxB;UAsBA;UAkCN;UAIA;UAMM;UA8BA;;UAmBN;UAOA;UAWA;;UAcA;UAcM;;UAON;;;;;;;;;;;;;;;;;iBAgKY,gCAAgC,kBAAkB,UAAU;EAChF,QAAQ;EACR,SAAS;EACT;EACA,UAAU;EACV,YAAY;;EAEZ,WAAW,cAAc;;EAEzB,sBAAsB;EACtB;EACA;IACE,QAAQ;;;UCrqBK,gCAAgC,WAAW,kBAAkB;;EAE5E;;EAEA,UAAU,eAAe,WAAW;;UAGrB,2BAA2B,kBAAkB,UAAU,mBAC9D,KAAK,mBAAmB,WAAW;;EAE3C,iBAAiB;;;;;;;;;EASjB,sBAAsB,gCAAgC,WAAW;;EAEjE,sBACE,SAAS,gBACT,UAAU,WACV,KAAK,WAAW,mBAAmB,WAAW,+BAC3C,QAAQ;;EAEb,UAAU,gBAAgB;EAC1B;EACA;;;EAGA;;;EAGA;;EAEA,WAAW,cAAc;;;;;;;;;;;EAWzB,qBAAqB;IACnB;IACA;IACA,YAAY;MACV;MACA,UAAU,eAAe,WAAW;MACpC;;IAEF,SAAS;;IAET,aAAa;IACb;QACI,QAAQ,cAAc;;;;;;;;;;;;;;;;;;EAkB5B,oBAAoB,UAAU,eAAe,WAAW;;;;;;;;;;;;;EAaxD,eAAe;;;;;;;;;;;;EAYf,eAAe;;KAGL,uBACV,kBAAkB,UAClB,aACE,2BAA2B,WAAW;UAEzB,sBAAsB,WAAW,kBAAkB;EAClE,aAAa;IACX,QAAQ;IACR,UAAU;MACR;MACA,SAAS;MACT,UAAU,eAAe,WAAW;;;;EAIxC,iBAAiB;EACjB,eAAe;EACf;;;;EAIA;;;;EAIA;EACA,kBAAkB,eAAe,WAAW;;EAE5C,MAAM;;;;EAIN,gBAAgB;;;;;;EAMhB,gBAAgB;;;;;iBAMI,gBAAgB,kBAAkB,UAAU,WAChE,MAAM,uBAAuB,WAAW,aACvC,QAAQ,sBAAsB,WAAW;;;KCtLhC,0BACV,kBAAkB,UAClB,aACE,uBAAuB,WAAW;;;EAGpC,kBAAkB;;;;;;;;;EASlB;;;;EAIA,MAAM,KAAK,WAAW;;;;;EAKtB;;EAEA;EACA;;;;;;;;EAQA,cAAc,eAAe,gBAAgB,iBAAiB,mBAAmB;;UAGlE,yBAAyB,WAAW,kBAAkB,kBAC7D,sBAAsB,WAAW;EACzC,mBAAmB,eAAe,WAAW;EAC7C,iBAAiB,eAAe,WAAW;EAC3C,uBAAuB,eAAe,WAAW;EACjD,qBAAqB;EACrB,YAAY,QAAQ,WAAW,KAAK,WAAW;;;;EAI/C;;;;;EAKA;EACA,WAAW,kBAAkB;;;;;iBAMT,mBAAmB,kBAAkB,UAAU,WACnE,MAAM,0BAA0B,WAAW,aAC1C,QAAQ,yBAAyB,WAAW;;;UC1D9B;;;;;;;WAON;;WAEA,yBAAyB;;;;;WAKzB;;;;;;;;;;;;;;;;;;;;;;;;iBAiFK,oBAAoB,qCAAqC;;;;;iBAiCzD,4BACd,oCACA,OAAM;;;;;;;;;;;;iBAyBQ,yBACd,OAAM,2BACJ,SAAS;;;;;KCzID,6BAA6B;UAExB,gBAAgB,WAAW,kBAAkB,WAAW;;;;EAIvE,MAAM;;;EAGN,aAAa;;EAEb;;EAEA;EACA;EACA;;;EAGA,UAAU;;;;;;;;;;;;;;;;EAgBV;IAAY;;IAAwB;IAAuB;;;;;;EAK3D;;EAEA,aAAa,UAAU;;;EAGvB,cAAc;IAAS,UAAU;IAAW,UAAU;;;EAEtD,aAAa;EACb;IAAmB;IAAc,QAAQ,EAAE;;;;;;;;;;;;;iBAoB7B,SAAS,qBAAqB,kBAAkB,WAAW,UACzE,cACA,gBACA,MAAM,gBAAgB,WAAW,aAChC,YAAY,WAAW;;;UCxGT;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;;WAEN,WAAW;;WAEX;WACA;WACA;;WAEA;aACE;aACA;;;UAII;WACN,SAAS;WACT,uBAAuB;;WAEvB,qBAAqB;;;;;;;;;;iBAWhB,yCAAyC;EACvD,SAAS;EACT,UAAU;IACR;;;;KCfQ,6BAA6B,kBAAkB,UAAU,aAAa,KAChF,mBAAmB,WAAW;;UAKf;;EAEf;;EAEA,gBAAgB;EAChB;EACA;;UAGe;EACf;;EAEA;EACA;EACA;EACA;EACA;;EAEA;;UAGe;EACf;EACA;;UAGe;EACf;EACA;;UAGe;;EAEf;;EAEA;;EAEA;EACA;;EAEA;EACA;EACA;;UAGe;;EAEf,QAAQ;;EAER,SAAS;;EAET,UAAU;;EAEV,SAAS;;EAET;;EAEA;EACA;;EAEA;EACA;;EAEA;;EAEA;;;;;;EAMA;EACA;EACA,aAAa;;EAEb,eAAe;;EAEf,0BAA0B;;EAE1B,kBAAkB;;EAElB,oBAAoB;;;UAIL,wBAAwB,kBAAkB,UAAU;;WAE1D,iBAAiB;;WAEjB,yBAAyB;;WAEzB,6BAA6B;;WAE7B,sBACP,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;;WAEJ,iBAAiB,YAAY,WAAW;;WAExC;WACA;;WAEA,YAAY,SAAS,6BAA6B,WAAW;;WAE7D,YAAY;;UAGN;;EAEf,eAAe;;EAEf,MAAM;;EAEN;;EAEA,aAAa;;EAEb,gBAAgB;;;UAID,mBAAmB,kBAAkB,WAAW,UAAU;;EAEzE;EACA,WACE,OAAO,wBAAwB,WAAW,eACvC,QAAQ;;UAGE;EACf;;EAEA;;EAEA;;EAEA;;;EAGA;IAAU;IAAa;;;EAEvB,kBAAkB;;EAElB;;EAEA,aAAa;;EAEb,gBAAgB;IACd;IACA;IACA;IACA;;EAEF,eAAe;;EAEf;;UAGe;;EAEf;EACA;;EAEA;EACA;EACA;;EAEA;;UAGe;;EAEf,QAAQ;EACR,MAAM;;EAEN,UAAU;EACV;;EAEA,kBAAkB;;EAElB,UAAU;;EAEV,WAAW;;EAEX;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,eAAe;;UAGA,kCAAkC,kBAAkB,UAAU,mBACrE,KAAK,mBAAmB,WAAW;EAC3C,SAAS,mBAAmB,WAAW;EACvC,iBAAiB;;EAEjB,gBAAgB;;EAEhB,oBAAoB;;EAEpB,eAAe;;EAEf,sBACE,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;EACb,QAAQ,YAAY,WAAW;;;EAG/B;;EAEA,yBAAyB,6BAA6B,WAAW;;EAEjE;;;EAGA;;EAEA;;;;;EAKA,sBAAsB;;;;;iBAMF,2BAA2B,kBAAkB,UAAU,WAC3E,MAAM,kCAAkC,WAAW,aAClD,QAAQ;;iBAqxBK,sBAAsB,SAAS,oBAAoB;;iBAYnD,kCACd,SAAS,mBACT,mBAAmB,gBAClB;;iBA0Ba,uBACd,SAAS;EAAgB;EAAe,MAAM;KAC7C;;;KCxkCS;KAEA;UAEK;EACf,MAAM;EACN,UAAU;EACV;;;EAGA,UAAU;;UAGK;EACf,QAAQ;;EAER,QAAQ,OAAO;;EAEf,aAAa;;UAGE;EACf,MAAM;EACN;EACA;EACA;;UAGe;;;;;;;;;;EAUf;IACE;IACA;;IAEA;;;;;;;;;;;;;EAcF;IACE;IACA;IACA;IACA;;;;;;;;;;EAWF;IACE,WAAW,KAAK;IAChB;IACA;IACA;IACA;;;;;;;;iBASY,YAAY,MAAM,aAAa,OAAM,gBAAqB;;;KClG9D;UAUK;EACf,UAAU;;EAEV;;;;;EAKA;;EAEA;;EAEA;;UAGe,oBAAoB;EACnC,SAAS;;UAGM;EACf;EACA,UAAU;EACV;EACA;EACA;;UAGe;EACf,UAAU;EACV,oBAAoB,OAAO;EAC3B;;;cA0DW,yBAAyB;iBA6FtB,eAAe,aAAY,gBAAqB;;;;;iBAkBhD,mBACd,gBACA,qBACA,QAAQ,cACP;;iBA4Fa,cAAc,UAAU,mBAAmB;;;UCrQ1C;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;EAIA,cAAc,SAAS;;EAEvB;;EAEA;;EAEA;;EAEA;;EAEA,UAAU,YAAY;;EAEtB;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA;EACA;EACA;EACA;;UAGe;EACf;IACE;IACA;;EAEF;IACE;IACA;IACA;IACA;MACE;MACA;MACA;MACA;;;EAGJ;;UAGe;EACf;EACA,MAAM;EACN;EACA,aAAa;;;;;;;UAQE;EACf;;EAEA;EACA;EACA;EACA;;EAEA;EACA;;EAEA;EACA;;EAEA;;EAEA,YAAY;;EAEZ,qBAAqB;;EAErB,UAAU;;EAEV;;EAEA;IACE,UAAU;IACV;IACA;IACA,mBAAmB;;;;;;EAMrB;;EAEA;;EAEA;;;EAGA;;EAEA,SAAS;EACT;EACA;;UAGe,wBAAwB,WAAW,kBAAkB;EACpE;EACA;EACA;EACA,iBAAiB;EACjB,eAAe;EACf;EACA;;EAEA,wBAAwB,eAAe,WAAW;;EAElD,aAAa;IACX;IACA,YAAY;IACZ;;;IAGA,UAAU;MACR;MACA,SAAS;MACT,UAAU,eAAe,WAAW;;;EAGxC,MAAM;;;;EAIN;EACA,mBAAmB,eAAe,WAAW;EAC7C,iBAAiB,eAAe,WAAW;EAC3C,qBAAqB;EACrB,uBAAuB,eAAe,WAAW;;EAEjD,cAAc,cAAc;EAC5B;EACA;EACA,qBAAqB;;UAGN,6BAA6B,WAAW,kBAAkB;EACzE;EACA;EACA;EACA,iBAAiB;EACjB,QAAQ,yBAAyB,WAAW;EAC5C,cAAc,cAAc;EAC5B;EACA;;;iBAIc,6BAA6B,WAAW,kBAAkB,UACxE,OAAO,6BAA6B,WAAW,aAC9C,wBAAwB,WAAW;;iBA4CtB,0BAA0B,WAAW,kBAAkB,UACrE,MAAM,wBAAwB,WAAW,aACxC;;iBAqRa,0BAA0B,WAAW,kBAAkB,UACrE,UAAU,eAAe,WAAW;;iBAqCtB,2BAA2B,QAAQ,uBAAuB;;;iBAc1D,gBAAgB;;;;;;;;;;;;iBA2JhB,oBACd,QAAQ,sBACR;EAAQ;IACP;;iBAiJa,qBAAqB;;;;iBAMrB,oBAAoB;UAInB;EACf,QAAQ;EACR,OAAO;;EAEP;EACA;;UAGe,uBAAuB,WAAW,kBAAkB,kBAC3D,wBAAwB,WAAW;;EAE3C,SAAS;;;EAGT,eAAe;;;;;;;;;;;;iBA+EK,mBAAmB,WAAW,kBAAkB,UACpE,MAAM,uBAAuB,WAAW,aACvC,QAAQ"}
@@ -1,6 +1,6 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
2
  import { r as canonicalString } from "./canonical-DPyQ_rpt.js";
3
- import { a as isToolSpan, i as isLlmSpan, r as isJudgeSpan } from "./schema-CdIX2aHu.js";
3
+ import { a as isToolSpan, i as isLlmSpan, r as isJudgeSpan } from "./schema-CSf6qWgZ.js";
4
4
  //#region src/trace/query.ts
5
5
  /**
6
6
  * Typed query helpers over TraceStore.
@@ -131,4 +131,4 @@ function runMetricExtractor(metric) {
131
131
  //#endregion
132
132
  export { hasCapturedToolArgs as a, llmSpans as c, runsForScenario as d, toolSpans as f, groupBy as i, runFailureClass as l, aggregateLlm as n, isRunMetric as o, argHash as r, judgeSpans as s, RUN_METRICS as t, runMetricExtractor as u };
133
133
 
134
- //# sourceMappingURL=query-BPGMVlbM.js.map
134
+ //# sourceMappingURL=query-D1nLIKt7.js.map