@tangle-network/agent-eval 0.161.1 → 0.170.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (307) hide show
  1. package/CHANGELOG.md +156 -0
  2. package/README.md +2 -0
  3. package/dist/{active-curriculum-CD5TU2yW.js → active-curriculum-OjIWgrUJ.js} +2 -2
  4. package/dist/{active-curriculum-CD5TU2yW.js.map → active-curriculum-OjIWgrUJ.js.map} +1 -1
  5. package/dist/adapters/http.d.ts +108 -0
  6. package/dist/adapters/http.d.ts.map +1 -0
  7. package/dist/adapters/http.js +208 -0
  8. package/dist/adapters/http.js.map +1 -0
  9. package/dist/analyst/index.d.ts +40 -70
  10. package/dist/analyst/index.d.ts.map +1 -1
  11. package/dist/analyst/index.js +18 -311
  12. package/dist/analyst/index.js.map +1 -1
  13. package/dist/{backend-integrity-DxuQCu_A.d.ts → backend-integrity-e79K3UPD.d.ts} +3 -3
  14. package/dist/{backend-integrity-DxuQCu_A.d.ts.map → backend-integrity-e79K3UPD.d.ts.map} +1 -1
  15. package/dist/{baseline-BhPRQBVn.js → baseline-BC-eBZ7U.js} +2 -2
  16. package/dist/{baseline-BhPRQBVn.js.map → baseline-BC-eBZ7U.js.map} +1 -1
  17. package/dist/{benchmark-BhT16ep9.js → benchmark-C4wk_Sjr.js} +10 -3
  18. package/dist/benchmark-C4wk_Sjr.js.map +1 -0
  19. package/dist/{benchmark-command-BDC3Gocz.js → benchmark-command-BA7qOdWw.js} +239 -253
  20. package/dist/benchmark-command-BA7qOdWw.js.map +1 -0
  21. package/dist/{benchmark-CGPp-kDC.d.ts → benchmark-h-h4bfqj.d.ts} +3 -3
  22. package/dist/{benchmark-CGPp-kDC.d.ts.map → benchmark-h-h4bfqj.d.ts.map} +1 -1
  23. package/dist/benchmarks/index.d.ts +5 -5
  24. package/dist/benchmarks/index.js +3 -3
  25. package/dist/builder-eval/index.d.ts +3 -3
  26. package/dist/builder-eval/index.d.ts.map +1 -1
  27. package/dist/builder-eval/index.js +22 -8
  28. package/dist/builder-eval/index.js.map +1 -1
  29. package/dist/campaign/index.d.ts +8 -8
  30. package/dist/campaign/index.js +9 -9
  31. package/dist/{campaign-BSmOwskD.js → campaign-BeCbxFqs.js} +19 -18
  32. package/dist/campaign-BeCbxFqs.js.map +1 -0
  33. package/dist/{canonical-IL-Bu-14.js → canonical-DPyQ_rpt.js} +22 -2
  34. package/dist/{canonical-IL-Bu-14.js.map → canonical-DPyQ_rpt.js.map} +1 -1
  35. package/dist/{chat-client-DlMlAeYI.js → chat-client-DEtybj5i.js} +5 -5
  36. package/dist/{chat-client-DlMlAeYI.js.map → chat-client-DEtybj5i.js.map} +1 -1
  37. package/dist/{chat-json-call-6g5sJobJ.js → chat-json-call-5Jxna-aV.js} +2 -2
  38. package/dist/{chat-json-call-6g5sJobJ.js.map → chat-json-call-5Jxna-aV.js.map} +1 -1
  39. package/dist/cli.js +3 -3
  40. package/dist/{client-CX7KqIdB.js → client-BvwNkIRN.js} +2 -2
  41. package/dist/{client-CX7KqIdB.js.map → client-BvwNkIRN.js.map} +1 -1
  42. package/dist/{client-L9VVPkim.d.ts → client-_Fsa5c2_.d.ts} +4 -4
  43. package/dist/{client-L9VVPkim.d.ts.map → client-_Fsa5c2_.d.ts.map} +1 -1
  44. package/dist/contract/index.d.ts +12 -703
  45. package/dist/contract/index.js +14 -14
  46. package/dist/{counterfactual-BaFUWK3H.d.ts → counterfactual-Bee5_BIn.d.ts} +4 -4
  47. package/dist/{counterfactual-BaFUWK3H.d.ts.map → counterfactual-Bee5_BIn.d.ts.map} +1 -1
  48. package/dist/{counterfactual-D_VWavVm.js → counterfactual-Bjq1mlUu.js} +2 -2
  49. package/dist/{counterfactual-D_VWavVm.js.map → counterfactual-Bjq1mlUu.js.map} +1 -1
  50. package/dist/{default-registry-G9CKMNkc.d.ts → default-registry-ovxrOP0_.d.ts} +6 -6
  51. package/dist/{default-registry-G9CKMNkc.d.ts.map → default-registry-ovxrOP0_.d.ts.map} +1 -1
  52. package/dist/{define-agent-eval-h-s-sI-v.js → define-agent-eval-Clj-8igZ.js} +27 -15
  53. package/dist/{define-agent-eval-h-s-sI-v.js.map → define-agent-eval-Clj-8igZ.js.map} +1 -1
  54. package/dist/{define-agent-eval-Dx1JnPEa.d.ts → define-agent-eval-DVJm8Xlh.d.ts} +7 -7
  55. package/dist/{define-agent-eval-Dx1JnPEa.d.ts.map → define-agent-eval-DVJm8Xlh.d.ts.map} +1 -1
  56. package/dist/{descriptive-jDOuI6mz.js → descriptive-1V17A-qa.js} +2 -2
  57. package/dist/{descriptive-jDOuI6mz.js.map → descriptive-1V17A-qa.js.map} +1 -1
  58. package/dist/{dspy-rlm-engine-DptEII26.js → dspy-rlm-engine-CS3qcCEk.js} +3 -10
  59. package/dist/dspy-rlm-engine-CS3qcCEk.js.map +1 -0
  60. package/dist/{emitter-D_jYSGRd.d.ts → emitter-Bvnu0VzL.d.ts} +3 -3
  61. package/dist/{emitter-D_jYSGRd.d.ts.map → emitter-Bvnu0VzL.d.ts.map} +1 -1
  62. package/dist/{emitter-BpYFQPj4.js → emitter-DeQHiDMm.js} +13 -6
  63. package/dist/emitter-DeQHiDMm.js.map +1 -0
  64. package/dist/{engine-Cu5qD5Fc.d.ts → engine-D12Rb6WB.d.ts} +7 -7
  65. package/dist/{engine-Cu5qD5Fc.d.ts.map → engine-D12Rb6WB.d.ts.map} +1 -1
  66. package/dist/{eval-campaign-BsXWL2-2.js → eval-campaign-JDTeE6Pl.js} +5 -5
  67. package/dist/{eval-campaign-BsXWL2-2.js.map → eval-campaign-JDTeE6Pl.js.map} +1 -1
  68. package/dist/{exact-types-qnexxJ1Z.d.ts → exact-types-BEecmnWm.d.ts} +2 -2
  69. package/dist/{exact-types-qnexxJ1Z.d.ts.map → exact-types-BEecmnWm.d.ts.map} +1 -1
  70. package/dist/experiment/index.d.ts +5 -5
  71. package/dist/experiment/index.js +11 -11
  72. package/dist/{experiment-tracker-Ym6rEQT1.js → experiment-tracker-BKEumQug.js} +2 -2
  73. package/dist/{experiment-tracker-Ym6rEQT1.js.map → experiment-tracker-BKEumQug.js.map} +1 -1
  74. package/dist/{experiment-tracker-DCO6Cz4s.d.ts → experiment-tracker-Dm8yQMqb.d.ts} +2 -2
  75. package/dist/{experiment-tracker-DCO6Cz4s.d.ts.map → experiment-tracker-Dm8yQMqb.d.ts.map} +1 -1
  76. package/dist/{external-optimizer-process-WosTBChy.js → external-optimizer-process-CQxylYeG.js} +4 -11
  77. package/dist/external-optimizer-process-CQxylYeG.js.map +1 -0
  78. package/dist/{external-optimizer-subprocess-BIWbHpgD.js → external-optimizer-subprocess-Cex8Da2i.js} +26 -12
  79. package/dist/external-optimizer-subprocess-Cex8Da2i.js.map +1 -0
  80. package/dist/{failure-cluster-CXL8NbEw.d.ts → failure-cluster-6YSvsKlp.d.ts} +3 -3
  81. package/dist/{failure-cluster-CXL8NbEw.d.ts.map → failure-cluster-6YSvsKlp.d.ts.map} +1 -1
  82. package/dist/{feedback-trajectory-B3ZHaHV_.d.ts → feedback-trajectory-DIqpCyF0.d.ts} +6 -6
  83. package/dist/{feedback-trajectory-B3ZHaHV_.d.ts.map → feedback-trajectory-DIqpCyF0.d.ts.map} +1 -1
  84. package/dist/fuzz.d.ts +2 -3
  85. package/dist/fuzz.d.ts.map +1 -1
  86. package/dist/fuzz.js +4 -10
  87. package/dist/fuzz.js.map +1 -1
  88. package/dist/{skillopt-optimization-method-x7TTF23P.d.ts → heldout-gate-Bn7_xWCv.d.ts} +111 -111
  89. package/dist/heldout-gate-Bn7_xWCv.d.ts.map +1 -0
  90. package/dist/hosted/index.d.ts +2 -2
  91. package/dist/hosted/index.js +1 -1
  92. package/dist/index-Bfs5aufo.d.ts +704 -0
  93. package/dist/index-Bfs5aufo.d.ts.map +1 -0
  94. package/dist/{index-CGtH1piv.d.ts → index-CM-SM00y.d.ts} +7 -38
  95. package/dist/index-CM-SM00y.d.ts.map +1 -0
  96. package/dist/{index-D-V8gCs_.d.ts → index-DBbivBNs.d.ts} +30 -22
  97. package/dist/index-DBbivBNs.d.ts.map +1 -0
  98. package/dist/{index-D-IiQIBB.d.ts → index-DMoxLG8P.d.ts} +3 -3
  99. package/dist/{index-D-IiQIBB.d.ts.map → index-DMoxLG8P.d.ts.map} +1 -1
  100. package/dist/{index-D_P7Ye43.d.ts → index-DNgf5gyG.d.ts} +2 -2
  101. package/dist/{index-D_P7Ye43.d.ts.map → index-DNgf5gyG.d.ts.map} +1 -1
  102. package/dist/index.d.ts +67 -37
  103. package/dist/index.d.ts.map +1 -1
  104. package/dist/index.js +47 -41
  105. package/dist/index.js.map +1 -1
  106. package/dist/{insight-report-DRe8LB6d.d.ts → insight-report-08F022xN.d.ts} +4 -4
  107. package/dist/{insight-report-DRe8LB6d.d.ts.map → insight-report-08F022xN.d.ts.map} +1 -1
  108. package/dist/{integrity-DUNX9Fao.d.ts → integrity-B_EDELom.d.ts} +2 -2
  109. package/dist/{integrity-DUNX9Fao.d.ts.map → integrity-B_EDELom.d.ts.map} +1 -1
  110. package/dist/{internal-BDHPCnjk.js → internal-BMFSR8Ns.js} +3 -19
  111. package/dist/internal-BMFSR8Ns.js.map +1 -0
  112. package/dist/{judge-calibration-zZjLz8hr.js → judge-calibration-BnpVKtnb.js} +3 -13
  113. package/dist/judge-calibration-BnpVKtnb.js.map +1 -0
  114. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -1
  115. package/dist/{kind-factory-DY8FdoXf.js → kind-factory-DMeEoMQZ.js} +3 -10
  116. package/dist/kind-factory-DMeEoMQZ.js.map +1 -0
  117. package/dist/ledger-core/index.js +2 -2
  118. package/dist/{ledger-core-BOzlRygb.js → ledger-core-PIfjCbKn.js} +2 -2
  119. package/dist/{ledger-core-BOzlRygb.js.map → ledger-core-PIfjCbKn.js.map} +1 -1
  120. package/dist/{llm-client-hgDieDNN.js → llm-client-BFMRpmqb.js} +8 -11
  121. package/dist/llm-client-BFMRpmqb.js.map +1 -0
  122. package/dist/{llm-judge-BhasIPFT.js → llm-judge-DbJdo8Nj.js} +101 -23
  123. package/dist/llm-judge-DbJdo8Nj.js.map +1 -0
  124. package/dist/matrix/index.d.ts +2 -2
  125. package/dist/{matrix-eXKRMHnL.d.ts → matrix-BpI5Trmo.d.ts} +3 -3
  126. package/dist/{matrix-eXKRMHnL.d.ts.map → matrix-BpI5Trmo.d.ts.map} +1 -1
  127. package/dist/meta-eval/index.d.ts +5 -4
  128. package/dist/meta-eval/index.d.ts.map +1 -1
  129. package/dist/meta-eval/index.js +14 -22
  130. package/dist/meta-eval/index.js.map +1 -1
  131. package/dist/{mint-DfODW1KW.js → mint-DjfDUMHr.js} +2 -2
  132. package/dist/{mint-DfODW1KW.js.map → mint-DjfDUMHr.js.map} +1 -1
  133. package/dist/multishot/golden/index.d.ts +1 -1
  134. package/dist/multishot/index.d.ts +2 -2
  135. package/dist/openapi.json +4 -4
  136. package/dist/{paired-arms-D-XRF_fy.js → paired-arms-D4aeIHUy.js} +3 -3
  137. package/dist/{paired-arms-D-XRF_fy.js.map → paired-arms-D4aeIHUy.js.map} +1 -1
  138. package/dist/{paired-tests-BHIhYVdu.js → paired-tests-C8iCsioC.js} +3 -3
  139. package/dist/{paired-tests-BHIhYVdu.js.map → paired-tests-C8iCsioC.js.map} +1 -1
  140. package/dist/pipelines/index.d.ts +7 -6
  141. package/dist/pipelines/index.d.ts.map +1 -1
  142. package/dist/pipelines/index.js +5 -20
  143. package/dist/pipelines/index.js.map +1 -1
  144. package/dist/{power-and-mde-CHIrXJll.js → power-and-mde-B8F2RdcD.js} +3 -3
  145. package/dist/{power-and-mde-CHIrXJll.js.map → power-and-mde-B8F2RdcD.js.map} +1 -1
  146. package/dist/{power-preflight-DEw-uC7q.js → power-preflight-CFXm0Vjo.js} +3 -3
  147. package/dist/{power-preflight-DEw-uC7q.js.map → power-preflight-CFXm0Vjo.js.map} +1 -1
  148. package/dist/{pareto-BqNW3LJR.d.ts → power-preflight-Ptse_Kq7.d.ts} +43 -43
  149. package/dist/power-preflight-Ptse_Kq7.d.ts.map +1 -0
  150. package/dist/{pre-registration-KN9jkh58.js → pre-registration-D94b7Of5.js} +2 -2
  151. package/dist/{pre-registration-KN9jkh58.js.map → pre-registration-D94b7Of5.js.map} +1 -1
  152. package/dist/{pre-registration-CzFCcwYk.d.ts → pre-registration-DHz6P_6f.d.ts} +2 -2
  153. package/dist/{pre-registration-CzFCcwYk.d.ts.map → pre-registration-DHz6P_6f.d.ts.map} +1 -1
  154. package/dist/{produced-state-DZ89riy5.js → produced-state-CtSIp5cQ.js} +5 -5
  155. package/dist/{produced-state-DZ89riy5.js.map → produced-state-CtSIp5cQ.js.map} +1 -1
  156. package/dist/profile-cell.js +1 -1
  157. package/dist/{promotion-policy-DtnOIZvk.d.ts → promotion-policy-CkXSgKkF.d.ts} +3 -3
  158. package/dist/{promotion-policy-DtnOIZvk.d.ts.map → promotion-policy-CkXSgKkF.d.ts.map} +1 -1
  159. package/dist/{promotion-policy-xzA40Evo.js → promotion-policy-LY9mVQ7W.js} +3 -3
  160. package/dist/{promotion-policy-xzA40Evo.js.map → promotion-policy-LY9mVQ7W.js.map} +1 -1
  161. package/dist/{run-score-lDzV0X8j.js → proposal-findings-bko3GGy-.js} +2 -31
  162. package/dist/proposal-findings-bko3GGy-.js.map +1 -0
  163. package/dist/{transient-failure-DKF5Mofa.d.ts → provenance-CIRUardl.d.ts} +891 -891
  164. package/dist/provenance-CIRUardl.d.ts.map +1 -0
  165. package/dist/{query-CHmMP42p.js → query-BPGMVlbM.js} +46 -4
  166. package/dist/query-BPGMVlbM.js.map +1 -0
  167. package/dist/{query-DxPYqpmT.d.ts → query-Na5gEIGd.d.ts} +21 -4
  168. package/dist/query-Na5gEIGd.d.ts.map +1 -0
  169. package/dist/random-Dn5fPWkt.js +21 -0
  170. package/dist/random-Dn5fPWkt.js.map +1 -0
  171. package/dist/record-id-DUgsK5qp.js +17 -0
  172. package/dist/record-id-DUgsK5qp.js.map +1 -0
  173. package/dist/{registry-8You7OK1.d.ts → registry-xEb_xfns.d.ts} +3 -3
  174. package/dist/{registry-8You7OK1.d.ts.map → registry-xEb_xfns.d.ts.map} +1 -1
  175. package/dist/{release-confidence-DKfD2RYU.js → release-confidence-CzUHc4z4.js} +4 -4
  176. package/dist/{release-confidence-DKfD2RYU.js.map → release-confidence-CzUHc4z4.js.map} +1 -1
  177. package/dist/{release-confidence-Dqt0NFep.d.ts → release-confidence-D6lQw_o7.d.ts} +4 -4
  178. package/dist/{release-confidence-Dqt0NFep.d.ts.map → release-confidence-D6lQw_o7.d.ts.map} +1 -1
  179. package/dist/reporting.d.ts +3 -3
  180. package/dist/reporting.js +6 -6
  181. package/dist/{researcher-Cz565b7D.d.ts → researcher-CMUTQXD7.d.ts} +6 -6
  182. package/dist/{researcher-Cz565b7D.d.ts.map → researcher-CMUTQXD7.d.ts.map} +1 -1
  183. package/dist/{reward-hacking-MBf7qpSB.d.ts → reward-hacking-CgPRUesA.d.ts} +2 -2
  184. package/dist/{reward-hacking-MBf7qpSB.d.ts.map → reward-hacking-CgPRUesA.d.ts.map} +1 -1
  185. package/dist/{reward-hacking-t4lB1yt8.js → reward-hacking-SkxYgT0x.js} +3 -3
  186. package/dist/{reward-hacking-t4lB1yt8.js.map → reward-hacking-SkxYgT0x.js.map} +1 -1
  187. package/dist/rl.d.ts +8 -8
  188. package/dist/rl.d.ts.map +1 -1
  189. package/dist/rl.js +13 -18
  190. package/dist/rl.js.map +1 -1
  191. package/dist/rollout/index.d.ts +1 -1
  192. package/dist/rollout/index.js +2 -2
  193. package/dist/{rollout-Dm2tSdiQ.js → rollout-Crypdx8s.js} +2 -2
  194. package/dist/{rollout-Dm2tSdiQ.js.map → rollout-Crypdx8s.js.map} +1 -1
  195. package/dist/{rubric-predictive-validity-CK8SCOg-.js → rubric-predictive-validity-2D5Gw9z9.js} +3 -3
  196. package/dist/{rubric-predictive-validity-CK8SCOg-.js.map → rubric-predictive-validity-2D5Gw9z9.js.map} +1 -1
  197. package/dist/{rubric-predictive-validity-CxycqzX5.d.ts → rubric-predictive-validity-DluJLCKQ.d.ts} +2 -2
  198. package/dist/{rubric-predictive-validity-CxycqzX5.d.ts.map → rubric-predictive-validity-DluJLCKQ.d.ts.map} +1 -1
  199. package/dist/{run-record-BC0ebuRP.js → run-record-DLORoL7t.js} +2 -2
  200. package/dist/{run-record-BC0ebuRP.js.map → run-record-DLORoL7t.js.map} +1 -1
  201. package/dist/{run-record-VVy4T9OW.d.ts → run-record-DQjRcYwA.d.ts} +3 -3
  202. package/dist/{run-record-VVy4T9OW.d.ts.map → run-record-DQjRcYwA.d.ts.map} +1 -1
  203. package/dist/{schema-k6ZBftVv.js → schema-CdIX2aHu.js} +5 -1
  204. package/dist/{schema-k6ZBftVv.js.map → schema-CdIX2aHu.js.map} +1 -1
  205. package/dist/{schema-Bjgdsn73.d.ts → schema-DID1Cqct.d.ts} +7 -3
  206. package/dist/{schema-Bjgdsn73.d.ts.map → schema-DID1Cqct.d.ts.map} +1 -1
  207. package/dist/{semantic-concept-judge-BSkKKHeq.js → semantic-concept-judge-I36eejJx.js} +3 -3
  208. package/dist/{semantic-concept-judge-BSkKKHeq.js.map → semantic-concept-judge-I36eejJx.js.map} +1 -1
  209. package/dist/{sequential-rYW-Ophm.js → sequential-B51qAYE4.js} +4 -4
  210. package/dist/{sequential-rYW-Ophm.js.map → sequential-B51qAYE4.js.map} +1 -1
  211. package/dist/{server-BtFd4uzB.js → server-CCEnywOR.js} +27 -19
  212. package/dist/server-CCEnywOR.js.map +1 -0
  213. package/dist/{skillopt-optimization-method-DbaekMcn.js → skillopt-optimization-method-B2R9C5aG.js} +11 -11
  214. package/dist/{skillopt-optimization-method-DbaekMcn.js.map → skillopt-optimization-method-B2R9C5aG.js.map} +1 -1
  215. package/dist/{statistical-heldout-Cy3EhjlC.d.ts → statistical-heldout-DFS7QGpS.d.ts} +3 -3
  216. package/dist/{statistical-heldout-Cy3EhjlC.d.ts.map → statistical-heldout-DFS7QGpS.d.ts.map} +1 -1
  217. package/dist/{store-B06JdC56.d.ts → store-Cq9oOrI1.d.ts} +2 -2
  218. package/dist/{store-B06JdC56.d.ts.map → store-Cq9oOrI1.d.ts.map} +1 -1
  219. package/dist/{store-otlp-C_Rq5I4D.js → store-otlp-CHjBvWQY.js} +2 -2
  220. package/dist/{store-otlp-C_Rq5I4D.js.map → store-otlp-CHjBvWQY.js.map} +1 -1
  221. package/dist/{store-tool-spans-DPUG7UUY.d.ts → store-tool-spans-B2DJ_82T.d.ts} +102 -36
  222. package/dist/store-tool-spans-B2DJ_82T.d.ts.map +1 -0
  223. package/dist/{store-tool-spans-Dlh9vkFK.js → store-tool-spans-B9o6tU8f.js} +23 -19
  224. package/dist/store-tool-spans-B9o6tU8f.js.map +1 -0
  225. package/dist/storyboard/index.d.ts +1 -1
  226. package/dist/{student-t-CvBq2mve.js → student-t-BA-Uy51p.js} +2 -2
  227. package/dist/{student-t-CvBq2mve.js.map → student-t-BA-Uy51p.js.map} +1 -1
  228. package/dist/{summary-report-BI5hUtvK.js → summary-report-Bgh8CpNK.js} +7 -7
  229. package/dist/{summary-report-BI5hUtvK.js.map → summary-report-Bgh8CpNK.js.map} +1 -1
  230. package/dist/{summary-report-CC07PhEL.d.ts → summary-report-DRstQNBX.d.ts} +3 -3
  231. package/dist/{summary-report-CC07PhEL.d.ts.map → summary-report-DRstQNBX.d.ts.map} +1 -1
  232. package/dist/supervisor-run/index.js +1 -1
  233. package/dist/{task-failure-attributes-DTl-7-Kw.js → task-failure-attributes-CBGtLS_H.js} +3 -3
  234. package/dist/{task-failure-attributes-DTl-7-Kw.js.map → task-failure-attributes-CBGtLS_H.js.map} +1 -1
  235. package/dist/{tool-groups-Ci8i9ErB.d.ts → tool-groups-BnXlCJZQ.d.ts} +3 -3
  236. package/dist/tool-groups-BnXlCJZQ.d.ts.map +1 -0
  237. package/dist/{tool-waste-Dro0gJi3.d.ts → tool-waste-BrmLKxMw.d.ts} +4 -4
  238. package/dist/{tool-waste-Dro0gJi3.d.ts.map → tool-waste-BrmLKxMw.d.ts.map} +1 -1
  239. package/dist/{tool-waste-BqzmVdJk.js → tool-waste-CwGHzBzX.js} +4 -4
  240. package/dist/{tool-waste-BqzmVdJk.js.map → tool-waste-CwGHzBzX.js.map} +1 -1
  241. package/dist/trace-repair/index.d.ts +3 -3
  242. package/dist/trace-repair/index.d.ts.map +1 -1
  243. package/dist/trace-repair/index.js +5 -4
  244. package/dist/trace-repair/index.js.map +1 -1
  245. package/dist/traces.d.ts +60 -62
  246. package/dist/traces.d.ts.map +1 -1
  247. package/dist/traces.js +13 -22
  248. package/dist/traces.js.map +1 -1
  249. package/dist/{trajectory-Bi157Gun.d.ts → trajectory-r1bQqvBQ.d.ts} +3 -3
  250. package/dist/{trajectory-Bi157Gun.d.ts.map → trajectory-r1bQqvBQ.d.ts.map} +1 -1
  251. package/dist/trajectory-replay/index.d.ts +3 -3
  252. package/dist/trajectory-replay/index.d.ts.map +1 -1
  253. package/dist/trajectory-replay/index.js +4 -14
  254. package/dist/trajectory-replay/index.js.map +1 -1
  255. package/dist/types-Bfk0uxRj.d.ts.map +1 -1
  256. package/dist/{types-BI4fT3HN.js → types-CiWITkGo.js} +11 -2
  257. package/dist/types-CiWITkGo.js.map +1 -0
  258. package/dist/{types-D9ssmxKL.d.ts → types-DMoNFDWi.d.ts} +6 -3
  259. package/dist/{types-D9ssmxKL.d.ts.map → types-DMoNFDWi.d.ts.map} +1 -1
  260. package/dist/{types-D4s7Z6nq.d.ts → types-Dy237wiH.d.ts} +3 -3
  261. package/dist/{types-D4s7Z6nq.d.ts.map → types-Dy237wiH.d.ts.map} +1 -1
  262. package/dist/{types-BPb2Kf_C2.d.ts → types-i21ccEkr.d.ts} +3 -3
  263. package/dist/types-i21ccEkr.d.ts.map +1 -0
  264. package/dist/{verdict-B0xltqu6.js → verdict-BQ3pCFf8.js} +2 -2
  265. package/dist/{verdict-B0xltqu6.js.map → verdict-BQ3pCFf8.js.map} +1 -1
  266. package/dist/{verdict-cache-CdVVTVmn.js → verdict-cache-B3eCVQtY.js} +2 -2
  267. package/dist/{verdict-cache-CdVVTVmn.js.map → verdict-cache-B3eCVQtY.js.map} +1 -1
  268. package/dist/wire/index.d.ts +23 -8
  269. package/dist/wire/index.d.ts.map +1 -1
  270. package/dist/wire/index.js +2 -2
  271. package/docs/code-agent-intake.md +64 -0
  272. package/docs/concepts.md +1 -1
  273. package/docs/design/statistics-decisions.md +1 -1
  274. package/docs/distributed-driver.md +3 -6
  275. package/docs/public-api.md +122 -106
  276. package/docs/wire-protocol.md +5 -3
  277. package/package.json +9 -2
  278. package/dist/benchmark-BhT16ep9.js.map +0 -1
  279. package/dist/benchmark-command-BDC3Gocz.js.map +0 -1
  280. package/dist/campaign-BSmOwskD.js.map +0 -1
  281. package/dist/capture-fetch-CqwsJkkG.d.ts +0 -68
  282. package/dist/capture-fetch-CqwsJkkG.d.ts.map +0 -1
  283. package/dist/contract/index.d.ts.map +0 -1
  284. package/dist/dspy-rlm-engine-DptEII26.js.map +0 -1
  285. package/dist/emitter-BpYFQPj4.js.map +0 -1
  286. package/dist/external-optimizer-process-WosTBChy.js.map +0 -1
  287. package/dist/external-optimizer-subprocess-BIWbHpgD.js.map +0 -1
  288. package/dist/index-CGtH1piv.d.ts.map +0 -1
  289. package/dist/index-D-V8gCs_.d.ts.map +0 -1
  290. package/dist/index-vrJugRal.d.ts +0 -1
  291. package/dist/internal-BDHPCnjk.js.map +0 -1
  292. package/dist/judge-calibration-zZjLz8hr.js.map +0 -1
  293. package/dist/kind-factory-DY8FdoXf.js.map +0 -1
  294. package/dist/llm-client-hgDieDNN.js.map +0 -1
  295. package/dist/llm-judge-BhasIPFT.js.map +0 -1
  296. package/dist/pareto-BqNW3LJR.d.ts.map +0 -1
  297. package/dist/query-CHmMP42p.js.map +0 -1
  298. package/dist/query-DxPYqpmT.d.ts.map +0 -1
  299. package/dist/run-score-lDzV0X8j.js.map +0 -1
  300. package/dist/server-BtFd4uzB.js.map +0 -1
  301. package/dist/skillopt-optimization-method-x7TTF23P.d.ts.map +0 -1
  302. package/dist/store-tool-spans-DPUG7UUY.d.ts.map +0 -1
  303. package/dist/store-tool-spans-Dlh9vkFK.js.map +0 -1
  304. package/dist/tool-groups-Ci8i9ErB.d.ts.map +0 -1
  305. package/dist/transient-failure-DKF5Mofa.d.ts.map +0 -1
  306. package/dist/types-BI4fT3HN.js.map +0 -1
  307. package/dist/types-BPb2Kf_C2.d.ts.map +0 -1
@@ -1 +1 @@
1
- {"version":3,"file":"run-record-BC0ebuRP.js","names":[],"sources":["../src/run-record.ts"],"sourcesContent":["/**\n * Paper-grade RunRecord schema + runtime validator.\n *\n * Every run that participates in a promotion gate, paper table, or\n * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory\n * fields are exactly those the paper \"Two Loops, Three Roles\" requires\n * for reproducibility: who/what/when/cost/seed/hash, plus the search vs\n * holdout split tag. A task score is optional because execution-only records\n * must preserve missing labels instead of converting errors into zero quality.\n *\n * This is intentionally NOT a replacement for the rich `Run` /\n * `ProposeReviewReport` / `ScenarioResult` types already in the\n * package. Those are runtime structures with full provenance. A\n * `RunRecord` is the analysis-time projection — the JSON-friendly\n * row you'd put in a parquet file or paste into a notebook.\n *\n * Validate at the boundary:\n *\n * const rec = validateRunRecord(rawJson) // throws on missing\n * const ok = isRunRecord(rawJson) // boolean check\n * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }\n *\n * The validator runs in pure TS — zod is intentionally NOT a\n * dependency. Round-trip tested in `tests/run-record.test.ts`.\n */\n\nimport type { AgentProfileCell } from './agent-profile-cell'\nimport { validateAgentProfileCell } from './agent-profile-cell'\nimport type { CostProvenance } from './cost-ledger'\nimport { ValidationError } from './errors'\n// Value import of a leaf module that itself imports only this file's TYPES —\n// no runtime cycle. It keeps the raw split-score derivation spelled in exactly\n// one place (see `rollout/score-derivation-guard`).\nimport { observedScore } from './rollout/reward'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the\n * combined train+test pool that the optimizer is allowed to read. */\nexport type RunSplitTag = 'search' | 'dev' | 'holdout'\n\n/**\n * Explicit execution-lifecycle result for a run.\n *\n * This is separate from task quality (`outcome`) and failure classification.\n * Producers set it only from root-run or process evidence.\n */\nexport type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown'\n\n/** Explicit model value for a row that never produced a served model snapshot. */\nexport const UNKNOWN_MODEL = 'unknown'\n\nexport interface RunTokenUsage {\n input: number\n /** All generated tokens charged as output, including reasoning tokens. */\n output: number\n /** Present only when one or more paid calls did not report token usage.\n * In that case, every numeric field is a known subtotal, not a measured total. */\n tokensKnown?: false\n /** Reasoning-token subset of `output`, when the provider reports it. */\n reasoning?: number\n /** Prompt tokens served from a provider cache. */\n cached?: number\n /** Prompt tokens written into a provider cache. */\n cacheWrite?: number\n}\n\n/** How a run's USD amount was obtained. */\nexport type RunCostProvenance = CostProvenance\n\nexport interface RunJudgeMetadata {\n model: string\n promptVersion: string\n /** [0,1] confidence the judge declared. Constant judge confidence\n * across many runs is a fallback signal (see `canary.ts`). */\n confidence: number\n /** True if the judge degraded to a fallback path (rules-only,\n * prior-call cache, etc.). The canary uses this to alert. */\n fallback: boolean\n}\n\n/**\n * Per-judge / per-dimension breakdown for runs scored by an ensemble of\n * judges over a multi-dimensional rubric.\n *\n * The collapsed `outcome.searchScore` / `holdoutScore` carries the\n * composite the gate uses. The full breakdown belongs here so consumers\n * can answer \"which judge disagreed?\", \"which dimension dragged the\n * composite down?\", and \"did half the panel fail?\" without re-running.\n *\n * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and\n * `composite` are convenience projections — derivable but precomputed so\n * downstream IRR primitives (`interRaterReliability`,\n * `corpusInterRaterAgreement`) and reporters don't pay the same\n * aggregation twice.\n *\n * Fail-loud discipline: judges that errored out land in `failedJudges`\n * by id. A missing key in `perJudge` is ambiguous (silent zero vs not\n * run); the explicit list makes a partial-failure recorded as such.\n */\nexport interface JudgeScoresRecord {\n /** Per-judge per-dimension scores. `{ \"kimi-k2.6\": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */\n perJudge: Record<string, Record<string, number>>\n /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */\n perDimMean: Record<string, number>\n /** Composite mean across successful judges. Mirrors the task score only\n * when `failedJudges` is empty. */\n composite: number\n /** Judges that errored or returned an unparseable verdict. Recorded\n * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,\n * not inferred from missing keys in `perJudge`. */\n failedJudges?: string[]\n /** Free-form notes the judges emitted (joined across judges or\n * first-judge only — consumer's choice). */\n notes?: string\n}\n\nexport interface RunOutcome {\n /** Score on the search/optimization split. Optional for holdout-only and\n * execution-only records. */\n searchScore?: number\n /** Score on the held-out split. Optional for search-only and execution-only\n * records. When both scores are absent, the run is explicitly unlabeled. */\n holdoutScore?: number\n /** Bag of any other metric the run produced — judge dimensions,\n * pass/fail counters, latency stats, etc. Numeric only — keeps\n * reporters honest. */\n raw: Record<string, number>\n /** Per-judge / per-dim breakdown. Consumers writing ensemble\n * judgements populate this; substrate primitives like\n * `interRaterReliability` and `corpusInterRaterAgreement` accept\n * these records as input. Optional — single-judge or scalar-only\n * runs leave it unset. */\n judgeScores?: JudgeScoresRecord\n /** Authenticity / realness verdict — did the run build the REAL thing on the\n * intended infra, or fake it (see `./authenticity`)? Optional: only domains\n * with an authenticity config populate it. Carried in the corpus so the\n * flywheel / off-policy learning can optimize for real completion, not gamed\n * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run\n * must not count as a real success regardless of `score`. */\n realness?: { score: number; gated: boolean; reason?: string }\n}\n\n/**\n * Mandatory paper-grade fields for a single evaluation run. Optional\n * fields are extension points; mandatory fields throw if missing.\n *\n * Hash discipline:\n * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the\n * model (after any steering bundle merge).\n * - `configHash` is the sha256 of the effective run config (model,\n * temperature, tools, judges, splits). The pair (promptHash,\n * configHash) uniquely identifies an experiment cell.\n *\n * Model snapshot discipline:\n * - successful rows MUST encode a snapshot version. Bare aliases like\n * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.\n * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.\n * - a failed, cancelled, incomplete, or otherwise unknown row may use\n * `UNKNOWN_MODEL` when no served model was observed. This is an explicit\n * absence marker, not a fabricated snapshot.\n */\nexport interface RunRecord {\n /** UUID for the run. */\n runId: string\n /** Logical experiment grouping (a treatment vs a baseline within\n * the same sweep should share `experimentId`). */\n experimentId: string\n /** Stable identifier for the candidate (variant) being run. The\n * promotion gate compares two `candidateId`s on matched items. */\n candidateId: string\n /** RNG seed for the run. Always recorded — silent re-seeding is\n * the most common cause of non-reproducible numbers. */\n seed: number\n /** Model identifier WITH snapshot version. */\n model: string\n /** sha256 of the effective prompt (post-steering). */\n promptHash: string\n /** sha256 of the effective config. */\n configHash: string\n /** Git SHA the harness was run from. */\n commitSha: string\n /** End-to-end wall-clock duration in milliseconds. */\n wallMs: number\n /** Time spent queued before execution started, if known. */\n queueMs?: number\n /** Total USD cost, or null when the producer could not capture one. */\n costUsd: number | null\n /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */\n costProvenance: RunCostProvenance\n /** Token usage breakdown. */\n tokenUsage: RunTokenUsage\n /** Root-run or process terminal result. Never inferred from a child span. */\n terminalOutcome: RunTerminalOutcome\n /** Root-run or process failure reason. Valid only for a failed, cancelled,\n * or incomplete terminal result; never populated from a child span. */\n terminalFailureReason?: string\n /** Judge-side metadata, if a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /** Per-split scores + raw bag. */\n outcome: RunOutcome\n /** Canonical task-failure class drawn from the shared\n * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result\n * evidence. Execution errors belong in\n * `outcome.raw.execution_error_count`. */\n failureClass?: FailureClass\n /** Free-form task-failure detail scoped under a non-success\n * `failureClass`. It is invalid without that class. */\n failureMode?: string\n /** Which split this run was drawn from. */\n splitTag: RunSplitTag\n /**\n * Stable scenario identifier the run observed or was scored against.\n * Comparison primitives match this identity rather than input order.\n */\n scenarioId: string\n /**\n * Canonical identity for the agent profile cell that produced this row:\n * profile artifact hash plus optional harness/model/prompt/reporting\n * dimensions. Use `agentProfile.cellId` to group persona sweeps and\n * longitudinal reports by the complete source profile, not by a loose\n * candidate label or opaque config hash.\n */\n agentProfile?: AgentProfileCell\n}\n\n/**\n * Canonical task-result classification.\n *\n * A producer may omit classification, record explicit success, or attach\n * domain-specific detail to a non-success class. Detail can never stand alone.\n * Execution errors belong in `outcome.raw.execution_error_count`.\n */\nexport type RunTaskFailure =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | {\n failureClass: Exclude<FailureClass, 'success'>\n failureMode?: string\n }\n\n/**\n * Return task quality, preferring held-out evidence when both scores exist.\n *\n * RAW: no realness protection is applied. Built on `observedScore` rather\n * than repeating the split derivation, so only `rollout/reward.ts` reads the\n * raw fields. Anything that becomes training data must use `trainingScore` or\n * `trainingReward` instead.\n */\nexport function runTaskScore(record: RunRecord): number | undefined {\n const score = observedScore(record)\n return typeof score === 'number' && Number.isFinite(score) ? score : undefined\n}\n\n// ── Validation ───────────────────────────────────────────────────────\n\nconst MANDATORY_TOP_LEVEL = [\n 'runId',\n 'experimentId',\n 'candidateId',\n 'seed',\n 'model',\n 'promptHash',\n 'configHash',\n 'commitSha',\n 'wallMs',\n 'costUsd',\n 'costProvenance',\n 'tokenUsage',\n 'terminalOutcome',\n 'outcome',\n 'splitTag',\n 'scenarioId',\n] as const\n\nconst SPLIT_TAGS: ReadonlyArray<RunSplitTag> = ['search', 'dev', 'holdout']\nconst TERMINAL_OUTCOMES: ReadonlyArray<RunTerminalOutcome> = [\n 'succeeded',\n 'failed',\n 'cancelled',\n 'incomplete',\n 'unknown',\n]\n\nexport class RunRecordValidationError extends ValidationError {\n readonly path: string\n constructor(message: string, path = '') {\n super(path ? `${message} (at ${path})` : message)\n this.path = path\n }\n}\n\n/**\n * Strict validator. Throws `RunRecordValidationError` on the first\n * missing or wrongly-typed field. Returns the input cast to\n * `RunRecord` on success — the validator does not coerce.\n */\nexport function validateRunRecord(input: unknown): RunRecord {\n if (input === null || typeof input !== 'object') {\n throw new RunRecordValidationError('expected object')\n }\n const obj = input as Record<string, unknown>\n\n for (const key of MANDATORY_TOP_LEVEL) {\n if (!(key in obj)) {\n throw new RunRecordValidationError(`missing mandatory field \"${key}\"`)\n }\n }\n\n expectString(obj.runId, 'runId')\n expectString(obj.experimentId, 'experimentId')\n expectString(obj.candidateId, 'candidateId')\n expectFiniteNumber(obj.seed, 'seed')\n expectString(obj.model, 'model')\n expectString(obj.promptHash, 'promptHash')\n expectString(obj.configHash, 'configHash')\n expectString(obj.commitSha, 'commitSha')\n expectNonNegativeNumber(obj.wallMs, 'wallMs')\n if (obj.queueMs !== undefined) expectNonNegativeNumber(obj.queueMs, 'queueMs')\n validateCost(obj.costUsd, obj.costProvenance)\n\n // Snapshot discipline: successful rows require a served model snapshot.\n // Non-success rows may carry the explicit absence marker when execution\n // stopped before a model identity was observed.\n if (\n !modelHasSnapshot(obj.model as string) &&\n !(obj.model === UNKNOWN_MODEL && obj.terminalOutcome !== 'succeeded')\n ) {\n throw new RunRecordValidationError(\n `model \"${obj.model}\" lacks a snapshot version (use 'name@YYYY-MM-DD' or 'name-YYYYMMDD', or '${UNKNOWN_MODEL}' for a non-success row without a served model)`,\n 'model',\n )\n }\n\n // Token usage.\n const tu = obj.tokenUsage\n if (tu === null || typeof tu !== 'object') {\n throw new RunRecordValidationError('tokenUsage must be an object', 'tokenUsage')\n }\n const tuRec = tu as Record<string, unknown>\n expectNonNegativeNumber(tuRec.input, 'tokenUsage.input')\n expectNonNegativeNumber(tuRec.output, 'tokenUsage.output')\n if (tuRec.tokensKnown !== undefined && tuRec.tokensKnown !== false) {\n throw new RunRecordValidationError(\n 'tokensKnown must be false when present; omit it when token usage is complete',\n 'tokenUsage.tokensKnown',\n )\n }\n if (tuRec.reasoning !== undefined) {\n expectNonNegativeNumber(tuRec.reasoning, 'tokenUsage.reasoning')\n if ((tuRec.reasoning as number) > (tuRec.output as number)) {\n throw new RunRecordValidationError(\n 'reasoning tokens must be a subset of output tokens',\n 'tokenUsage.reasoning',\n )\n }\n }\n if (tuRec.cached !== undefined) expectNonNegativeNumber(tuRec.cached, 'tokenUsage.cached')\n if (tuRec.cacheWrite !== undefined) {\n expectNonNegativeNumber(tuRec.cacheWrite, 'tokenUsage.cacheWrite')\n }\n\n // Judge metadata, optional.\n if (obj.judgeMetadata !== undefined) {\n const jm = obj.judgeMetadata\n if (jm === null || typeof jm !== 'object') {\n throw new RunRecordValidationError('judgeMetadata must be an object', 'judgeMetadata')\n }\n const jmRec = jm as Record<string, unknown>\n expectString(jmRec.model, 'judgeMetadata.model')\n expectString(jmRec.promptVersion, 'judgeMetadata.promptVersion')\n expectFiniteNumber(jmRec.confidence, 'judgeMetadata.confidence')\n if (typeof jmRec.fallback !== 'boolean') {\n throw new RunRecordValidationError(\n 'judgeMetadata.fallback must be boolean',\n 'judgeMetadata.fallback',\n )\n }\n }\n\n // Outcome.\n const out = obj.outcome\n if (out === null || typeof out !== 'object') {\n throw new RunRecordValidationError('outcome must be an object', 'outcome')\n }\n const outRec = out as Record<string, unknown>\n if (outRec.searchScore !== undefined)\n expectFiniteNumber(outRec.searchScore, 'outcome.searchScore')\n if (outRec.holdoutScore !== undefined)\n expectFiniteNumber(outRec.holdoutScore, 'outcome.holdoutScore')\n const raw = outRec.raw\n if (raw === null || typeof raw !== 'object') {\n throw new RunRecordValidationError('outcome.raw must be an object', 'outcome.raw')\n }\n for (const [k, v] of Object.entries(raw as Record<string, unknown>)) {\n expectFiniteNumber(v, `outcome.raw.${k}`)\n }\n // Realness verdict, optional.\n if (outRec.realness !== undefined) {\n const r = outRec.realness\n if (r === null || typeof r !== 'object') {\n throw new RunRecordValidationError('outcome.realness must be an object', 'outcome.realness')\n }\n const rr = r as Record<string, unknown>\n expectFiniteNumber(rr.score, 'outcome.realness.score')\n if (typeof rr.gated !== 'boolean') {\n throw new RunRecordValidationError(\n 'outcome.realness.gated must be a boolean',\n 'outcome.realness.gated',\n )\n }\n }\n\n // Per-judge / per-dim breakdown, optional.\n if (outRec.judgeScores !== undefined) {\n validateJudgeScores(outRec.judgeScores, 'outcome.judgeScores')\n }\n\n // Failure mode optional.\n if (\n obj.failureClass !== undefined &&\n (typeof obj.failureClass !== 'string' ||\n !FAILURE_CLASSES.includes(obj.failureClass as FailureClass))\n ) {\n throw new RunRecordValidationError(\n `failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n 'failureClass',\n )\n }\n if (obj.failureMode !== undefined) {\n expectString(obj.failureMode, 'failureMode')\n if (obj.failureClass === undefined || obj.failureClass === 'success') {\n throw new RunRecordValidationError(\n 'failureMode requires a non-success failureClass',\n 'failureMode',\n )\n }\n }\n\n if (\n typeof obj.terminalOutcome !== 'string' ||\n !TERMINAL_OUTCOMES.includes(obj.terminalOutcome as RunTerminalOutcome)\n ) {\n throw new RunRecordValidationError(\n `terminalOutcome must be one of ${TERMINAL_OUTCOMES.join(', ')}`,\n 'terminalOutcome',\n )\n }\n if (obj.terminalFailureReason !== undefined) {\n expectString(obj.terminalFailureReason, 'terminalFailureReason')\n if (\n obj.terminalOutcome !== 'failed' &&\n obj.terminalOutcome !== 'cancelled' &&\n obj.terminalOutcome !== 'incomplete'\n ) {\n throw new RunRecordValidationError(\n 'terminalFailureReason requires terminalOutcome failed, cancelled, or incomplete',\n 'terminalFailureReason',\n )\n }\n }\n\n if (obj.agentProfile !== undefined) {\n try {\n const profile = validateAgentProfileCell(obj.agentProfile)\n if (profile.model !== undefined && profile.model !== obj.model) {\n throw new RunRecordValidationError(\n `agentProfile.model \"${profile.model}\" does not match model \"${obj.model}\"`,\n 'agentProfile.model',\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== obj.promptHash) {\n throw new RunRecordValidationError(\n `agentProfile.promptHash \"${profile.promptHash}\" does not match promptHash \"${obj.promptHash}\"`,\n 'agentProfile.promptHash',\n )\n }\n } catch (error) {\n if (error instanceof RunRecordValidationError) throw error\n if (error instanceof Error) {\n throw new RunRecordValidationError(error.message, 'agentProfile')\n }\n throw error\n }\n }\n\n expectString(obj.scenarioId, 'scenarioId')\n\n // Split tag.\n if (typeof obj.splitTag !== 'string' || !SPLIT_TAGS.includes(obj.splitTag as RunSplitTag)) {\n throw new RunRecordValidationError(\n `splitTag must be one of ${SPLIT_TAGS.join(', ')}, got ${String(obj.splitTag)}`,\n 'splitTag',\n )\n }\n\n return input as RunRecord\n}\n\nfunction validateCost(costUsd: unknown, provenance: unknown): void {\n if (provenance === null || typeof provenance !== 'object') {\n throw new RunRecordValidationError('costProvenance must be an object', 'costProvenance')\n }\n const value = provenance as Record<string, unknown>\n if (value.kind !== 'observed' && value.kind !== 'estimated' && value.kind !== 'uncaptured') {\n throw new RunRecordValidationError(\n 'costProvenance.kind must be observed, estimated, or uncaptured',\n 'costProvenance.kind',\n )\n }\n // A record must never read as a total it cannot support, so an uncaptured\n // cost carries no number at all. A matrix `CellResult` keeps its known\n // subtotal instead, because a cost ceiling must charge the part it can see;\n // converting one to the other drops that subtotal.\n if (value.kind === 'uncaptured') {\n if (value.usd !== null) {\n throw new RunRecordValidationError(\n 'uncaptured costProvenance.usd must be null',\n 'costProvenance.usd',\n )\n }\n if (costUsd !== null) {\n throw new RunRecordValidationError('uncaptured cost requires costUsd to be null', 'costUsd')\n }\n return\n }\n expectNonNegativeNumber(costUsd, 'costUsd')\n expectNonNegativeNumber(value.usd, 'costProvenance.usd')\n if (value.usd !== costUsd) {\n throw new RunRecordValidationError(\n 'costProvenance.usd must equal costUsd',\n 'costProvenance.usd',\n )\n }\n}\n\n/** Boolean validator — convenience for filtering arrays. */\nexport function isRunRecord(input: unknown): input is RunRecord {\n try {\n validateRunRecord(input)\n return true\n } catch {\n return false\n }\n}\n\n/** Non-throwing validator — returns a discriminated union. */\nexport function parseRunRecordSafe(\n input: unknown,\n): { ok: true; value: RunRecord } | { ok: false; error: RunRecordValidationError } {\n try {\n return { ok: true, value: validateRunRecord(input) }\n } catch (e) {\n if (e instanceof RunRecordValidationError) return { ok: false, error: e }\n throw e\n }\n}\n\n/** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */\nexport function roundTripRunRecord(record: RunRecord): RunRecord {\n const json = JSON.stringify(record)\n return validateRunRecord(JSON.parse(json))\n}\n\n// ── Internals ────────────────────────────────────────────────────────\n\nfunction expectString(value: unknown, path: string): void {\n if (typeof value !== 'string' || value.length === 0) {\n throw new RunRecordValidationError(`expected non-empty string`, path)\n }\n}\n\nfunction expectFiniteNumber(value: unknown, path: string): void {\n if (typeof value !== 'number' || !Number.isFinite(value)) {\n throw new RunRecordValidationError(`expected finite number`, path)\n }\n}\n\nfunction expectNonNegativeNumber(value: unknown, path: string): void {\n expectFiniteNumber(value, path)\n if ((value as number) < 0) {\n throw new RunRecordValidationError('expected non-negative number', path)\n }\n}\n\nfunction validateJudgeScores(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object') {\n throw new RunRecordValidationError('judgeScores must be an object', path)\n }\n const rec = value as Record<string, unknown>\n\n const perJudge = rec.perJudge\n if (perJudge === null || typeof perJudge !== 'object') {\n throw new RunRecordValidationError('perJudge must be an object', `${path}.perJudge`)\n }\n for (const [judgeId, dims] of Object.entries(perJudge as Record<string, unknown>)) {\n if (dims === null || typeof dims !== 'object') {\n throw new RunRecordValidationError(\n 'per-judge entry must be an object of dimension scores',\n `${path}.perJudge.${judgeId}`,\n )\n }\n for (const [dim, score] of Object.entries(dims as Record<string, unknown>)) {\n expectFiniteNumber(score, `${path}.perJudge.${judgeId}.${dim}`)\n }\n }\n\n const perDimMean = rec.perDimMean\n if (perDimMean === null || typeof perDimMean !== 'object') {\n throw new RunRecordValidationError('perDimMean must be an object', `${path}.perDimMean`)\n }\n for (const [dim, mean] of Object.entries(perDimMean as Record<string, unknown>)) {\n expectFiniteNumber(mean, `${path}.perDimMean.${dim}`)\n }\n\n expectFiniteNumber(rec.composite, `${path}.composite`)\n\n if (rec.failedJudges !== undefined) {\n if (!Array.isArray(rec.failedJudges)) {\n throw new RunRecordValidationError(\n 'failedJudges must be an array of strings',\n `${path}.failedJudges`,\n )\n }\n for (let i = 0; i < rec.failedJudges.length; i++) {\n const id = rec.failedJudges[i]\n if (typeof id !== 'string' || id.length === 0) {\n throw new RunRecordValidationError(\n 'failedJudges entry must be a non-empty string',\n `${path}.failedJudges[${i}]`,\n )\n }\n }\n }\n\n if (rec.notes !== undefined && typeof rec.notes !== 'string') {\n throw new RunRecordValidationError('notes must be a string', `${path}.notes`)\n }\n}\n\n/**\n * Snapshot check for provider model identifiers. Accepts ISO and compact\n * dates, Router's `-MMDD` snapshots, one opaque `@token`, and Vertex-style\n * `:date-token` suffixes. Routing selectors such as `@preset/name` are not\n * immutable model identities.\n */\nexport function modelHasSnapshot(model: string): boolean {\n if (model.length === 0 || model.trim() !== model) return false\n\n const opaqueAt = model.lastIndexOf('@')\n if (opaqueAt > 0) {\n const base = model.slice(0, opaqueAt)\n const token = model.slice(opaqueAt + 1)\n if (!base.includes('@') && /^[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(token)) {\n return true\n }\n }\n\n const isoDate = model.match(/-(\\d{4})-(\\d{2})-(\\d{2})$/u)\n if (isoDate && validSnapshotDate(isoDate[1]!, isoDate[2]!, isoDate[3]!)) return true\n\n const compactDate = model.match(/-(\\d{4})(\\d{2})(\\d{2})$/u)\n if (compactDate && validSnapshotDate(compactDate[1]!, compactDate[2]!, compactDate[3]!)) {\n return true\n }\n\n const routerDate = model.match(/-(\\d{2})(\\d{2})$/u)\n if (routerDate && validSnapshotDate(undefined, routerDate[1]!, routerDate[2]!)) return true\n\n return /:date-[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(model)\n}\n\nfunction validSnapshotDate(year: string | undefined, month: string, day: string): boolean {\n const monthNumber = Number(month)\n const dayNumber = Number(day)\n if (!Number.isInteger(monthNumber) || monthNumber < 1 || monthNumber > 12) return false\n\n const yearNumber = year === undefined ? undefined : Number(year)\n const leapYear =\n yearNumber === undefined ||\n (yearNumber % 4 === 0 && (yearNumber % 100 !== 0 || yearNumber % 400 === 0))\n const daysInMonth = [31, leapYear ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31]\n return Number.isInteger(dayNumber) && dayNumber >= 1 && dayNumber <= daysInMonth[monthNumber - 1]!\n}\n"],"mappings":";;;;;;AAiDA,MAAa,gBAAgB;;;;;;;;;AAuM7B,SAAgB,aAAa,QAAuC;CAClE,MAAM,QAAQ,cAAc,MAAM;CAClC,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,IAAI,QAAQ,KAAA;AACvE;AAIA,MAAM,sBAAsB;CAC1B;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,aAAyC;CAAC;CAAU;CAAO;AAAS;AAC1E,MAAM,oBAAuD;CAC3D;CACA;CACA;CACA;CACA;AACF;AAEA,IAAa,2BAAb,cAA8C,gBAAgB;CAC5D;CACA,YAAY,SAAiB,OAAO,IAAI;EACtC,MAAM,OAAO,GAAG,QAAQ,OAAO,KAAK,KAAK,OAAO;EAChD,KAAK,OAAO;CACd;AACF;;;;;;AAOA,SAAgB,kBAAkB,OAA2B;CAC3D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iBAAiB;CAEtD,MAAM,MAAM;CAEZ,KAAK,MAAM,OAAO,qBAChB,IAAI,EAAE,OAAO,MACX,MAAM,IAAI,yBAAyB,4BAA4B,IAAI,EAAE;CAIzE,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,cAAc,cAAc;CAC7C,aAAa,IAAI,aAAa,aAAa;CAC3C,mBAAmB,IAAI,MAAM,MAAM;CACnC,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,WAAW,WAAW;CACvC,wBAAwB,IAAI,QAAQ,QAAQ;CAC5C,IAAI,IAAI,YAAY,KAAA,GAAW,wBAAwB,IAAI,SAAS,SAAS;CAC7E,aAAa,IAAI,SAAS,IAAI,cAAc;CAK5C,IACE,CAAC,iBAAiB,IAAI,KAAe,KACrC,EAAE,IAAI,UAAA,aAA2B,IAAI,oBAAoB,cAEzD,MAAM,IAAI,yBACR,UAAU,IAAI,MAAM,4EAA4E,cAAc,kDAC9G,OACF;CAIF,MAAM,KAAK,IAAI;CACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,gCAAgC,YAAY;CAEjF,MAAM,QAAQ;CACd,wBAAwB,MAAM,OAAO,kBAAkB;CACvD,wBAAwB,MAAM,QAAQ,mBAAmB;CACzD,IAAI,MAAM,gBAAgB,KAAA,KAAa,MAAM,gBAAgB,OAC3D,MAAM,IAAI,yBACR,gFACA,wBACF;CAEF,IAAI,MAAM,cAAc,KAAA,GAAW;EACjC,wBAAwB,MAAM,WAAW,sBAAsB;EAC/D,IAAK,MAAM,YAAwB,MAAM,QACvC,MAAM,IAAI,yBACR,sDACA,sBACF;CAEJ;CACA,IAAI,MAAM,WAAW,KAAA,GAAW,wBAAwB,MAAM,QAAQ,mBAAmB;CACzF,IAAI,MAAM,eAAe,KAAA,GACvB,wBAAwB,MAAM,YAAY,uBAAuB;CAInE,IAAI,IAAI,kBAAkB,KAAA,GAAW;EACnC,MAAM,KAAK,IAAI;EACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,mCAAmC,eAAe;EAEvF,MAAM,QAAQ;EACd,aAAa,MAAM,OAAO,qBAAqB;EAC/C,aAAa,MAAM,eAAe,6BAA6B;EAC/D,mBAAmB,MAAM,YAAY,0BAA0B;EAC/D,IAAI,OAAO,MAAM,aAAa,WAC5B,MAAM,IAAI,yBACR,0CACA,wBACF;CAEJ;CAGA,MAAM,MAAM,IAAI;CAChB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,6BAA6B,SAAS;CAE3E,MAAM,SAAS;CACf,IAAI,OAAO,gBAAgB,KAAA,GACzB,mBAAmB,OAAO,aAAa,qBAAqB;CAC9D,IAAI,OAAO,iBAAiB,KAAA,GAC1B,mBAAmB,OAAO,cAAc,sBAAsB;CAChE,MAAM,MAAM,OAAO;CACnB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,iCAAiC,aAAa;CAEnF,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,GAA8B,GAChE,mBAAmB,GAAG,eAAe,GAAG;CAG1C,IAAI,OAAO,aAAa,KAAA,GAAW;EACjC,MAAM,IAAI,OAAO;EACjB,IAAI,MAAM,QAAQ,OAAO,MAAM,UAC7B,MAAM,IAAI,yBAAyB,sCAAsC,kBAAkB;EAE7F,MAAM,KAAK;EACX,mBAAmB,GAAG,OAAO,wBAAwB;EACrD,IAAI,OAAO,GAAG,UAAU,WACtB,MAAM,IAAI,yBACR,4CACA,wBACF;CAEJ;CAGA,IAAI,OAAO,gBAAgB,KAAA,GACzB,oBAAoB,OAAO,aAAa,qBAAqB;CAI/D,IACE,IAAI,iBAAiB,KAAA,MACpB,OAAO,IAAI,iBAAiB,YAC3B,CAAC,gBAAgB,SAAS,IAAI,YAA4B,IAE5D,MAAM,IAAI,yBACR,+BAA+B,gBAAgB,KAAK,IAAI,KACxD,cACF;CAEF,IAAI,IAAI,gBAAgB,KAAA,GAAW;EACjC,aAAa,IAAI,aAAa,aAAa;EAC3C,IAAI,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,WACzD,MAAM,IAAI,yBACR,mDACA,aACF;CAEJ;CAEA,IACE,OAAO,IAAI,oBAAoB,YAC/B,CAAC,kBAAkB,SAAS,IAAI,eAAqC,GAErE,MAAM,IAAI,yBACR,kCAAkC,kBAAkB,KAAK,IAAI,KAC7D,iBACF;CAEF,IAAI,IAAI,0BAA0B,KAAA,GAAW;EAC3C,aAAa,IAAI,uBAAuB,uBAAuB;EAC/D,IACE,IAAI,oBAAoB,YACxB,IAAI,oBAAoB,eACxB,IAAI,oBAAoB,cAExB,MAAM,IAAI,yBACR,mFACA,uBACF;CAEJ;CAEA,IAAI,IAAI,iBAAiB,KAAA,GACvB,IAAI;EACF,MAAM,UAAU,yBAAyB,IAAI,YAAY;EACzD,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,IAAI,OACvD,MAAM,IAAI,yBACR,uBAAuB,QAAQ,MAAM,0BAA0B,IAAI,MAAM,IACzE,oBACF;EAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,IAAI,YACjE,MAAM,IAAI,yBACR,4BAA4B,QAAQ,WAAW,+BAA+B,IAAI,WAAW,IAC7F,yBACF;CAEJ,SAAS,OAAO;EACd,IAAI,iBAAiB,0BAA0B,MAAM;EACrD,IAAI,iBAAiB,OACnB,MAAM,IAAI,yBAAyB,MAAM,SAAS,cAAc;EAElE,MAAM;CACR;CAGF,aAAa,IAAI,YAAY,YAAY;CAGzC,IAAI,OAAO,IAAI,aAAa,YAAY,CAAC,WAAW,SAAS,IAAI,QAAuB,GACtF,MAAM,IAAI,yBACR,2BAA2B,WAAW,KAAK,IAAI,EAAE,QAAQ,OAAO,IAAI,QAAQ,KAC5E,UACF;CAGF,OAAO;AACT;AAEA,SAAS,aAAa,SAAkB,YAA2B;CACjE,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,oCAAoC,gBAAgB;CAEzF,MAAM,QAAQ;CACd,IAAI,MAAM,SAAS,cAAc,MAAM,SAAS,eAAe,MAAM,SAAS,cAC5E,MAAM,IAAI,yBACR,kEACA,qBACF;CAMF,IAAI,MAAM,SAAS,cAAc;EAC/B,IAAI,MAAM,QAAQ,MAChB,MAAM,IAAI,yBACR,8CACA,oBACF;EAEF,IAAI,YAAY,MACd,MAAM,IAAI,yBAAyB,+CAA+C,SAAS;EAE7F;CACF;CACA,wBAAwB,SAAS,SAAS;CAC1C,wBAAwB,MAAM,KAAK,oBAAoB;CACvD,IAAI,MAAM,QAAQ,SAChB,MAAM,IAAI,yBACR,yCACA,oBACF;AAEJ;;AAGA,SAAgB,YAAY,OAAoC;CAC9D,IAAI;EACF,kBAAkB,KAAK;EACvB,OAAO;CACT,QAAQ;EACN,OAAO;CACT;AACF;;AAGA,SAAgB,mBACd,OACiF;CACjF,IAAI;EACF,OAAO;GAAE,IAAI;GAAM,OAAO,kBAAkB,KAAK;EAAE;CACrD,SAAS,GAAG;EACV,IAAI,aAAa,0BAA0B,OAAO;GAAE,IAAI;GAAO,OAAO;EAAE;EACxE,MAAM;CACR;AACF;;AAGA,SAAgB,mBAAmB,QAA8B;CAC/D,MAAM,OAAO,KAAK,UAAU,MAAM;CAClC,OAAO,kBAAkB,KAAK,MAAM,IAAI,CAAC;AAC3C;AAIA,SAAS,aAAa,OAAgB,MAAoB;CACxD,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,yBAAyB,6BAA6B,IAAI;AAExE;AAEA,SAAS,mBAAmB,OAAgB,MAAoB;CAC9D,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GACrD,MAAM,IAAI,yBAAyB,0BAA0B,IAAI;AAErE;AAEA,SAAS,wBAAwB,OAAgB,MAAoB;CACnE,mBAAmB,OAAO,IAAI;CAC9B,IAAK,QAAmB,GACtB,MAAM,IAAI,yBAAyB,gCAAgC,IAAI;AAE3E;AAEA,SAAS,oBAAoB,OAAgB,MAAoB;CAC/D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iCAAiC,IAAI;CAE1E,MAAM,MAAM;CAEZ,MAAM,WAAW,IAAI;CACrB,IAAI,aAAa,QAAQ,OAAO,aAAa,UAC3C,MAAM,IAAI,yBAAyB,8BAA8B,GAAG,KAAK,UAAU;CAErF,KAAK,MAAM,CAAC,SAAS,SAAS,OAAO,QAAQ,QAAmC,GAAG;EACjF,IAAI,SAAS,QAAQ,OAAO,SAAS,UACnC,MAAM,IAAI,yBACR,yDACA,GAAG,KAAK,YAAY,SACtB;EAEF,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,IAA+B,GACvE,mBAAmB,OAAO,GAAG,KAAK,YAAY,QAAQ,GAAG,KAAK;CAElE;CAEA,MAAM,aAAa,IAAI;CACvB,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,gCAAgC,GAAG,KAAK,YAAY;CAEzF,KAAK,MAAM,CAAC,KAAK,SAAS,OAAO,QAAQ,UAAqC,GAC5E,mBAAmB,MAAM,GAAG,KAAK,cAAc,KAAK;CAGtD,mBAAmB,IAAI,WAAW,GAAG,KAAK,WAAW;CAErD,IAAI,IAAI,iBAAiB,KAAA,GAAW;EAClC,IAAI,CAAC,MAAM,QAAQ,IAAI,YAAY,GACjC,MAAM,IAAI,yBACR,4CACA,GAAG,KAAK,cACV;EAEF,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,aAAa,QAAQ,KAAK;GAChD,MAAM,KAAK,IAAI,aAAa;GAC5B,IAAI,OAAO,OAAO,YAAY,GAAG,WAAW,GAC1C,MAAM,IAAI,yBACR,iDACA,GAAG,KAAK,gBAAgB,EAAE,EAC5B;EAEJ;CACF;CAEA,IAAI,IAAI,UAAU,KAAA,KAAa,OAAO,IAAI,UAAU,UAClD,MAAM,IAAI,yBAAyB,0BAA0B,GAAG,KAAK,OAAO;AAEhF;;;;;;;AAQA,SAAgB,iBAAiB,OAAwB;CACvD,IAAI,MAAM,WAAW,KAAK,MAAM,KAAK,MAAM,OAAO,OAAO;CAEzD,MAAM,WAAW,MAAM,YAAY,GAAG;CACtC,IAAI,WAAW,GAAG;EAChB,MAAM,OAAO,MAAM,MAAM,GAAG,QAAQ;EACpC,MAAM,QAAQ,MAAM,MAAM,WAAW,CAAC;EACtC,IAAI,CAAC,KAAK,SAAS,GAAG,KAAK,gDAAgD,KAAK,KAAK,GACnF,OAAO;CAEX;CAEA,MAAM,UAAU,MAAM,MAAM,4BAA4B;CACxD,IAAI,WAAW,kBAAkB,QAAQ,IAAK,QAAQ,IAAK,QAAQ,EAAG,GAAG,OAAO;CAEhF,MAAM,cAAc,MAAM,MAAM,0BAA0B;CAC1D,IAAI,eAAe,kBAAkB,YAAY,IAAK,YAAY,IAAK,YAAY,EAAG,GACpF,OAAO;CAGT,MAAM,aAAa,MAAM,MAAM,mBAAmB;CAClD,IAAI,cAAc,kBAAkB,KAAA,GAAW,WAAW,IAAK,WAAW,EAAG,GAAG,OAAO;CAEvF,OAAO,qDAAqD,KAAK,KAAK;AACxE;AAEA,SAAS,kBAAkB,MAA0B,OAAe,KAAsB;CACxF,MAAM,cAAc,OAAO,KAAK;CAChC,MAAM,YAAY,OAAO,GAAG;CAC5B,IAAI,CAAC,OAAO,UAAU,WAAW,KAAK,cAAc,KAAK,cAAc,IAAI,OAAO;CAElF,MAAM,aAAa,SAAS,KAAA,IAAY,KAAA,IAAY,OAAO,IAAI;CAI/D,MAAM,cAAc;EAAC;EAFnB,eAAe,KAAA,KACd,aAAa,MAAM,MAAM,aAAa,QAAQ,KAAK,aAAa,QAAQ,KACvC,KAAK;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;CAAE;CACnF,OAAO,OAAO,UAAU,SAAS,KAAK,aAAa,KAAK,aAAa,YAAY,cAAc;AACjG"}
1
+ {"version":3,"file":"run-record-DLORoL7t.js","names":[],"sources":["../src/run-record.ts"],"sourcesContent":["/**\n * Paper-grade RunRecord schema + runtime validator.\n *\n * Every run that participates in a promotion gate, paper table, or\n * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory\n * fields are exactly those the paper \"Two Loops, Three Roles\" requires\n * for reproducibility: who/what/when/cost/seed/hash, plus the search vs\n * holdout split tag. A task score is optional because execution-only records\n * must preserve missing labels instead of converting errors into zero quality.\n *\n * This is intentionally NOT a replacement for the rich `Run` /\n * `ProposeReviewReport` / `ScenarioResult` types already in the\n * package. Those are runtime structures with full provenance. A\n * `RunRecord` is the analysis-time projection — the JSON-friendly\n * row you'd put in a parquet file or paste into a notebook.\n *\n * Validate at the boundary:\n *\n * const rec = validateRunRecord(rawJson) // throws on missing\n * const ok = isRunRecord(rawJson) // boolean check\n * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }\n *\n * The validator runs in pure TS — zod is intentionally NOT a\n * dependency. Round-trip tested in `tests/run-record.test.ts`.\n */\n\nimport type { AgentProfileCell } from './agent-profile-cell'\nimport { validateAgentProfileCell } from './agent-profile-cell'\nimport type { CostProvenance } from './cost-ledger'\nimport { ValidationError } from './errors'\n// Value import of a leaf module that itself imports only this file's TYPES —\n// no runtime cycle. It keeps the raw split-score derivation spelled in exactly\n// one place (see `rollout/score-derivation-guard`).\nimport { observedScore } from './rollout/reward'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the\n * combined train+test pool that the optimizer is allowed to read. */\nexport type RunSplitTag = 'search' | 'dev' | 'holdout'\n\n/**\n * Explicit execution-lifecycle result for a run.\n *\n * This is separate from task quality (`outcome`) and failure classification.\n * Producers set it only from root-run or process evidence.\n */\nexport type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown'\n\n/** Explicit model value for a row that never produced a served model snapshot. */\nexport const UNKNOWN_MODEL = 'unknown'\n\nexport interface RunTokenUsage {\n input: number\n /** All generated tokens charged as output, including reasoning tokens. */\n output: number\n /** Present only when one or more paid calls did not report token usage.\n * In that case, every numeric field is a known subtotal, not a measured total. */\n tokensKnown?: false\n /** Reasoning-token subset of `output`, when the provider reports it. */\n reasoning?: number\n /** Prompt tokens served from a provider cache. */\n cached?: number\n /** Prompt tokens written into a provider cache. */\n cacheWrite?: number\n}\n\n/** How a run's USD amount was obtained. */\nexport type RunCostProvenance = CostProvenance\n\nexport interface RunJudgeMetadata {\n model: string\n promptVersion: string\n /** [0,1] confidence the judge declared. Constant judge confidence\n * across many runs is a fallback signal (see `canary.ts`). */\n confidence: number\n /** True if the judge degraded to a fallback path (rules-only,\n * prior-call cache, etc.). The canary uses this to alert. */\n fallback: boolean\n}\n\n/**\n * Per-judge / per-dimension breakdown for runs scored by an ensemble of\n * judges over a multi-dimensional rubric.\n *\n * The collapsed `outcome.searchScore` / `holdoutScore` carries the\n * composite the gate uses. The full breakdown belongs here so consumers\n * can answer \"which judge disagreed?\", \"which dimension dragged the\n * composite down?\", and \"did half the panel fail?\" without re-running.\n *\n * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and\n * `composite` are convenience projections — derivable but precomputed so\n * downstream IRR primitives (`interRaterReliability`,\n * `corpusInterRaterAgreement`) and reporters don't pay the same\n * aggregation twice.\n *\n * Fail-loud discipline: judges that errored out land in `failedJudges`\n * by id. A missing key in `perJudge` is ambiguous (silent zero vs not\n * run); the explicit list makes a partial-failure recorded as such.\n */\nexport interface JudgeScoresRecord {\n /** Per-judge per-dimension scores. `{ \"kimi-k2.6\": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */\n perJudge: Record<string, Record<string, number>>\n /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */\n perDimMean: Record<string, number>\n /** Composite mean across successful judges. Mirrors the task score only\n * when `failedJudges` is empty. */\n composite: number\n /** Judges that errored or returned an unparseable verdict. Recorded\n * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,\n * not inferred from missing keys in `perJudge`. */\n failedJudges?: string[]\n /** Free-form notes the judges emitted (joined across judges or\n * first-judge only — consumer's choice). */\n notes?: string\n}\n\nexport interface RunOutcome {\n /** Score on the search/optimization split. Optional for holdout-only and\n * execution-only records. */\n searchScore?: number\n /** Score on the held-out split. Optional for search-only and execution-only\n * records. When both scores are absent, the run is explicitly unlabeled. */\n holdoutScore?: number\n /** Bag of any other metric the run produced — judge dimensions,\n * pass/fail counters, latency stats, etc. Numeric only — keeps\n * reporters honest. */\n raw: Record<string, number>\n /** Per-judge / per-dim breakdown. Consumers writing ensemble\n * judgements populate this; substrate primitives like\n * `interRaterReliability` and `corpusInterRaterAgreement` accept\n * these records as input. Optional — single-judge or scalar-only\n * runs leave it unset. */\n judgeScores?: JudgeScoresRecord\n /** Authenticity / realness verdict — did the run build the REAL thing on the\n * intended infra, or fake it (see `./authenticity`)? Optional: only domains\n * with an authenticity config populate it. Carried in the corpus so the\n * flywheel / off-policy learning can optimize for real completion, not gamed\n * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run\n * must not count as a real success regardless of `score`. */\n realness?: { score: number; gated: boolean; reason?: string }\n}\n\n/**\n * Mandatory paper-grade fields for a single evaluation run. Optional\n * fields are extension points; mandatory fields throw if missing.\n *\n * Hash discipline:\n * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the\n * model (after any steering bundle merge).\n * - `configHash` is the sha256 of the effective run config (model,\n * temperature, tools, judges, splits). The pair (promptHash,\n * configHash) uniquely identifies an experiment cell.\n *\n * Model snapshot discipline:\n * - successful rows MUST encode a snapshot version. Bare aliases like\n * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.\n * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.\n * - a failed, cancelled, incomplete, or otherwise unknown row may use\n * `UNKNOWN_MODEL` when no served model was observed. This is an explicit\n * absence marker, not a fabricated snapshot.\n */\nexport interface RunRecord {\n /** UUID for the run. */\n runId: string\n /** Logical experiment grouping (a treatment vs a baseline within\n * the same sweep should share `experimentId`). */\n experimentId: string\n /** Stable identifier for the candidate (variant) being run. The\n * promotion gate compares two `candidateId`s on matched items. */\n candidateId: string\n /** RNG seed for the run. Always recorded — silent re-seeding is\n * the most common cause of non-reproducible numbers. */\n seed: number\n /** Model identifier WITH snapshot version. */\n model: string\n /** sha256 of the effective prompt (post-steering). */\n promptHash: string\n /** sha256 of the effective config. */\n configHash: string\n /** Git SHA the harness was run from. */\n commitSha: string\n /** End-to-end wall-clock duration in milliseconds. */\n wallMs: number\n /** Time spent queued before execution started, if known. */\n queueMs?: number\n /** Total USD cost, or null when the producer could not capture one. */\n costUsd: number | null\n /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */\n costProvenance: RunCostProvenance\n /** Token usage breakdown. */\n tokenUsage: RunTokenUsage\n /** Root-run or process terminal result. Never inferred from a child span. */\n terminalOutcome: RunTerminalOutcome\n /** Root-run or process failure reason. Valid only for a failed, cancelled,\n * or incomplete terminal result; never populated from a child span. */\n terminalFailureReason?: string\n /** Judge-side metadata, if a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /** Per-split scores + raw bag. */\n outcome: RunOutcome\n /** Canonical task-failure class drawn from the shared\n * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result\n * evidence. Execution errors belong in\n * `outcome.raw.execution_error_count`. */\n failureClass?: FailureClass\n /** Free-form task-failure detail scoped under a non-success\n * `failureClass`. It is invalid without that class. */\n failureMode?: string\n /** Which split this run was drawn from. */\n splitTag: RunSplitTag\n /**\n * Stable scenario identifier the run observed or was scored against.\n * Comparison primitives match this identity rather than input order.\n */\n scenarioId: string\n /**\n * Canonical identity for the agent profile cell that produced this row:\n * profile artifact hash plus optional harness/model/prompt/reporting\n * dimensions. Use `agentProfile.cellId` to group persona sweeps and\n * longitudinal reports by the complete source profile, not by a loose\n * candidate label or opaque config hash.\n */\n agentProfile?: AgentProfileCell\n}\n\n/**\n * Canonical task-result classification.\n *\n * A producer may omit classification, record explicit success, or attach\n * domain-specific detail to a non-success class. Detail can never stand alone.\n * Execution errors belong in `outcome.raw.execution_error_count`.\n */\nexport type RunTaskFailure =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | {\n failureClass: Exclude<FailureClass, 'success'>\n failureMode?: string\n }\n\n/**\n * Return task quality, preferring held-out evidence when both scores exist.\n *\n * RAW: no realness protection is applied. Built on `observedScore` rather\n * than repeating the split derivation, so only `rollout/reward.ts` reads the\n * raw fields. Anything that becomes training data must use `trainingScore` or\n * `trainingReward` instead.\n */\nexport function runTaskScore(record: RunRecord): number | undefined {\n const score = observedScore(record)\n return typeof score === 'number' && Number.isFinite(score) ? score : undefined\n}\n\n// ── Validation ───────────────────────────────────────────────────────\n\nconst MANDATORY_TOP_LEVEL = [\n 'runId',\n 'experimentId',\n 'candidateId',\n 'seed',\n 'model',\n 'promptHash',\n 'configHash',\n 'commitSha',\n 'wallMs',\n 'costUsd',\n 'costProvenance',\n 'tokenUsage',\n 'terminalOutcome',\n 'outcome',\n 'splitTag',\n 'scenarioId',\n] as const\n\nconst SPLIT_TAGS: ReadonlyArray<RunSplitTag> = ['search', 'dev', 'holdout']\nconst TERMINAL_OUTCOMES: ReadonlyArray<RunTerminalOutcome> = [\n 'succeeded',\n 'failed',\n 'cancelled',\n 'incomplete',\n 'unknown',\n]\n\nexport class RunRecordValidationError extends ValidationError {\n readonly path: string\n constructor(message: string, path = '') {\n super(path ? `${message} (at ${path})` : message)\n this.path = path\n }\n}\n\n/**\n * Strict validator. Throws `RunRecordValidationError` on the first\n * missing or wrongly-typed field. Returns the input cast to\n * `RunRecord` on success — the validator does not coerce.\n */\nexport function validateRunRecord(input: unknown): RunRecord {\n if (input === null || typeof input !== 'object') {\n throw new RunRecordValidationError('expected object')\n }\n const obj = input as Record<string, unknown>\n\n for (const key of MANDATORY_TOP_LEVEL) {\n if (!(key in obj)) {\n throw new RunRecordValidationError(`missing mandatory field \"${key}\"`)\n }\n }\n\n expectString(obj.runId, 'runId')\n expectString(obj.experimentId, 'experimentId')\n expectString(obj.candidateId, 'candidateId')\n expectFiniteNumber(obj.seed, 'seed')\n expectString(obj.model, 'model')\n expectString(obj.promptHash, 'promptHash')\n expectString(obj.configHash, 'configHash')\n expectString(obj.commitSha, 'commitSha')\n expectNonNegativeNumber(obj.wallMs, 'wallMs')\n if (obj.queueMs !== undefined) expectNonNegativeNumber(obj.queueMs, 'queueMs')\n validateCost(obj.costUsd, obj.costProvenance)\n\n // Snapshot discipline: successful rows require a served model snapshot.\n // Non-success rows may carry the explicit absence marker when execution\n // stopped before a model identity was observed.\n if (\n !modelHasSnapshot(obj.model as string) &&\n !(obj.model === UNKNOWN_MODEL && obj.terminalOutcome !== 'succeeded')\n ) {\n throw new RunRecordValidationError(\n `model \"${obj.model}\" lacks a snapshot version (use 'name@YYYY-MM-DD' or 'name-YYYYMMDD', or '${UNKNOWN_MODEL}' for a non-success row without a served model)`,\n 'model',\n )\n }\n\n // Token usage.\n const tu = obj.tokenUsage\n if (tu === null || typeof tu !== 'object') {\n throw new RunRecordValidationError('tokenUsage must be an object', 'tokenUsage')\n }\n const tuRec = tu as Record<string, unknown>\n expectNonNegativeNumber(tuRec.input, 'tokenUsage.input')\n expectNonNegativeNumber(tuRec.output, 'tokenUsage.output')\n if (tuRec.tokensKnown !== undefined && tuRec.tokensKnown !== false) {\n throw new RunRecordValidationError(\n 'tokensKnown must be false when present; omit it when token usage is complete',\n 'tokenUsage.tokensKnown',\n )\n }\n if (tuRec.reasoning !== undefined) {\n expectNonNegativeNumber(tuRec.reasoning, 'tokenUsage.reasoning')\n if ((tuRec.reasoning as number) > (tuRec.output as number)) {\n throw new RunRecordValidationError(\n 'reasoning tokens must be a subset of output tokens',\n 'tokenUsage.reasoning',\n )\n }\n }\n if (tuRec.cached !== undefined) expectNonNegativeNumber(tuRec.cached, 'tokenUsage.cached')\n if (tuRec.cacheWrite !== undefined) {\n expectNonNegativeNumber(tuRec.cacheWrite, 'tokenUsage.cacheWrite')\n }\n\n // Judge metadata, optional.\n if (obj.judgeMetadata !== undefined) {\n const jm = obj.judgeMetadata\n if (jm === null || typeof jm !== 'object') {\n throw new RunRecordValidationError('judgeMetadata must be an object', 'judgeMetadata')\n }\n const jmRec = jm as Record<string, unknown>\n expectString(jmRec.model, 'judgeMetadata.model')\n expectString(jmRec.promptVersion, 'judgeMetadata.promptVersion')\n expectFiniteNumber(jmRec.confidence, 'judgeMetadata.confidence')\n if (typeof jmRec.fallback !== 'boolean') {\n throw new RunRecordValidationError(\n 'judgeMetadata.fallback must be boolean',\n 'judgeMetadata.fallback',\n )\n }\n }\n\n // Outcome.\n const out = obj.outcome\n if (out === null || typeof out !== 'object') {\n throw new RunRecordValidationError('outcome must be an object', 'outcome')\n }\n const outRec = out as Record<string, unknown>\n if (outRec.searchScore !== undefined)\n expectFiniteNumber(outRec.searchScore, 'outcome.searchScore')\n if (outRec.holdoutScore !== undefined)\n expectFiniteNumber(outRec.holdoutScore, 'outcome.holdoutScore')\n const raw = outRec.raw\n if (raw === null || typeof raw !== 'object') {\n throw new RunRecordValidationError('outcome.raw must be an object', 'outcome.raw')\n }\n for (const [k, v] of Object.entries(raw as Record<string, unknown>)) {\n expectFiniteNumber(v, `outcome.raw.${k}`)\n }\n // Realness verdict, optional.\n if (outRec.realness !== undefined) {\n const r = outRec.realness\n if (r === null || typeof r !== 'object') {\n throw new RunRecordValidationError('outcome.realness must be an object', 'outcome.realness')\n }\n const rr = r as Record<string, unknown>\n expectFiniteNumber(rr.score, 'outcome.realness.score')\n if (typeof rr.gated !== 'boolean') {\n throw new RunRecordValidationError(\n 'outcome.realness.gated must be a boolean',\n 'outcome.realness.gated',\n )\n }\n }\n\n // Per-judge / per-dim breakdown, optional.\n if (outRec.judgeScores !== undefined) {\n validateJudgeScores(outRec.judgeScores, 'outcome.judgeScores')\n }\n\n // Failure mode optional.\n if (\n obj.failureClass !== undefined &&\n (typeof obj.failureClass !== 'string' ||\n !FAILURE_CLASSES.includes(obj.failureClass as FailureClass))\n ) {\n throw new RunRecordValidationError(\n `failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n 'failureClass',\n )\n }\n if (obj.failureMode !== undefined) {\n expectString(obj.failureMode, 'failureMode')\n if (obj.failureClass === undefined || obj.failureClass === 'success') {\n throw new RunRecordValidationError(\n 'failureMode requires a non-success failureClass',\n 'failureMode',\n )\n }\n }\n\n if (\n typeof obj.terminalOutcome !== 'string' ||\n !TERMINAL_OUTCOMES.includes(obj.terminalOutcome as RunTerminalOutcome)\n ) {\n throw new RunRecordValidationError(\n `terminalOutcome must be one of ${TERMINAL_OUTCOMES.join(', ')}`,\n 'terminalOutcome',\n )\n }\n if (obj.terminalFailureReason !== undefined) {\n expectString(obj.terminalFailureReason, 'terminalFailureReason')\n if (\n obj.terminalOutcome !== 'failed' &&\n obj.terminalOutcome !== 'cancelled' &&\n obj.terminalOutcome !== 'incomplete'\n ) {\n throw new RunRecordValidationError(\n 'terminalFailureReason requires terminalOutcome failed, cancelled, or incomplete',\n 'terminalFailureReason',\n )\n }\n }\n\n if (obj.agentProfile !== undefined) {\n try {\n const profile = validateAgentProfileCell(obj.agentProfile)\n if (profile.model !== undefined && profile.model !== obj.model) {\n throw new RunRecordValidationError(\n `agentProfile.model \"${profile.model}\" does not match model \"${obj.model}\"`,\n 'agentProfile.model',\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== obj.promptHash) {\n throw new RunRecordValidationError(\n `agentProfile.promptHash \"${profile.promptHash}\" does not match promptHash \"${obj.promptHash}\"`,\n 'agentProfile.promptHash',\n )\n }\n } catch (error) {\n if (error instanceof RunRecordValidationError) throw error\n if (error instanceof Error) {\n throw new RunRecordValidationError(error.message, 'agentProfile')\n }\n throw error\n }\n }\n\n expectString(obj.scenarioId, 'scenarioId')\n\n // Split tag.\n if (typeof obj.splitTag !== 'string' || !SPLIT_TAGS.includes(obj.splitTag as RunSplitTag)) {\n throw new RunRecordValidationError(\n `splitTag must be one of ${SPLIT_TAGS.join(', ')}, got ${String(obj.splitTag)}`,\n 'splitTag',\n )\n }\n\n return input as RunRecord\n}\n\nfunction validateCost(costUsd: unknown, provenance: unknown): void {\n if (provenance === null || typeof provenance !== 'object') {\n throw new RunRecordValidationError('costProvenance must be an object', 'costProvenance')\n }\n const value = provenance as Record<string, unknown>\n if (value.kind !== 'observed' && value.kind !== 'estimated' && value.kind !== 'uncaptured') {\n throw new RunRecordValidationError(\n 'costProvenance.kind must be observed, estimated, or uncaptured',\n 'costProvenance.kind',\n )\n }\n // A record must never read as a total it cannot support, so an uncaptured\n // cost carries no number at all. A matrix `CellResult` keeps its known\n // subtotal instead, because a cost ceiling must charge the part it can see;\n // converting one to the other drops that subtotal.\n if (value.kind === 'uncaptured') {\n if (value.usd !== null) {\n throw new RunRecordValidationError(\n 'uncaptured costProvenance.usd must be null',\n 'costProvenance.usd',\n )\n }\n if (costUsd !== null) {\n throw new RunRecordValidationError('uncaptured cost requires costUsd to be null', 'costUsd')\n }\n return\n }\n expectNonNegativeNumber(costUsd, 'costUsd')\n expectNonNegativeNumber(value.usd, 'costProvenance.usd')\n if (value.usd !== costUsd) {\n throw new RunRecordValidationError(\n 'costProvenance.usd must equal costUsd',\n 'costProvenance.usd',\n )\n }\n}\n\n/** Boolean validator — convenience for filtering arrays. */\nexport function isRunRecord(input: unknown): input is RunRecord {\n try {\n validateRunRecord(input)\n return true\n } catch {\n return false\n }\n}\n\n/** Non-throwing validator — returns a discriminated union. */\nexport function parseRunRecordSafe(\n input: unknown,\n): { ok: true; value: RunRecord } | { ok: false; error: RunRecordValidationError } {\n try {\n return { ok: true, value: validateRunRecord(input) }\n } catch (e) {\n if (e instanceof RunRecordValidationError) return { ok: false, error: e }\n throw e\n }\n}\n\n/** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */\nexport function roundTripRunRecord(record: RunRecord): RunRecord {\n const json = JSON.stringify(record)\n return validateRunRecord(JSON.parse(json))\n}\n\n// ── Internals ────────────────────────────────────────────────────────\n\nfunction expectString(value: unknown, path: string): void {\n if (typeof value !== 'string' || value.length === 0) {\n throw new RunRecordValidationError(`expected non-empty string`, path)\n }\n}\n\nfunction expectFiniteNumber(value: unknown, path: string): void {\n if (typeof value !== 'number' || !Number.isFinite(value)) {\n throw new RunRecordValidationError(`expected finite number`, path)\n }\n}\n\nfunction expectNonNegativeNumber(value: unknown, path: string): void {\n expectFiniteNumber(value, path)\n if ((value as number) < 0) {\n throw new RunRecordValidationError('expected non-negative number', path)\n }\n}\n\nfunction validateJudgeScores(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object') {\n throw new RunRecordValidationError('judgeScores must be an object', path)\n }\n const rec = value as Record<string, unknown>\n\n const perJudge = rec.perJudge\n if (perJudge === null || typeof perJudge !== 'object') {\n throw new RunRecordValidationError('perJudge must be an object', `${path}.perJudge`)\n }\n for (const [judgeId, dims] of Object.entries(perJudge as Record<string, unknown>)) {\n if (dims === null || typeof dims !== 'object') {\n throw new RunRecordValidationError(\n 'per-judge entry must be an object of dimension scores',\n `${path}.perJudge.${judgeId}`,\n )\n }\n for (const [dim, score] of Object.entries(dims as Record<string, unknown>)) {\n expectFiniteNumber(score, `${path}.perJudge.${judgeId}.${dim}`)\n }\n }\n\n const perDimMean = rec.perDimMean\n if (perDimMean === null || typeof perDimMean !== 'object') {\n throw new RunRecordValidationError('perDimMean must be an object', `${path}.perDimMean`)\n }\n for (const [dim, mean] of Object.entries(perDimMean as Record<string, unknown>)) {\n expectFiniteNumber(mean, `${path}.perDimMean.${dim}`)\n }\n\n expectFiniteNumber(rec.composite, `${path}.composite`)\n\n if (rec.failedJudges !== undefined) {\n if (!Array.isArray(rec.failedJudges)) {\n throw new RunRecordValidationError(\n 'failedJudges must be an array of strings',\n `${path}.failedJudges`,\n )\n }\n for (let i = 0; i < rec.failedJudges.length; i++) {\n const id = rec.failedJudges[i]\n if (typeof id !== 'string' || id.length === 0) {\n throw new RunRecordValidationError(\n 'failedJudges entry must be a non-empty string',\n `${path}.failedJudges[${i}]`,\n )\n }\n }\n }\n\n if (rec.notes !== undefined && typeof rec.notes !== 'string') {\n throw new RunRecordValidationError('notes must be a string', `${path}.notes`)\n }\n}\n\n/**\n * Snapshot check for provider model identifiers. Accepts ISO and compact\n * dates, Router's `-MMDD` snapshots, one opaque `@token`, and Vertex-style\n * `:date-token` suffixes. Routing selectors such as `@preset/name` are not\n * immutable model identities.\n */\nexport function modelHasSnapshot(model: string): boolean {\n if (model.length === 0 || model.trim() !== model) return false\n\n const opaqueAt = model.lastIndexOf('@')\n if (opaqueAt > 0) {\n const base = model.slice(0, opaqueAt)\n const token = model.slice(opaqueAt + 1)\n if (!base.includes('@') && /^[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(token)) {\n return true\n }\n }\n\n const isoDate = model.match(/-(\\d{4})-(\\d{2})-(\\d{2})$/u)\n if (isoDate && validSnapshotDate(isoDate[1]!, isoDate[2]!, isoDate[3]!)) return true\n\n const compactDate = model.match(/-(\\d{4})(\\d{2})(\\d{2})$/u)\n if (compactDate && validSnapshotDate(compactDate[1]!, compactDate[2]!, compactDate[3]!)) {\n return true\n }\n\n const routerDate = model.match(/-(\\d{2})(\\d{2})$/u)\n if (routerDate && validSnapshotDate(undefined, routerDate[1]!, routerDate[2]!)) return true\n\n return /:date-[A-Za-z0-9](?:[A-Za-z0-9._-]*[A-Za-z0-9])?$/u.test(model)\n}\n\nfunction validSnapshotDate(year: string | undefined, month: string, day: string): boolean {\n const monthNumber = Number(month)\n const dayNumber = Number(day)\n if (!Number.isInteger(monthNumber) || monthNumber < 1 || monthNumber > 12) return false\n\n const yearNumber = year === undefined ? undefined : Number(year)\n const leapYear =\n yearNumber === undefined ||\n (yearNumber % 4 === 0 && (yearNumber % 100 !== 0 || yearNumber % 400 === 0))\n const daysInMonth = [31, leapYear ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31]\n return Number.isInteger(dayNumber) && dayNumber >= 1 && dayNumber <= daysInMonth[monthNumber - 1]!\n}\n"],"mappings":";;;;;;AAiDA,MAAa,gBAAgB;;;;;;;;;AAuM7B,SAAgB,aAAa,QAAuC;CAClE,MAAM,QAAQ,cAAc,MAAM;CAClC,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,IAAI,QAAQ,KAAA;AACvE;AAIA,MAAM,sBAAsB;CAC1B;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,aAAyC;CAAC;CAAU;CAAO;AAAS;AAC1E,MAAM,oBAAuD;CAC3D;CACA;CACA;CACA;CACA;AACF;AAEA,IAAa,2BAAb,cAA8C,gBAAgB;CAC5D;CACA,YAAY,SAAiB,OAAO,IAAI;EACtC,MAAM,OAAO,GAAG,QAAQ,OAAO,KAAK,KAAK,OAAO;EAChD,KAAK,OAAO;CACd;AACF;;;;;;AAOA,SAAgB,kBAAkB,OAA2B;CAC3D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iBAAiB;CAEtD,MAAM,MAAM;CAEZ,KAAK,MAAM,OAAO,qBAChB,IAAI,EAAE,OAAO,MACX,MAAM,IAAI,yBAAyB,4BAA4B,IAAI,EAAE;CAIzE,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,cAAc,cAAc;CAC7C,aAAa,IAAI,aAAa,aAAa;CAC3C,mBAAmB,IAAI,MAAM,MAAM;CACnC,aAAa,IAAI,OAAO,OAAO;CAC/B,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,YAAY,YAAY;CACzC,aAAa,IAAI,WAAW,WAAW;CACvC,wBAAwB,IAAI,QAAQ,QAAQ;CAC5C,IAAI,IAAI,YAAY,KAAA,GAAW,wBAAwB,IAAI,SAAS,SAAS;CAC7E,aAAa,IAAI,SAAS,IAAI,cAAc;CAK5C,IACE,CAAC,iBAAiB,IAAI,KAAe,KACrC,EAAE,IAAI,UAAA,aAA2B,IAAI,oBAAoB,cAEzD,MAAM,IAAI,yBACR,UAAU,IAAI,MAAM,4EAA4E,cAAc,kDAC9G,OACF;CAIF,MAAM,KAAK,IAAI;CACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,gCAAgC,YAAY;CAEjF,MAAM,QAAQ;CACd,wBAAwB,MAAM,OAAO,kBAAkB;CACvD,wBAAwB,MAAM,QAAQ,mBAAmB;CACzD,IAAI,MAAM,gBAAgB,KAAA,KAAa,MAAM,gBAAgB,OAC3D,MAAM,IAAI,yBACR,gFACA,wBACF;CAEF,IAAI,MAAM,cAAc,KAAA,GAAW;EACjC,wBAAwB,MAAM,WAAW,sBAAsB;EAC/D,IAAK,MAAM,YAAwB,MAAM,QACvC,MAAM,IAAI,yBACR,sDACA,sBACF;CAEJ;CACA,IAAI,MAAM,WAAW,KAAA,GAAW,wBAAwB,MAAM,QAAQ,mBAAmB;CACzF,IAAI,MAAM,eAAe,KAAA,GACvB,wBAAwB,MAAM,YAAY,uBAAuB;CAInE,IAAI,IAAI,kBAAkB,KAAA,GAAW;EACnC,MAAM,KAAK,IAAI;EACf,IAAI,OAAO,QAAQ,OAAO,OAAO,UAC/B,MAAM,IAAI,yBAAyB,mCAAmC,eAAe;EAEvF,MAAM,QAAQ;EACd,aAAa,MAAM,OAAO,qBAAqB;EAC/C,aAAa,MAAM,eAAe,6BAA6B;EAC/D,mBAAmB,MAAM,YAAY,0BAA0B;EAC/D,IAAI,OAAO,MAAM,aAAa,WAC5B,MAAM,IAAI,yBACR,0CACA,wBACF;CAEJ;CAGA,MAAM,MAAM,IAAI;CAChB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,6BAA6B,SAAS;CAE3E,MAAM,SAAS;CACf,IAAI,OAAO,gBAAgB,KAAA,GACzB,mBAAmB,OAAO,aAAa,qBAAqB;CAC9D,IAAI,OAAO,iBAAiB,KAAA,GAC1B,mBAAmB,OAAO,cAAc,sBAAsB;CAChE,MAAM,MAAM,OAAO;CACnB,IAAI,QAAQ,QAAQ,OAAO,QAAQ,UACjC,MAAM,IAAI,yBAAyB,iCAAiC,aAAa;CAEnF,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,GAA8B,GAChE,mBAAmB,GAAG,eAAe,GAAG;CAG1C,IAAI,OAAO,aAAa,KAAA,GAAW;EACjC,MAAM,IAAI,OAAO;EACjB,IAAI,MAAM,QAAQ,OAAO,MAAM,UAC7B,MAAM,IAAI,yBAAyB,sCAAsC,kBAAkB;EAE7F,MAAM,KAAK;EACX,mBAAmB,GAAG,OAAO,wBAAwB;EACrD,IAAI,OAAO,GAAG,UAAU,WACtB,MAAM,IAAI,yBACR,4CACA,wBACF;CAEJ;CAGA,IAAI,OAAO,gBAAgB,KAAA,GACzB,oBAAoB,OAAO,aAAa,qBAAqB;CAI/D,IACE,IAAI,iBAAiB,KAAA,MACpB,OAAO,IAAI,iBAAiB,YAC3B,CAAC,gBAAgB,SAAS,IAAI,YAA4B,IAE5D,MAAM,IAAI,yBACR,+BAA+B,gBAAgB,KAAK,IAAI,KACxD,cACF;CAEF,IAAI,IAAI,gBAAgB,KAAA,GAAW;EACjC,aAAa,IAAI,aAAa,aAAa;EAC3C,IAAI,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,WACzD,MAAM,IAAI,yBACR,mDACA,aACF;CAEJ;CAEA,IACE,OAAO,IAAI,oBAAoB,YAC/B,CAAC,kBAAkB,SAAS,IAAI,eAAqC,GAErE,MAAM,IAAI,yBACR,kCAAkC,kBAAkB,KAAK,IAAI,KAC7D,iBACF;CAEF,IAAI,IAAI,0BAA0B,KAAA,GAAW;EAC3C,aAAa,IAAI,uBAAuB,uBAAuB;EAC/D,IACE,IAAI,oBAAoB,YACxB,IAAI,oBAAoB,eACxB,IAAI,oBAAoB,cAExB,MAAM,IAAI,yBACR,mFACA,uBACF;CAEJ;CAEA,IAAI,IAAI,iBAAiB,KAAA,GACvB,IAAI;EACF,MAAM,UAAU,yBAAyB,IAAI,YAAY;EACzD,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,IAAI,OACvD,MAAM,IAAI,yBACR,uBAAuB,QAAQ,MAAM,0BAA0B,IAAI,MAAM,IACzE,oBACF;EAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,IAAI,YACjE,MAAM,IAAI,yBACR,4BAA4B,QAAQ,WAAW,+BAA+B,IAAI,WAAW,IAC7F,yBACF;CAEJ,SAAS,OAAO;EACd,IAAI,iBAAiB,0BAA0B,MAAM;EACrD,IAAI,iBAAiB,OACnB,MAAM,IAAI,yBAAyB,MAAM,SAAS,cAAc;EAElE,MAAM;CACR;CAGF,aAAa,IAAI,YAAY,YAAY;CAGzC,IAAI,OAAO,IAAI,aAAa,YAAY,CAAC,WAAW,SAAS,IAAI,QAAuB,GACtF,MAAM,IAAI,yBACR,2BAA2B,WAAW,KAAK,IAAI,EAAE,QAAQ,OAAO,IAAI,QAAQ,KAC5E,UACF;CAGF,OAAO;AACT;AAEA,SAAS,aAAa,SAAkB,YAA2B;CACjE,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,oCAAoC,gBAAgB;CAEzF,MAAM,QAAQ;CACd,IAAI,MAAM,SAAS,cAAc,MAAM,SAAS,eAAe,MAAM,SAAS,cAC5E,MAAM,IAAI,yBACR,kEACA,qBACF;CAMF,IAAI,MAAM,SAAS,cAAc;EAC/B,IAAI,MAAM,QAAQ,MAChB,MAAM,IAAI,yBACR,8CACA,oBACF;EAEF,IAAI,YAAY,MACd,MAAM,IAAI,yBAAyB,+CAA+C,SAAS;EAE7F;CACF;CACA,wBAAwB,SAAS,SAAS;CAC1C,wBAAwB,MAAM,KAAK,oBAAoB;CACvD,IAAI,MAAM,QAAQ,SAChB,MAAM,IAAI,yBACR,yCACA,oBACF;AAEJ;;AAGA,SAAgB,YAAY,OAAoC;CAC9D,IAAI;EACF,kBAAkB,KAAK;EACvB,OAAO;CACT,QAAQ;EACN,OAAO;CACT;AACF;;AAGA,SAAgB,mBACd,OACiF;CACjF,IAAI;EACF,OAAO;GAAE,IAAI;GAAM,OAAO,kBAAkB,KAAK;EAAE;CACrD,SAAS,GAAG;EACV,IAAI,aAAa,0BAA0B,OAAO;GAAE,IAAI;GAAO,OAAO;EAAE;EACxE,MAAM;CACR;AACF;;AAGA,SAAgB,mBAAmB,QAA8B;CAC/D,MAAM,OAAO,KAAK,UAAU,MAAM;CAClC,OAAO,kBAAkB,KAAK,MAAM,IAAI,CAAC;AAC3C;AAIA,SAAS,aAAa,OAAgB,MAAoB;CACxD,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,yBAAyB,6BAA6B,IAAI;AAExE;AAEA,SAAS,mBAAmB,OAAgB,MAAoB;CAC9D,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GACrD,MAAM,IAAI,yBAAyB,0BAA0B,IAAI;AAErE;AAEA,SAAS,wBAAwB,OAAgB,MAAoB;CACnE,mBAAmB,OAAO,IAAI;CAC9B,IAAK,QAAmB,GACtB,MAAM,IAAI,yBAAyB,gCAAgC,IAAI;AAE3E;AAEA,SAAS,oBAAoB,OAAgB,MAAoB;CAC/D,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,yBAAyB,iCAAiC,IAAI;CAE1E,MAAM,MAAM;CAEZ,MAAM,WAAW,IAAI;CACrB,IAAI,aAAa,QAAQ,OAAO,aAAa,UAC3C,MAAM,IAAI,yBAAyB,8BAA8B,GAAG,KAAK,UAAU;CAErF,KAAK,MAAM,CAAC,SAAS,SAAS,OAAO,QAAQ,QAAmC,GAAG;EACjF,IAAI,SAAS,QAAQ,OAAO,SAAS,UACnC,MAAM,IAAI,yBACR,yDACA,GAAG,KAAK,YAAY,SACtB;EAEF,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,IAA+B,GACvE,mBAAmB,OAAO,GAAG,KAAK,YAAY,QAAQ,GAAG,KAAK;CAElE;CAEA,MAAM,aAAa,IAAI;CACvB,IAAI,eAAe,QAAQ,OAAO,eAAe,UAC/C,MAAM,IAAI,yBAAyB,gCAAgC,GAAG,KAAK,YAAY;CAEzF,KAAK,MAAM,CAAC,KAAK,SAAS,OAAO,QAAQ,UAAqC,GAC5E,mBAAmB,MAAM,GAAG,KAAK,cAAc,KAAK;CAGtD,mBAAmB,IAAI,WAAW,GAAG,KAAK,WAAW;CAErD,IAAI,IAAI,iBAAiB,KAAA,GAAW;EAClC,IAAI,CAAC,MAAM,QAAQ,IAAI,YAAY,GACjC,MAAM,IAAI,yBACR,4CACA,GAAG,KAAK,cACV;EAEF,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,aAAa,QAAQ,KAAK;GAChD,MAAM,KAAK,IAAI,aAAa;GAC5B,IAAI,OAAO,OAAO,YAAY,GAAG,WAAW,GAC1C,MAAM,IAAI,yBACR,iDACA,GAAG,KAAK,gBAAgB,EAAE,EAC5B;EAEJ;CACF;CAEA,IAAI,IAAI,UAAU,KAAA,KAAa,OAAO,IAAI,UAAU,UAClD,MAAM,IAAI,yBAAyB,0BAA0B,GAAG,KAAK,OAAO;AAEhF;;;;;;;AAQA,SAAgB,iBAAiB,OAAwB;CACvD,IAAI,MAAM,WAAW,KAAK,MAAM,KAAK,MAAM,OAAO,OAAO;CAEzD,MAAM,WAAW,MAAM,YAAY,GAAG;CACtC,IAAI,WAAW,GAAG;EAChB,MAAM,OAAO,MAAM,MAAM,GAAG,QAAQ;EACpC,MAAM,QAAQ,MAAM,MAAM,WAAW,CAAC;EACtC,IAAI,CAAC,KAAK,SAAS,GAAG,KAAK,gDAAgD,KAAK,KAAK,GACnF,OAAO;CAEX;CAEA,MAAM,UAAU,MAAM,MAAM,4BAA4B;CACxD,IAAI,WAAW,kBAAkB,QAAQ,IAAK,QAAQ,IAAK,QAAQ,EAAG,GAAG,OAAO;CAEhF,MAAM,cAAc,MAAM,MAAM,0BAA0B;CAC1D,IAAI,eAAe,kBAAkB,YAAY,IAAK,YAAY,IAAK,YAAY,EAAG,GACpF,OAAO;CAGT,MAAM,aAAa,MAAM,MAAM,mBAAmB;CAClD,IAAI,cAAc,kBAAkB,KAAA,GAAW,WAAW,IAAK,WAAW,EAAG,GAAG,OAAO;CAEvF,OAAO,qDAAqD,KAAK,KAAK;AACxE;AAEA,SAAS,kBAAkB,MAA0B,OAAe,KAAsB;CACxF,MAAM,cAAc,OAAO,KAAK;CAChC,MAAM,YAAY,OAAO,GAAG;CAC5B,IAAI,CAAC,OAAO,UAAU,WAAW,KAAK,cAAc,KAAK,cAAc,IAAI,OAAO;CAElF,MAAM,aAAa,SAAS,KAAA,IAAY,KAAA,IAAY,OAAO,IAAI;CAI/D,MAAM,cAAc;EAAC;EAFnB,eAAe,KAAA,KACd,aAAa,MAAM,MAAM,aAAa,QAAQ,KAAK,aAAa,QAAQ,KACvC,KAAK;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;EAAI;CAAE;CACnF,OAAO,OAAO,UAAU,SAAS,KAAK,aAAa,KAAK,aAAa,YAAY,cAAc;AACjG"}
@@ -1,7 +1,7 @@
1
1
  import { c as ValidationError } from "./errors-DEE6u6ot.js";
2
- import { r as AgentProfileCell } from "./agent-profile-cell-CTOZJUuE.js";
3
2
  import { p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
4
- import { o as FailureClass } from "./schema-Bjgdsn73.js";
3
+ import { r as AgentProfileCell } from "./agent-profile-cell-CTOZJUuE.js";
4
+ import { o as FailureClass } from "./schema-DID1Cqct.js";
5
5
  //#region src/run-record.d.ts
6
6
  /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
7
7
  * combined train+test pool that the optimizer is allowed to read. */
@@ -244,4 +244,4 @@ declare function roundTripRunRecord(record: RunRecord): RunRecord;
244
244
  declare function modelHasSnapshot(model: string): boolean;
245
245
  //#endregion
246
246
  export { validateRunRecord as _, RunRecord as a, RunTaskFailure as c, UNKNOWN_MODEL as d, isRunRecord as f, runTaskScore as g, roundTripRunRecord as h, RunOutcome as i, RunTerminalOutcome as l, parseRunRecordSafe as m, RunCostProvenance as n, RunRecordValidationError as o, modelHasSnapshot as p, RunJudgeMetadata as r, RunSplitTag as s, JudgeScoresRecord as t, RunTokenUsage as u };
247
- //# sourceMappingURL=run-record-VVy4T9OW.d.ts.map
247
+ //# sourceMappingURL=run-record-DQjRcYwA.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"run-record-VVy4T9OW.d.ts","names":[],"sources":["../src/run-record.ts"],"mappings":";;;;;;;KAsCY;;;;;;;KAQA;;cAGC;UAEI;EACf;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;;KAIU,oBAAoB;UAEf;EACf;EACA;;;EAGA;;;EAGA;;;;;;;;;;;;;;;;;;;;;UAsBe;;EAEf,UAAU,eAAe;;EAEzB,YAAY;;;EAGZ;;;;EAIA;;;EAGA;;UAGe;;;EAGf;;;EAGA;;;;EAIA,KAAK;;;;;;EAML,cAAc;;;;;;;EAOd;IAAa;IAAe;IAAgB;;;;;;;;;;;;;;;;;;;;;;UAsB7B;;EAEf;;;EAGA;;;EAGA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,gBAAgB;;EAEhB,YAAY;;EAEZ,iBAAiB;;;EAGjB;;EAEA,gBAAgB;;EAEhB,SAAS;;;;;EAKT,eAAe;;;EAGf;;EAEA,UAAU;;;;;EAKV;;;;;;;;EAQA,eAAe;;;;;;;;;KAUL;EACN;EAA0B;;EAC1B;EAAyB;;EAEzB,cAAc,QAAQ;EACtB;;;;;;;;;;iBAWU,aAAa,QAAQ;cAmCxB,iCAAiC;WACnC;EACT,YAAY,iBAAiB;;;;;;;iBAWf,kBAAkB,iBAAiB;;iBAgPnC,YAAY,iBAAiB,SAAS;;iBAUtC,mBACd;EACG;EAAU,OAAO;;EAAgB;EAAW,OAAO;;;iBAUxC,mBAAmB,QAAQ,YAAY;;;;;;;iBAuFvC,iBAAiB"}
1
+ {"version":3,"file":"run-record-DQjRcYwA.d.ts","names":[],"sources":["../src/run-record.ts"],"mappings":";;;;;;;KAsCY;;;;;;;KAQA;;cAGC;UAEI;EACf;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;;KAIU,oBAAoB;UAEf;EACf;EACA;;;EAGA;;;EAGA;;;;;;;;;;;;;;;;;;;;;UAsBe;;EAEf,UAAU,eAAe;;EAEzB,YAAY;;;EAGZ;;;;EAIA;;;EAGA;;UAGe;;;EAGf;;;EAGA;;;;EAIA,KAAK;;;;;;EAML,cAAc;;;;;;;EAOd;IAAa;IAAe;IAAgB;;;;;;;;;;;;;;;;;;;;;;UAsB7B;;EAEf;;;EAGA;;;EAGA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,gBAAgB;;EAEhB,YAAY;;EAEZ,iBAAiB;;;EAGjB;;EAEA,gBAAgB;;EAEhB,SAAS;;;;;EAKT,eAAe;;;EAGf;;EAEA,UAAU;;;;;EAKV;;;;;;;;EAQA,eAAe;;;;;;;;;KAUL;EACN;EAA0B;;EAC1B;EAAyB;;EAEzB,cAAc,QAAQ;EACtB;;;;;;;;;;iBAWU,aAAa,QAAQ;cAmCxB,iCAAiC;WACnC;EACT,YAAY,iBAAiB;;;;;;;iBAWf,kBAAkB,iBAAiB;;iBAgPnC,YAAY,iBAAiB,SAAS;;iBAUtC,mBACd;EACG;EAAU,OAAO;;EAAgB;EAAW,OAAO;;;iBAUxC,mBAAmB,QAAQ,YAAY;;;;;;;iBAuFvC,iBAAiB"}
@@ -11,6 +11,10 @@
11
11
  * entities that OTEL leaves as free-form attributes.
12
12
  */
13
13
  const TRACE_SCHEMA_VERSION = "1.0.0";
14
+ /**
15
+ * The failure taxonomy. `FailureClass` derives from this array, so the type
16
+ * and the runtime list cannot name different sets.
17
+ */
14
18
  const FAILURE_CLASSES = [
15
19
  "success",
16
20
  "reasoning_error",
@@ -60,4 +64,4 @@ function isJudgeSpan(s) {
60
64
  //#endregion
61
65
  export { isToolSpan as a, isLlmSpan as i, TRACE_SCHEMA_VERSION as n, isJudgeSpan as r, FAILURE_CLASSES as t };
62
66
 
63
- //# sourceMappingURL=schema-k6ZBftVv.js.map
67
+ //# sourceMappingURL=schema-CdIX2aHu.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"schema-k6ZBftVv.js","names":[],"sources":["../src/trace/schema.ts"],"sourcesContent":["/**\n * TraceSchema v1 — the canonical data model for agent-eval.\n *\n * Every score, every failure class, every pipeline in the framework is\n * a view over this data. Shape it once, live with it.\n *\n * Wire-compatible with OpenTelemetry span semantics (see trace/otel.ts)\n * but extended with agent-specific span kinds (llm, tool, retrieval,\n * judge, sandbox) and first-class BudgetLedger / Artifact / JudgeVerdict\n * entities that OTEL leaves as free-form attributes.\n */\n\nexport const TRACE_SCHEMA_VERSION = '1.0.0'\n\n// ── Run ──────────────────────────────────────────────────────────────\n\nexport type RunStatus = 'running' | 'completed' | 'failed' | 'aborted'\n\nexport interface BudgetSpec {\n tokens?: number\n wallMs?: number\n calls?: number\n usd?: number\n}\n\nexport interface RunOutcome {\n score?: number\n pass?: boolean\n failureClass?: FailureClass\n notes?: string\n}\n\n/**\n * Layer — optional classification in a nested build workflow.\n * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).\n * `app-build`: sandbox harness that compiled + tested the generated scaffold.\n * `app-runtime`: a run of the generated agent against a domain scenario.\n * `meta`: any meta-eval (judge replay, correlation analysis).\n */\nexport type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom'\n\nexport interface Run {\n runId: string\n /**\n * Stable identifier of the scenario being executed.\n *\n * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts\n * input WITHOUT this field, substituting a sensible default\n * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no\n * curated scenario to anchor to (runtime / operator / meta-eval runs). This\n * keeps the persisted shape unambiguous for downstream filters + aggregations\n * while removing the boilerplate of inventing placeholder ids at the call site.\n */\n scenarioId: string\n variantId?: string\n datasetVersion?: string\n /** Git SHA of agent code at run time. */\n codeSha?: string\n /** Hash of the prompt template + any system prompt. */\n promptSha?: string\n /** Model id + date + system-prompt hash, concatenated. */\n modelFingerprint?: string\n seed?: number\n /** Arbitrary environment markers (shell, docker version, tz). */\n envFingerprint?: Record<string, string>\n /** Version of the redaction rules applied to this run. */\n redactionVersion?: string\n /** Parent run in a nested build workflow. A builder run's children are\n * app-build runs; those children are app-runtime runs. */\n parentRunId?: string\n /** Stable project identifier — groups runs across chats + sessions. */\n projectId?: string\n /** Chat/conversation identifier within a project. */\n chatId?: string\n /** Layer classification — hint for aggregation; not enforced. */\n layer?: RunLayer\n startedAt: number\n endedAt?: number\n status: RunStatus\n outcome?: RunOutcome\n budget?: BudgetSpec\n /** Free-form labels for downstream grouping. */\n tags?: Record<string, string>\n}\n\n// ── Spans (hierarchical work units) ──────────────────────────────────\n\nexport type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom'\n\nexport type SpanStatus = 'ok' | 'error'\n\nexport interface SpanBase {\n spanId: string\n parentSpanId?: string\n runId: string\n kind: SpanKind\n name: string\n startedAt: number\n endedAt?: number\n status?: SpanStatus\n error?: string\n /** Anything not covered by typed fields. Kept deliberately free-form. */\n attributes?: Record<string, unknown>\n}\n\nexport interface Message {\n role: 'system' | 'user' | 'assistant' | 'tool'\n content: string\n tokens?: number\n /** Multi-modal content descriptors; blobs themselves live in Artifacts. */\n images?: Array<{ artifactId?: string; url?: string; mime?: string }>\n}\n\nexport interface LlmSpan extends SpanBase {\n kind: 'llm'\n model: string\n messages: Message[]\n output?: string\n inputTokens?: number\n /** All generated tokens, including the reasoning subset when present. */\n outputTokens?: number\n cachedTokens?: number\n cacheWriteTokens?: number\n /** Reasoning-token subset of `outputTokens`. */\n reasoningTokens?: number\n costUsd?: number\n finishReason?: string\n}\n\nexport interface ToolSpan extends SpanBase {\n kind: 'tool'\n toolName: string\n args: unknown\n /** False when the source observed the call but did not capture its arguments. */\n argsCaptured?: boolean\n result?: unknown\n latencyMs?: number\n}\n\nexport interface RetrievalSpan extends SpanBase {\n kind: 'retrieval'\n query: string\n hits: Array<{ docId: string; score: number; content?: string }>\n}\n\nexport interface JudgeSpan extends SpanBase {\n kind: 'judge'\n judgeId: string\n /** Span this judgment applies to. */\n targetSpanId: string\n dimension: string\n /** Numeric score (free-range; interpretation up to the judge). */\n score: number\n rationale?: string\n evidence?: string\n}\n\nexport interface SandboxSpan extends SpanBase {\n kind: 'sandbox'\n image?: string\n command?: string\n exitCode?: number\n testsTotal?: number\n testsPassed?: number\n stdoutHash?: string\n stderrHash?: string\n /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */\n wallMs?: number\n}\n\nexport interface GenericSpan extends SpanBase {\n kind: 'agent' | 'custom'\n}\n\nexport type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan\n\n// ── Events (point-in-time occurrences within a span) ─────────────────\n\nexport type EventKind =\n | 'log'\n | 'error'\n | 'budget_decrement'\n | 'budget_breach'\n | 'state_mutation'\n | 'policy_violation'\n | 'redaction_applied'\n | 'custom'\n\nexport interface TraceEvent {\n eventId: string\n runId: string\n spanId?: string\n kind: EventKind\n timestamp: number\n payload: Record<string, unknown>\n}\n\n// ── Budget ledger (running token/wall/call/$ accounting) ─────────────\n\nexport interface BudgetLedgerEntry {\n runId: string\n dimension: keyof BudgetSpec\n limit: number\n consumed: number\n remaining: number\n timestamp: number\n breached: boolean\n /** Span that triggered this entry, if any. */\n spanId?: string\n}\n\n// ── Artifacts (blobs addressed by hash) ──────────────────────────────\n\nexport interface Artifact {\n artifactId: string\n runId: string\n spanId?: string\n contentType: string\n sizeBytes: number\n /** sha256 in hex. */\n hash: string\n /** External storage URL (R2, S3, filesystem path). */\n storageUrl?: string\n /** Inline content for small blobs — keep under ~64KB. */\n inlineContent?: string\n}\n\n// ── Failure taxonomy ─────────────────────────────────────────────────\n\nexport type FailureClass =\n | 'success'\n | 'reasoning_error'\n | 'tool_selection_error'\n | 'tool_argument_error'\n | 'tool_recovery_failure'\n | 'hallucination'\n | 'instruction_following'\n | 'safety_refusal_miss'\n | 'policy_violation'\n | 'budget_exceeded'\n | 'format_drift'\n | 'permission_escalation'\n | 'pii_leak'\n | 'cost_overrun'\n | 'timeout'\n | 'sandbox_failure'\n | 'missing_user_data'\n | 'missing_domain_data'\n | 'missing_codebase_context'\n | 'missing_runtime_context'\n | 'missing_credentials'\n | 'missing_integration_connection'\n | 'missing_integration_scope'\n | 'integration_approval_required'\n | 'integration_auth_expired'\n | 'integration_provider_failure'\n | 'bad_integration_manifest'\n | 'unsafe_integration_write_denied'\n | 'stale_external_data'\n | 'bad_retrieval'\n | 'insufficient_evidence'\n | 'contradictory_evidence'\n | 'ambiguous_user_intent'\n | 'knowledge_readiness_blocked'\n | 'unknown'\n\nexport const FAILURE_CLASSES: readonly FailureClass[] = [\n 'success',\n 'reasoning_error',\n 'tool_selection_error',\n 'tool_argument_error',\n 'tool_recovery_failure',\n 'hallucination',\n 'instruction_following',\n 'safety_refusal_miss',\n 'policy_violation',\n 'budget_exceeded',\n 'format_drift',\n 'permission_escalation',\n 'pii_leak',\n 'cost_overrun',\n 'timeout',\n 'sandbox_failure',\n 'missing_user_data',\n 'missing_domain_data',\n 'missing_codebase_context',\n 'missing_runtime_context',\n 'missing_credentials',\n 'missing_integration_connection',\n 'missing_integration_scope',\n 'integration_approval_required',\n 'integration_auth_expired',\n 'integration_provider_failure',\n 'bad_integration_manifest',\n 'unsafe_integration_write_denied',\n 'stale_external_data',\n 'bad_retrieval',\n 'insufficient_evidence',\n 'contradictory_evidence',\n 'ambiguous_user_intent',\n 'knowledge_readiness_blocked',\n 'unknown',\n] as const\n\n// ── Helpers ──────────────────────────────────────────────────────────\n\nexport function isLlmSpan(s: Span): s is LlmSpan {\n return s.kind === 'llm'\n}\nexport function isToolSpan(s: Span): s is ToolSpan {\n return s.kind === 'tool'\n}\nexport function isJudgeSpan(s: Span): s is JudgeSpan {\n return s.kind === 'judge'\n}\n"],"mappings":";;;;;;;;;;;;AAYA,MAAa,uBAAuB;AA8PpC,MAAa,kBAA2C;CACtD;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAIA,SAAgB,UAAU,GAAuB;CAC/C,OAAO,EAAE,SAAS;AACpB;AACA,SAAgB,WAAW,GAAwB;CACjD,OAAO,EAAE,SAAS;AACpB;AACA,SAAgB,YAAY,GAAyB;CACnD,OAAO,EAAE,SAAS;AACpB"}
1
+ {"version":3,"file":"schema-CdIX2aHu.js","names":[],"sources":["../src/trace/schema.ts"],"sourcesContent":["/**\n * TraceSchema v1 — the canonical data model for agent-eval.\n *\n * Every score, every failure class, every pipeline in the framework is\n * a view over this data. Shape it once, live with it.\n *\n * Wire-compatible with OpenTelemetry span semantics (see trace/otel.ts)\n * but extended with agent-specific span kinds (llm, tool, retrieval,\n * judge, sandbox) and first-class BudgetLedger / Artifact / JudgeVerdict\n * entities that OTEL leaves as free-form attributes.\n */\n\nexport const TRACE_SCHEMA_VERSION = '1.0.0'\n\n// ── Run ──────────────────────────────────────────────────────────────\n\nexport type RunStatus = 'running' | 'completed' | 'failed' | 'aborted'\n\nexport interface BudgetSpec {\n tokens?: number\n wallMs?: number\n calls?: number\n usd?: number\n}\n\nexport interface RunOutcome {\n score?: number\n pass?: boolean\n failureClass?: FailureClass\n notes?: string\n}\n\n/**\n * Layer — optional classification in a nested build workflow.\n * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).\n * `app-build`: sandbox harness that compiled + tested the generated scaffold.\n * `app-runtime`: a run of the generated agent against a domain scenario.\n * `meta`: any meta-eval (judge replay, correlation analysis).\n */\nexport type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom'\n\nexport interface Run {\n runId: string\n /**\n * Stable identifier of the scenario being executed.\n *\n * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts\n * input WITHOUT this field, substituting a sensible default\n * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no\n * curated scenario to anchor to (runtime / operator / meta-eval runs). This\n * keeps the persisted shape unambiguous for downstream filters + aggregations\n * while removing the boilerplate of inventing placeholder ids at the call site.\n */\n scenarioId: string\n variantId?: string\n datasetVersion?: string\n /** Git SHA of agent code at run time. */\n codeSha?: string\n /** Hash of the prompt template + any system prompt. */\n promptSha?: string\n /** Model id + date + system-prompt hash, concatenated. */\n modelFingerprint?: string\n seed?: number\n /** Arbitrary environment markers (shell, docker version, tz). */\n envFingerprint?: Record<string, string>\n /** Version of the redaction rules applied to this run. */\n redactionVersion?: string\n /** Parent run in a nested build workflow. A builder run's children are\n * app-build runs; those children are app-runtime runs. */\n parentRunId?: string\n /** Stable project identifier — groups runs across chats + sessions. */\n projectId?: string\n /** Chat/conversation identifier within a project. */\n chatId?: string\n /** Layer classification — hint for aggregation; not enforced. */\n layer?: RunLayer\n startedAt: number\n endedAt?: number\n status: RunStatus\n outcome?: RunOutcome\n budget?: BudgetSpec\n /** Free-form labels for downstream grouping. */\n tags?: Record<string, string>\n}\n\n// ── Spans (hierarchical work units) ──────────────────────────────────\n\nexport type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom'\n\nexport type SpanStatus = 'ok' | 'error'\n\nexport interface SpanBase {\n spanId: string\n parentSpanId?: string\n runId: string\n kind: SpanKind\n name: string\n startedAt: number\n endedAt?: number\n status?: SpanStatus\n error?: string\n /** Anything not covered by typed fields. Kept deliberately free-form. */\n attributes?: Record<string, unknown>\n}\n\nexport interface Message {\n role: 'system' | 'user' | 'assistant' | 'tool'\n content: string\n tokens?: number\n /** Multi-modal content descriptors; blobs themselves live in Artifacts. */\n images?: Array<{ artifactId?: string; url?: string; mime?: string }>\n}\n\nexport interface LlmSpan extends SpanBase {\n kind: 'llm'\n model: string\n messages: Message[]\n output?: string\n inputTokens?: number\n /** All generated tokens, including the reasoning subset when present. */\n outputTokens?: number\n cachedTokens?: number\n cacheWriteTokens?: number\n /** Reasoning-token subset of `outputTokens`. */\n reasoningTokens?: number\n costUsd?: number\n finishReason?: string\n}\n\nexport interface ToolSpan extends SpanBase {\n kind: 'tool'\n toolName: string\n args: unknown\n /** False when the source observed the call but did not capture its arguments. */\n argsCaptured?: boolean\n result?: unknown\n latencyMs?: number\n}\n\nexport interface RetrievalSpan extends SpanBase {\n kind: 'retrieval'\n query: string\n hits: Array<{ docId: string; score: number; content?: string }>\n}\n\nexport interface JudgeSpan extends SpanBase {\n kind: 'judge'\n judgeId: string\n /** Span this judgment applies to. */\n targetSpanId: string\n dimension: string\n /** Numeric score (free-range; interpretation up to the judge). */\n score: number\n rationale?: string\n evidence?: string\n}\n\nexport interface SandboxSpan extends SpanBase {\n kind: 'sandbox'\n image?: string\n command?: string\n exitCode?: number\n testsTotal?: number\n testsPassed?: number\n stdoutHash?: string\n stderrHash?: string\n /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */\n wallMs?: number\n}\n\nexport interface GenericSpan extends SpanBase {\n kind: 'agent' | 'custom'\n}\n\nexport type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan\n\n// ── Events (point-in-time occurrences within a span) ─────────────────\n\nexport type EventKind =\n | 'log'\n | 'error'\n | 'budget_decrement'\n | 'budget_breach'\n | 'state_mutation'\n | 'policy_violation'\n | 'redaction_applied'\n | 'custom'\n\nexport interface TraceEvent {\n eventId: string\n runId: string\n spanId?: string\n kind: EventKind\n timestamp: number\n payload: Record<string, unknown>\n}\n\n// ── Budget ledger (running token/wall/call/$ accounting) ─────────────\n\nexport interface BudgetLedgerEntry {\n runId: string\n dimension: keyof BudgetSpec\n limit: number\n consumed: number\n remaining: number\n timestamp: number\n breached: boolean\n /** Span that triggered this entry, if any. */\n spanId?: string\n}\n\n// ── Artifacts (blobs addressed by hash) ──────────────────────────────\n\nexport interface Artifact {\n artifactId: string\n runId: string\n spanId?: string\n contentType: string\n sizeBytes: number\n /** sha256 in hex. */\n hash: string\n /** External storage URL (R2, S3, filesystem path). */\n storageUrl?: string\n /** Inline content for small blobs — keep under ~64KB. */\n inlineContent?: string\n}\n\n// ── Failure taxonomy ─────────────────────────────────────────────────\n\n/**\n * The failure taxonomy. `FailureClass` derives from this array, so the type\n * and the runtime list cannot name different sets.\n */\nexport const FAILURE_CLASSES = [\n 'success',\n 'reasoning_error',\n 'tool_selection_error',\n 'tool_argument_error',\n 'tool_recovery_failure',\n 'hallucination',\n 'instruction_following',\n 'safety_refusal_miss',\n 'policy_violation',\n 'budget_exceeded',\n 'format_drift',\n 'permission_escalation',\n 'pii_leak',\n 'cost_overrun',\n 'timeout',\n 'sandbox_failure',\n 'missing_user_data',\n 'missing_domain_data',\n 'missing_codebase_context',\n 'missing_runtime_context',\n 'missing_credentials',\n 'missing_integration_connection',\n 'missing_integration_scope',\n 'integration_approval_required',\n 'integration_auth_expired',\n 'integration_provider_failure',\n 'bad_integration_manifest',\n 'unsafe_integration_write_denied',\n 'stale_external_data',\n 'bad_retrieval',\n 'insufficient_evidence',\n 'contradictory_evidence',\n 'ambiguous_user_intent',\n 'knowledge_readiness_blocked',\n 'unknown',\n] as const\n\nexport type FailureClass = (typeof FAILURE_CLASSES)[number]\n\n// ── Helpers ──────────────────────────────────────────────────────────\n\nexport function isLlmSpan(s: Span): s is LlmSpan {\n return s.kind === 'llm'\n}\nexport function isToolSpan(s: Span): s is ToolSpan {\n return s.kind === 'tool'\n}\nexport function isJudgeSpan(s: Span): s is JudgeSpan {\n return s.kind === 'judge'\n}\n"],"mappings":";;;;;;;;;;;;AAYA,MAAa,uBAAuB;;;;;AA6NpC,MAAa,kBAAkB;CAC7B;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAMA,SAAgB,UAAU,GAAuB;CAC/C,OAAO,EAAE,SAAS;AACpB;AACA,SAAgB,WAAW,GAAwB;CACjD,OAAO,EAAE,SAAS;AACpB;AACA,SAAgB,YAAY,GAAyB;CACnD,OAAO,EAAE,SAAS;AACpB"}
@@ -194,11 +194,15 @@ interface Artifact {
194
194
  /** Inline content for small blobs — keep under ~64KB. */
195
195
  inlineContent?: string;
196
196
  }
197
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
198
- declare const FAILURE_CLASSES: readonly FailureClass[];
197
+ /**
198
+ * The failure taxonomy. `FailureClass` derives from this array, so the type
199
+ * and the runtime list cannot name different sets.
200
+ */
201
+ declare const FAILURE_CLASSES: readonly ['success', 'reasoning_error', 'tool_selection_error', 'tool_argument_error', 'tool_recovery_failure', 'hallucination', 'instruction_following', 'safety_refusal_miss', 'policy_violation', 'budget_exceeded', 'format_drift', 'permission_escalation', 'pii_leak', 'cost_overrun', 'timeout', 'sandbox_failure', 'missing_user_data', 'missing_domain_data', 'missing_codebase_context', 'missing_runtime_context', 'missing_credentials', 'missing_integration_connection', 'missing_integration_scope', 'integration_approval_required', 'integration_auth_expired', 'integration_provider_failure', 'bad_integration_manifest', 'unsafe_integration_write_denied', 'stale_external_data', 'bad_retrieval', 'insufficient_evidence', 'contradictory_evidence', 'ambiguous_user_intent', 'knowledge_readiness_blocked', 'unknown'];
202
+ type FailureClass = (typeof FAILURE_CLASSES)[number];
199
203
  declare function isLlmSpan(s: Span): s is LlmSpan;
200
204
  declare function isToolSpan(s: Span): s is ToolSpan;
201
205
  declare function isJudgeSpan(s: Span): s is JudgeSpan;
202
206
  //#endregion
203
207
  export { TraceEvent as C, isToolSpan as E, ToolSpan as S, isLlmSpan as T, Span as _, FAILURE_CLASSES as a, SpanStatus as b, JudgeSpan as c, RetrievalSpan as d, Run as f, SandboxSpan as g, RunStatus as h, EventKind as i, LlmSpan as l, RunOutcome as m, BudgetLedgerEntry as n, FailureClass as o, RunLayer as p, BudgetSpec as r, GenericSpan as s, Artifact as t, Message as u, SpanBase as v, isJudgeSpan as w, TRACE_SCHEMA_VERSION as x, SpanKind as y };
204
- //# sourceMappingURL=schema-Bjgdsn73.d.ts.map
208
+ //# sourceMappingURL=schema-DID1Cqct.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"schema-Bjgdsn73.d.ts","names":[],"sources":["../src/trace/schema.ts"],"mappings":";;;;;;;;;;;;cAYa;KAID;UAEK;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,eAAe;EACf;;;;;;;;;KAUU;UAEK;EACf;;;;;;;;;;;EAWA;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA;;EAEA,iBAAiB;;EAEjB;;;EAGA;;EAEA;;EAEA;;EAEA,QAAQ;EACR;EACA;EACA,QAAQ;EACR,UAAU;EACV,SAAS;;EAET,OAAO;;KAKG;KAEA;UAEK;EACf;EACA;EACA;EACA,MAAM;EACN;EACA;EACA;EACA,SAAS;EACT;;EAEA,aAAa;;UAGE;EACf;EACA;EACA;;EAEA,SAAS;IAAQ;IAAqB;IAAc;;;UAGrC,gBAAgB;EAC/B;EACA;EACA,UAAU;EACV;EACA;;EAEA;EACA;EACA;;EAEA;EACA;EACA;;UAGe,iBAAiB;EAChC;EACA;EACA;;EAEA;EACA;EACA;;UAGe,sBAAsB;EACrC;EACA;EACA,MAAM;IAAQ;IAAe;IAAe;;;UAG7B,kBAAkB;EACjC;EACA;;EAEA;EACA;;EAEA;EACA;EACA;;UAGe,oBAAoB;EACnC;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;UAGe,oBAAoB;EACnC;;KAGU,OAAO,UAAU,WAAW,gBAAgB,YAAY,cAAc;KAItE;UAUK;EACf;EACA;EACA;EACA,MAAM;EACN;EACA,SAAS;;UAKM;EACf;EACA,iBAAiB;EACjB;EACA;EACA;EACA;EACA;;EAEA;;UAKe;EACf;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;KAKU;cAqCC,0BAA0B;iBAwCvB,UAAU,GAAG,OAAO,KAAK;iBAGzB,WAAW,GAAG,OAAO,KAAK;iBAG1B,YAAY,GAAG,OAAO,KAAK"}
1
+ {"version":3,"file":"schema-DID1Cqct.d.ts","names":[],"sources":["../src/trace/schema.ts"],"mappings":";;;;;;;;;;;;cAYa;KAID;UAEK;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,eAAe;EACf;;;;;;;;;KAUU;UAEK;EACf;;;;;;;;;;;EAWA;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA;;EAEA,iBAAiB;;EAEjB;;;EAGA;;EAEA;;EAEA;;EAEA,QAAQ;EACR;EACA;EACA,QAAQ;EACR,UAAU;EACV,SAAS;;EAET,OAAO;;KAKG;KAEA;UAEK;EACf;EACA;EACA;EACA,MAAM;EACN;EACA;EACA;EACA,SAAS;EACT;;EAEA,aAAa;;UAGE;EACf;EACA;EACA;;EAEA,SAAS;IAAQ;IAAqB;IAAc;;;UAGrC,gBAAgB;EAC/B;EACA;EACA,UAAU;EACV;EACA;;EAEA;EACA;EACA;;EAEA;EACA;EACA;;UAGe,iBAAiB;EAChC;EACA;EACA;;EAEA;EACA;EACA;;UAGe,sBAAsB;EACrC;EACA;EACA,MAAM;IAAQ;IAAe;IAAe;;;UAG7B,kBAAkB;EACjC;EACA;;EAEA;EACA;;EAEA;EACA;EACA;;UAGe,oBAAoB;EACnC;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;UAGe,oBAAoB;EACnC;;KAGU,OAAO,UAAU,WAAW,gBAAgB,YAAY,cAAc;KAItE;UAUK;EACf;EACA;EACA;EACA,MAAM;EACN;EACA,SAAS;;UAKM;EACf;EACA,iBAAiB;EACjB;EACA;EACA;EACA;EACA;;EAEA;;UAKe;EACf;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;;;;;cASW;KAsCD,uBAAuB;iBAInB,UAAU,GAAG,OAAO,KAAK;iBAGzB,WAAW,GAAG,OAAO,KAAK;iBAG1B,YAAY,GAAG,OAAO,KAAK"}
@@ -1,6 +1,6 @@
1
1
  import { i as CostLedger } from "./cost-ledger-B1qx30B4.js";
2
- import { l as Mutex } from "./ledger-core-BOzlRygb.js";
3
- import { t as paidJsonChat } from "./chat-json-call-6g5sJobJ.js";
2
+ import { l as Mutex } from "./ledger-core-PIfjCbKn.js";
3
+ import { t as paidJsonChat } from "./chat-json-call-5Jxna-aV.js";
4
4
  import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
5
5
  import { dirname } from "node:path";
6
6
  //#region src/locked-jsonl-appender.ts
@@ -379,4 +379,4 @@ async function runSemanticConceptJudge(input, options) {
379
379
  //#endregion
380
380
  export { diffFindings as a, defaultIsMaterial as i, runSemanticConceptJudge as n, FindingsStore as r, SEMANTIC_CONCEPT_JUDGE_VERSION as t };
381
381
 
382
- //# sourceMappingURL=semantic-concept-judge-BSkKKHeq.js.map
382
+ //# sourceMappingURL=semantic-concept-judge-I36eejJx.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"semantic-concept-judge-BSkKKHeq.js","names":[],"sources":["../src/locked-jsonl-appender.ts","../src/analyst/findings-store.ts","../src/semantic-concept-judge.ts"],"sourcesContent":["/**\n * LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary\n * payloads. The reference-replay store does the same thing for typed\n * `ReferenceReplayRun` rows; this is the generic version used by\n * `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants\n * append-only durable telemetry without rolling its own lock.\n *\n * Locks are per absolute file path (process-local). Cross-process\n * concurrency is NOT addressed — that's an fcntl/flock problem.\n */\n\nimport { appendFileSync, existsSync, mkdirSync } from 'node:fs'\nimport { dirname } from 'node:path'\nimport { Mutex } from './concurrency'\n\nconst mutexes = new Map<string, Mutex>()\n\nfunction getMutex(path: string): Mutex {\n let m = mutexes.get(path)\n if (!m) {\n m = new Mutex()\n mutexes.set(path, m)\n }\n return m\n}\n\nexport class LockedJsonlAppender {\n private readonly mutex: Mutex\n constructor(public readonly path: string) {\n this.mutex = getMutex(path)\n if (!existsSync(dirname(path))) {\n mkdirSync(dirname(path), { recursive: true })\n }\n }\n\n async append(entry: unknown): Promise<void> {\n const line = `${JSON.stringify(entry)}\\n`\n await this.mutex.runExclusive(() => {\n appendFileSync(this.path, line)\n })\n }\n}\n\n/** Reset all internal mutex state — tests only. */\nexport function resetLockedAppendersForTesting(): void {\n mutexes.clear()\n}\n","/**\n * FindingsStore — durable persistence for AnalystFinding rows + a diff\n * helper so we can answer \"what changed since the last run?\" without\n * recomputing analysts.\n *\n * On-disk shape is JSONL: one finding per line, append-only, locked via\n * LockedJsonlAppender. Operators get crash-safety (no partial JSON),\n * cheap reads (sequential parse), and trivial backup (rsync the file).\n *\n * Reads are non-locking: a reader sees a consistent snapshot of all\n * fully-written lines and skips an incomplete trailing line if the\n * writer is mid-append. Cross-process locking is intentionally out of\n * scope (see locked-jsonl-appender.ts).\n *\n * The store is run-scoped: callers pass `runId` on append and on load,\n * which keeps multi-run files cleanly partitioned. The `diffFindings`\n * helper compares two run-id sets using stable `finding_id` semantics —\n * the diff is the cross-run signal the regression dashboard renders.\n */\n\nimport { existsSync, readFileSync } from 'node:fs'\n\nimport { LockedJsonlAppender } from '../locked-jsonl-appender'\nimport type { AnalystFinding } from './types'\n\n/**\n * One persisted row. We attach `run_id` on disk so a single file can\n * hold multiple runs and the diff helper can query without re-walking\n * separate files.\n */\nexport interface PersistedFinding extends AnalystFinding {\n run_id: string\n}\n\nexport class FindingsStore {\n private readonly appender: LockedJsonlAppender\n\n constructor(public readonly path: string) {\n this.appender = new LockedJsonlAppender(path)\n }\n\n async append(runId: string, findings: AnalystFinding[]): Promise<void> {\n for (const f of findings) {\n const row: PersistedFinding = { ...f, run_id: runId }\n await this.appender.append(row)\n }\n }\n\n /** Load every persisted finding. Discards malformed trailing lines silently. */\n loadAll(): PersistedFinding[] {\n if (!existsSync(this.path)) return []\n const raw = readFileSync(this.path, 'utf8')\n if (!raw) return []\n const out: PersistedFinding[] = []\n for (const line of raw.split('\\n')) {\n if (!line) continue\n try {\n out.push(JSON.parse(line) as PersistedFinding)\n } catch {\n // Skip torn trailing line — the lock guarantees no torn lines\n // mid-file, only at EOF when a writer is in-flight.\n }\n }\n return out\n }\n\n /** Filter to a single run. */\n loadRun(runId: string): PersistedFinding[] {\n return this.loadAll().filter((r) => r.run_id === runId)\n }\n}\n\n// ── Cross-run diff ──────────────────────────────────────────────────\n\nexport interface FindingsDiff {\n /** New finding ids in `current` that weren't in `previous`. */\n appeared: PersistedFinding[]\n /** Finding ids in `previous` that aren't in `current`. */\n disappeared: PersistedFinding[]\n /** Same finding id present in both runs and unchanged per the materiality test. */\n persisted: PersistedFinding[]\n /**\n * Same finding id in both runs but at least one non-identity field\n * shifted per `DiffPolicy.isMaterial`. Reported as [previous, current].\n */\n changed: Array<{ previous: PersistedFinding; current: PersistedFinding }>\n}\n\nexport interface DiffPolicy {\n /**\n * Predicate that decides whether two findings (same finding_id) count\n * as a material change. Defaults to {@link defaultIsMaterial}: severity\n * shift, confidence Δ > 0.05, or evidence count change. Compliance /\n * perf consumers MAY supply a stricter predicate (e.g. rationale text\n * diff, metric Δ thresholds).\n */\n isMaterial?: (previous: AnalystFinding, current: AnalystFinding) => boolean\n}\n\n/**\n * Default materiality test. Deliberately narrow so LLM-reword churn\n * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.\n */\nexport function defaultIsMaterial(a: AnalystFinding, b: AnalystFinding): boolean {\n if (a.severity !== b.severity) return true\n if (Math.abs((a.confidence ?? 0) - (b.confidence ?? 0)) > 0.05) return true\n if (a.evidence_refs.length !== b.evidence_refs.length) return true\n return false\n}\n\n/**\n * Diff two findings sets by stable finding_id. Callers typically load\n * the two run-id slices from the same store and pass them in.\n */\nexport function diffFindings(\n previous: PersistedFinding[],\n current: PersistedFinding[],\n policy: DiffPolicy = {},\n): FindingsDiff {\n const isMaterial = policy.isMaterial ?? defaultIsMaterial\n const prevById = new Map(previous.map((f) => [f.finding_id, f]))\n const curById = new Map(current.map((f) => [f.finding_id, f]))\n\n const appeared: PersistedFinding[] = []\n const disappeared: PersistedFinding[] = []\n const persisted: PersistedFinding[] = []\n const changed: FindingsDiff['changed'] = []\n\n for (const [id, cur] of curById) {\n const prev = prevById.get(id)\n if (!prev) {\n appeared.push(cur)\n continue\n }\n if (isMaterial(prev, cur)) {\n changed.push({ previous: prev, current: cur })\n } else {\n persisted.push(cur)\n }\n }\n for (const [id, prev] of prevById) {\n if (!curById.has(id)) disappeared.push(prev)\n }\n return { appeared, disappeared, persisted, changed }\n}\n","/**\n * Semantic concept judge — \"does the built artifact actually implement\n * the features the user asked for?\"\n *\n * Distinct from the domain/code/coherence judges in `judges.ts`:\n * - those judges score free-form conversational agent outputs along\n * quality dimensions (accuracy, depth, etc.)\n * - this judge scores a *built artifact* (served HTML + source files)\n * against an explicit list of expected concepts, returning per-concept\n * {present, score 0-10, evidence, severity}.\n *\n * The judge is strict about distinguishing (a) a working implementation\n * from (b) a keyword-present stub. \"// TODO: mint button\" is NOT present.\n * Only real, functional, wired-up code counts.\n *\n * Use via {@link createSemanticConceptJudge} or directly via\n * {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM\n * or JSON-parse errors so the caller can treat that as \"layer skipped\"\n * rather than \"layer failed\" in a multi-layer pipeline.\n */\n\nimport type { ChatClient } from './analyst/chat-client'\nimport { paidJsonChat } from './chat-json-call'\nimport {\n CostLedger,\n type CostLedgerHandle,\n type CostReceipt,\n type CustomTokenPricing,\n} from './cost-ledger'\nimport type { LlmCallRequest } from './llm-client'\nimport type { Severity } from './multi-layer-verifier'\n\n// ─── Types ──────────────────────────────────────────────────────────────\n\n/**\n * Implementation complexity class for weighted scoring.\n *\n * - `render` (default): the concept is a UI surface that displays static\n * data — render a list, show a counter, lay out a button. Single-file\n * work, no external integration.\n * - `integrate`: the concept requires wiring a real external system —\n * wallet connect (wagmi + RainbowKit + chain config), payment provider\n * (Stripe Elements + intent + webhook), an API client with auth.\n * Multi-file, library-knowledge, runtime correctness matters.\n * - `compute`: the concept requires algorithmic work — solver, simulator,\n * constraint propagation, ML inference. Correctness > UI polish.\n *\n * Default weights (when applied via `weightConcepts: 'complexity'`):\n * render=1.0, integrate=2.0, compute=2.5\n *\n * Cross-vertical scoring without complexity weighting silently inflates\n * the rate of UI-heavy verticals (healthcare, fintech dashboards) vs\n * integration-heavy verticals (DeFi, wallets) — all concepts treated\n * equally even though the agent does 2-3x the work for `integrate`.\n */\nexport type ConceptComplexity = 'render' | 'integrate' | 'compute'\n\nexport interface ConceptSpec {\n name: string\n /** Short hints that help the judge; not used for matching. */\n keywords?: string[]\n /** Optional explicit weight; default 1.0. Overrides complexity-derived weight. */\n weight?: number\n /** Implementation complexity class. Default `render`. */\n complexity?: ConceptComplexity\n}\n\nexport interface ConceptFinding {\n concept: string\n present: boolean\n /** 0..10. 10 = production-ready; 7 = functional thin; 4 = partial; 0 = absent. */\n score: number\n evidence: string\n severity: Severity\n}\n\nexport interface SemanticConceptJudgeInput {\n /** Full natural-language prompt the agent was handed. */\n userRequest: string\n /** Rendered HTML the preview returns (UI artifacts). Optional. */\n servedHtml?: string\n /** Top-level source files from the agent's workdir. */\n sourceFiles: Array<{ path: string; content: string }>\n /** The expected concept list. */\n expectedConcepts: ConceptSpec[]\n /** Free-form metadata (id, difficulty) to inject into the prompt. */\n artifactLabel?: string\n artifactDescription?: string\n}\n\nexport interface SemanticConceptJudgeResult {\n kind: 'semantic-concept'\n version: string\n /** Normalized 0..1 score — mean of per-concept scores / 10. */\n score: number\n presentCount: number\n totalCount: number\n findings: ConceptFinding[]\n summary: string\n durationMs: number\n costUsd: number | null\n /** False on LLM/JSON error — treat as \"skipped / unable to judge\" in pipelines. */\n available: boolean\n error?: string\n}\n\n/**\n * Score-aggregation strategy. `mean` averages 0-10 scores uniformly.\n * `complexity` applies the default weight table (render=1, integrate=2,\n * compute=2.5) unless a concept has an explicit `weight`. `explicit`\n * honors only `weight` (defaulting to 1 for unspecified).\n */\nexport type ConceptWeightStrategy = 'mean' | 'complexity' | 'explicit'\n\nexport const DEFAULT_COMPLEXITY_WEIGHTS: Record<ConceptComplexity, number> = {\n render: 1.0,\n integrate: 2.0,\n compute: 2.5,\n}\n\nexport interface SemanticConceptJudgeOptions {\n /** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */\n model?: string\n /** Per-call timeout. Default 300s. */\n timeoutMs?: number\n /** Provider-enforced output limit. Default 16000. */\n maxTokens?: number\n /** Pipeline budget for the prompt (source blob truncation). Default 45000. */\n maxSourceChars?: number\n /** Per-file cap before inclusion. Default 20000. */\n maxPerFileChars?: number\n /** HTML cap. Default 30000. */\n maxHtmlChars?: number\n /** Caller-owned transport. Required: agent-eval executes no paid model. */\n chat: ChatClient\n /** Endpoint rates used when the transport reports no billed amount. */\n pricing?: CustomTokenPricing\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /**\n * Score aggregation strategy. Default `mean` — uniform average across\n * concepts. Cross-vertical comparisons should use `complexity` to\n * neutralize the integrate-vs-render asymmetry.\n */\n weightConcepts?: ConceptWeightStrategy\n /** Override the default complexity → weight table. */\n complexityWeights?: Partial<Record<ConceptComplexity, number>>\n}\n\n// ─── Prompt assembly ────────────────────────────────────────────────────\n\nexport const SEMANTIC_CONCEPT_JUDGE_VERSION = 'semantic-concept-judge-v1-2026-04-24'\n\nconst DEFAULT_MAX_SOURCE = 45_000\nconst DEFAULT_MAX_HTML = 30_000\nconst DEFAULT_MAX_PER_FILE = 20_000\nconst DEFAULT_TIMEOUT = 300_000\nconst DEFAULT_MAX_TOKENS = 16_000\nconst DEFAULT_MODEL = 'claude-sonnet-4-6'\n\nconst SEMANTIC_SCHEMA = {\n type: 'object',\n additionalProperties: false,\n required: ['summary', 'concepts'],\n properties: {\n summary: { type: 'string', minLength: 20, maxLength: 600 },\n concepts: {\n type: 'array',\n minItems: 1,\n items: {\n type: 'object',\n additionalProperties: false,\n required: ['concept', 'present', 'score', 'evidence', 'severity'],\n properties: {\n concept: { type: 'string', minLength: 1, maxLength: 120 },\n present: { type: 'boolean' },\n score: { type: 'number', minimum: 0, maximum: 10 },\n evidence: { type: 'string', minLength: 5, maxLength: 400 },\n severity: { type: 'string', enum: ['critical', 'major', 'minor', 'info'] },\n },\n },\n },\n },\n}\n\nfunction truncate(body: string, cap: number, label: string): string {\n if (body.length <= cap) return body\n return `${body.slice(0, cap)}\\n… [truncated ${body.length - cap} chars of ${label}]`\n}\n\nfunction buildPrompt(\n input: SemanticConceptJudgeInput,\n opts: { maxPerFileChars: number; maxSourceChars: number; maxHtmlChars: number },\n): string {\n const sourceBlob = input.sourceFiles\n .filter((f) => f.content.length <= opts.maxPerFileChars)\n .map((f) => `--- FILE: ${f.path} ---\\n${f.content}`)\n .join('\\n\\n')\n\n const html = input.servedHtml ?? ''\n\n return `You are a strict code-review judge evaluating whether an agent's 0-to-1 build actually implements the features the user asked for.\n\nYou MUST distinguish:\n (a) WORKING code that implements the concept (rendered UI, wired handler, real API call),\n (b) KEYWORD-PRESENT stub (comments mentioning the concept, variable names, TODOs),\n (c) ABSENT (concept nowhere).\n\nA comment like \"// TODO: add mint button\" is NOT present — score 2-3. Only count a concept as present if there is real functional code: a rendered component, a call handler wired to state or a network call, a computed value actually used.\n\nUSER REQUEST (what the agent was asked to build):\n${input.userRequest}\n\n${input.artifactLabel ? `ARTIFACT METADATA:\\n name: ${input.artifactLabel}\\n description: ${input.artifactDescription ?? ''}\\n\\n` : ''}EXPECTED CONCEPTS (each must be graded independently):\n${input.expectedConcepts\n .map(\n (c, i) =>\n ` ${i + 1}. \"${c.name}\"${c.keywords?.length ? ` — hints: [${c.keywords.slice(0, 6).join(' | ')}]` : ''}`,\n )\n .join('\\n')}\n\n${html ? `SERVED HTML (what the preview returns when hit):\\n${truncate(html, opts.maxHtmlChars, 'HTML')}\\n\\n` : ''}SOURCE FILES (the agent's workdir):\n${truncate(sourceBlob, opts.maxSourceChars, 'source')}\n\nFor EACH concept, return:\n - concept: the concept name as given (match exactly)\n - present: boolean — does a working implementation exist?\n - score: 0-10 — 10 = production-ready; 7 = functional but thin; 4 = partial/stubbed; 2 = keyword-only comment; 0 = absent\n - evidence: cite \"<file>:<line>\" or \"served-html:<selector>\" pointing at the strongest supporting code. If the concept is absent or stubbed, explain what's missing.\n - severity:\n \"info\" when present: true AND score >= 7\n \"minor\" when present: true AND 4 <= score < 7\n \"major\" when present: false OR score < 4\n \"critical\" when the concept is not only absent but a core user flow depends on it\n\nAlso produce a \"summary\" (one sentence, 20-600 chars): overall verdict on whether this is a shippable implementation of the user request vs a keyword-dense placeholder.\n\nBE SKEPTICAL. Keyword matching already passed — your job is to catch what keyword matching misses. If the agent shipped a working build, say so. If it shipped a stub, say so. Don't grade on effort.\n\nReturn STRICT JSON. No prose outside the JSON.`\n}\n\n// ─── Runner ─────────────────────────────────────────────────────────────\n\n/**\n * Run the semantic concept judge. Soft-fails to available=false on\n * LLM/JSON errors — callers in a MultiLayerVerifier pipeline can treat\n * that as \"skip\" rather than \"fail.\"\n */\nexport async function runSemanticConceptJudge(\n input: SemanticConceptJudgeInput,\n options: SemanticConceptJudgeOptions,\n): Promise<SemanticConceptJudgeResult> {\n const start = Date.now()\n const totalCount = input.expectedConcepts.length\n\n if (totalCount === 0) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount: 0,\n findings: [],\n summary: 'no expected concepts declared',\n durationMs: 0,\n costUsd: null,\n available: false,\n error: 'no expected concepts declared',\n }\n }\n\n const opts = {\n chat: options.chat,\n model: options.model ?? options.chat.defaultModel ?? DEFAULT_MODEL,\n timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,\n maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,\n maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,\n maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,\n maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,\n ...(options.pricing ? { pricing: options.pricing } : {}),\n costLedger: options.costLedger ?? new CostLedger(),\n costPhase: options.costPhase ?? 'judge.semantic-concept',\n costTags: options.costTags ?? {},\n signal: options.signal ?? new AbortController().signal,\n weightConcepts: options.weightConcepts ?? 'mean',\n complexityWeights: { ...DEFAULT_COMPLEXITY_WEIGHTS, ...(options.complexityWeights ?? {}) },\n }\n\n // Build a name → weight map for aggregation. Mean strategy keeps every\n // weight at 1 (uniform average). Complexity strategy reads the table\n // and lets an explicit `weight` override. Explicit strategy uses ONLY\n // the spec's `weight` (defaulting to 1).\n const weightForConcept = (spec: ConceptSpec): number => {\n if (opts.weightConcepts === 'mean') return 1\n if (spec.weight != null) return spec.weight\n if (opts.weightConcepts === 'complexity') {\n return opts.complexityWeights[spec.complexity ?? 'render'] ?? 1\n }\n return 1\n }\n const weightByName = new Map<string, number>(\n input.expectedConcepts.map((c) => [c.name, weightForConcept(c)]),\n )\n\n let receipt: CostReceipt | undefined\n try {\n const request = {\n model: opts.model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You are a strict code-review judge. Return strict JSON only. No prose outside the JSON. A keyword in a comment is NOT a working implementation.',\n },\n { role: 'user' as const, content: buildPrompt(input, opts) },\n ],\n jsonSchema: { name: 'semantic_concept_judge', schema: SEMANTIC_SCHEMA },\n temperature: 0,\n maxTokens: opts.maxTokens,\n timeoutMs: opts.timeoutMs,\n } satisfies LlmCallRequest\n const paid = await paidJsonChat<{ summary: string; concepts: ConceptFinding[] }>({\n chat: opts.chat,\n request,\n ledger: opts.costLedger,\n channel: 'judge',\n phase: opts.costPhase,\n actor: 'semantic-concept',\n tags: opts.costTags,\n signal: opts.signal,\n ...(opts.pricing ? { pricing: opts.pricing } : {}),\n })\n receipt = paid.receipt\n if (!paid.succeeded) throw paid.error\n const { value } = paid\n\n if (!value?.concepts || !Array.isArray(value.concepts)) {\n throw new Error('judge returned malformed response — expected array under \"concepts\"')\n }\n\n const findings: ConceptFinding[] = value.concepts.map((c) => ({\n concept: String(c.concept),\n present: Boolean(c.present),\n score: Math.max(0, Math.min(10, Number(c.score ?? 0))),\n evidence: String(c.evidence ?? ''),\n severity: (['critical', 'major', 'minor', 'info'] as const).includes(c.severity)\n ? c.severity\n : 'info',\n }))\n\n const presentCount = findings.filter((f) => f.present && f.score >= 7).length\n let weightSum = 0\n let weightedScoreSum = 0\n for (const f of findings) {\n const w = weightByName.get(f.concept) ?? 1\n weightSum += w\n weightedScoreSum += w * f.score\n }\n const scoreAvg =\n weightSum > 0\n ? weightedScoreSum / weightSum\n : findings.reduce((a, f) => a + f.score, 0) / Math.max(1, findings.length)\n\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: Number((scoreAvg / 10).toFixed(3)),\n presentCount,\n totalCount,\n findings,\n summary: String(value.summary ?? ''),\n durationMs: Date.now() - start,\n costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,\n available: true,\n }\n } catch (err) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount,\n findings: [],\n summary: '',\n durationMs: Date.now() - start,\n costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,\n available: false,\n error: err instanceof Error ? err.message : String(err),\n }\n }\n}\n\n/**\n * Factory: pin LLM options once, return a closure that accepts inputs.\n * Convenient for pipelines that want to share a single LlmClient config.\n */\nexport function createSemanticConceptJudge(\n options: SemanticConceptJudgeOptions,\n): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult> {\n return (input) => runSemanticConceptJudge(input, options)\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAM,0BAAU,IAAI,IAAmB;AAEvC,SAAS,SAAS,MAAqB;CACrC,IAAI,IAAI,QAAQ,IAAI,IAAI;CACxB,IAAI,CAAC,GAAG;EACN,IAAI,IAAI,MAAM;EACd,QAAQ,IAAI,MAAM,CAAC;CACrB;CACA,OAAO;AACT;AAEA,IAAa,sBAAb,MAAiC;CAEH;CAD5B;CACA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,QAAQ,SAAS,IAAI;EAC1B,IAAI,CAAC,WAAW,QAAQ,IAAI,CAAC,GAC3B,UAAU,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAEhD;CAEA,MAAM,OAAO,OAA+B;EAC1C,MAAM,OAAO,GAAG,KAAK,UAAU,KAAK,EAAE;EACtC,MAAM,KAAK,MAAM,mBAAmB;GAClC,eAAe,KAAK,MAAM,IAAI;EAChC,CAAC;CACH;AACF;;;;;;;;;;;;;;;;;;;;;;ACPA,IAAa,gBAAb,MAA2B;CAGG;CAF5B;CAEA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,WAAW,IAAI,oBAAoB,IAAI;CAC9C;CAEA,MAAM,OAAO,OAAe,UAA2C;EACrE,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,MAAwB;IAAE,GAAG;IAAG,QAAQ;GAAM;GACpD,MAAM,KAAK,SAAS,OAAO,GAAG;EAChC;CACF;;CAGA,UAA8B;EAC5B,IAAI,CAAC,WAAW,KAAK,IAAI,GAAG,OAAO,CAAC;EACpC,MAAM,MAAM,aAAa,KAAK,MAAM,MAAM;EAC1C,IAAI,CAAC,KAAK,OAAO,CAAC;EAClB,MAAM,MAA0B,CAAC;EACjC,KAAK,MAAM,QAAQ,IAAI,MAAM,IAAI,GAAG;GAClC,IAAI,CAAC,MAAM;GACX,IAAI;IACF,IAAI,KAAK,KAAK,MAAM,IAAI,CAAqB;GAC/C,QAAQ,CAGR;EACF;EACA,OAAO;CACT;;CAGA,QAAQ,OAAmC;EACzC,OAAO,KAAK,QAAQ,CAAC,CAAC,QAAQ,MAAM,EAAE,WAAW,KAAK;CACxD;AACF;;;;;AAiCA,SAAgB,kBAAkB,GAAmB,GAA4B;CAC/E,IAAI,EAAE,aAAa,EAAE,UAAU,OAAO;CACtC,IAAI,KAAK,KAAK,EAAE,cAAc,MAAM,EAAE,cAAc,EAAE,IAAI,KAAM,OAAO;CACvE,IAAI,EAAE,cAAc,WAAW,EAAE,cAAc,QAAQ,OAAO;CAC9D,OAAO;AACT;;;;;AAMA,SAAgB,aACd,UACA,SACA,SAAqB,CAAC,GACR;CACd,MAAM,aAAa,OAAO,cAAc;CACxC,MAAM,WAAW,IAAI,IAAI,SAAS,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAC/D,MAAM,UAAU,IAAI,IAAI,QAAQ,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAE7D,MAAM,WAA+B,CAAC;CACtC,MAAM,cAAkC,CAAC;CACzC,MAAM,YAAgC,CAAC;CACvC,MAAM,UAAmC,CAAC;CAE1C,KAAK,MAAM,CAAC,IAAI,QAAQ,SAAS;EAC/B,MAAM,OAAO,SAAS,IAAI,EAAE;EAC5B,IAAI,CAAC,MAAM;GACT,SAAS,KAAK,GAAG;GACjB;EACF;EACA,IAAI,WAAW,MAAM,GAAG,GACtB,QAAQ,KAAK;GAAE,UAAU;GAAM,SAAS;EAAI,CAAC;OAE7C,UAAU,KAAK,GAAG;CAEtB;CACA,KAAK,MAAM,CAAC,IAAI,SAAS,UACvB,IAAI,CAAC,QAAQ,IAAI,EAAE,GAAG,YAAY,KAAK,IAAI;CAE7C,OAAO;EAAE;EAAU;EAAa;EAAW;CAAQ;AACrD;;;AC9BA,MAAa,6BAAgE;CAC3E,QAAQ;CACR,WAAW;CACX,SAAS;AACX;AAmCA,MAAa,iCAAiC;AAE9C,MAAM,qBAAqB;AAC3B,MAAM,mBAAmB;AACzB,MAAM,uBAAuB;AAC7B,MAAM,kBAAkB;AACxB,MAAM,qBAAqB;AAC3B,MAAM,gBAAgB;AAEtB,MAAM,kBAAkB;CACtB,MAAM;CACN,sBAAsB;CACtB,UAAU,CAAC,WAAW,UAAU;CAChC,YAAY;EACV,SAAS;GAAE,MAAM;GAAU,WAAW;GAAI,WAAW;EAAI;EACzD,UAAU;GACR,MAAM;GACN,UAAU;GACV,OAAO;IACL,MAAM;IACN,sBAAsB;IACtB,UAAU;KAAC;KAAW;KAAW;KAAS;KAAY;IAAU;IAChE,YAAY;KACV,SAAS;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACxD,SAAS,EAAE,MAAM,UAAU;KAC3B,OAAO;MAAE,MAAM;MAAU,SAAS;MAAG,SAAS;KAAG;KACjD,UAAU;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACzD,UAAU;MAAE,MAAM;MAAU,MAAM;OAAC;OAAY;OAAS;OAAS;MAAM;KAAE;IAC3E;GACF;EACF;CACF;AACF;AAEA,SAAS,SAAS,MAAc,KAAa,OAAuB;CAClE,IAAI,KAAK,UAAU,KAAK,OAAO;CAC/B,OAAO,GAAG,KAAK,MAAM,GAAG,GAAG,EAAE,iBAAiB,KAAK,SAAS,IAAI,YAAY,MAAM;AACpF;AAEA,SAAS,YACP,OACA,MACQ;CACR,MAAM,aAAa,MAAM,YACtB,QAAQ,MAAM,EAAE,QAAQ,UAAU,KAAK,eAAe,CAAC,CACvD,KAAK,MAAM,aAAa,EAAE,KAAK,QAAQ,EAAE,SAAS,CAAC,CACnD,KAAK,MAAM;CAEd,MAAM,OAAO,MAAM,cAAc;CAEjC,OAAO;;;;;;;;;;EAUP,MAAM,YAAY;;EAElB,MAAM,gBAAgB,+BAA+B,MAAM,cAAc,mBAAmB,MAAM,uBAAuB,GAAG,QAAQ,GAAG;EACvI,MAAM,iBACL,KACE,GAAG,MACF,KAAK,IAAI,EAAE,KAAK,EAAE,KAAK,GAAG,EAAE,UAAU,SAAS,cAAc,EAAE,SAAS,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,KAAK,EAAE,KAAK,IACzG,CAAC,CACA,KAAK,IAAI,EAAE;;EAEZ,OAAO,qDAAqD,SAAS,MAAM,KAAK,cAAc,MAAM,EAAE,QAAQ,GAAG;EACjH,SAAS,YAAY,KAAK,gBAAgB,QAAQ,EAAE;;;;;;;;;;;;;;;;;;AAkBtD;;;;;;AASA,eAAsB,wBACpB,OACA,SACqC;CACrC,MAAM,QAAQ,KAAK,IAAI;CACvB,MAAM,aAAa,MAAM,iBAAiB;CAE1C,IAAI,eAAe,GACjB,OAAO;EACL,MAAM;EACN,SAAS;EACT,OAAO;EACP,cAAc;EACd,YAAY;EACZ,UAAU,CAAC;EACX,SAAS;EACT,YAAY;EACZ,SAAS;EACT,WAAW;EACX,OAAO;CACT;CAGF,MAAM,OAAO;EACX,MAAM,QAAQ;EACd,OAAO,QAAQ,SAAS,QAAQ,KAAK,gBAAgB;EACrD,WAAW,QAAQ,aAAa;EAChC,WAAW,QAAQ,aAAa;EAChC,gBAAgB,QAAQ,kBAAkB;EAC1C,iBAAiB,QAAQ,mBAAmB;EAC5C,cAAc,QAAQ,gBAAgB;EACtC,GAAI,QAAQ,UAAU,EAAE,SAAS,QAAQ,QAAQ,IAAI,CAAC;EACtD,YAAY,QAAQ,cAAc,IAAI,WAAW;EACjD,WAAW,QAAQ,aAAa;EAChC,UAAU,QAAQ,YAAY,CAAC;EAC/B,QAAQ,QAAQ,UAAU,IAAI,gBAAgB,CAAC,CAAC;EAChD,gBAAgB,QAAQ,kBAAkB;EAC1C,mBAAmB;GAAE,GAAG;GAA4B,GAAI,QAAQ,qBAAqB,CAAC;EAAG;CAC3F;CAMA,MAAM,oBAAoB,SAA8B;EACtD,IAAI,KAAK,mBAAmB,QAAQ,OAAO;EAC3C,IAAI,KAAK,UAAU,MAAM,OAAO,KAAK;EACrC,IAAI,KAAK,mBAAmB,cAC1B,OAAO,KAAK,kBAAkB,KAAK,cAAc,aAAa;EAEhE,OAAO;CACT;CACA,MAAM,eAAe,IAAI,IACvB,MAAM,iBAAiB,KAAK,MAAM,CAAC,EAAE,MAAM,iBAAiB,CAAC,CAAC,CAAC,CACjE;CAEA,IAAI;CACJ,IAAI;EACF,MAAM,UAAU;GACd,OAAO,KAAK;GACZ,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IAAE,MAAM;IAAiB,SAAS,YAAY,OAAO,IAAI;GAAE,CAC7D;GACA,YAAY;IAAE,MAAM;IAA0B,QAAQ;GAAgB;GACtE,aAAa;GACb,WAAW,KAAK;GAChB,WAAW,KAAK;EAClB;EACA,MAAM,OAAO,MAAM,aAA8D;GAC/E,MAAM,KAAK;GACX;GACA,QAAQ,KAAK;GACb,SAAS;GACT,OAAO,KAAK;GACZ,OAAO;GACP,MAAM,KAAK;GACX,QAAQ,KAAK;GACb,GAAI,KAAK,UAAU,EAAE,SAAS,KAAK,QAAQ,IAAI,CAAC;EAClD,CAAC;EACD,UAAU,KAAK;EACf,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;EAChC,MAAM,EAAE,UAAU;EAElB,IAAI,CAAC,OAAO,YAAY,CAAC,MAAM,QAAQ,MAAM,QAAQ,GACnD,MAAM,IAAI,MAAM,uEAAqE;EAGvF,MAAM,WAA6B,MAAM,SAAS,KAAK,OAAO;GAC5D,SAAS,OAAO,EAAE,OAAO;GACzB,SAAS,QAAQ,EAAE,OAAO;GAC1B,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,IAAI,OAAO,EAAE,SAAS,CAAC,CAAC,CAAC;GACrD,UAAU,OAAO,EAAE,YAAY,EAAE;GACjC,UAAW;IAAC;IAAY;IAAS;IAAS;GAAM,CAAC,CAAW,SAAS,EAAE,QAAQ,IAC3E,EAAE,WACF;EACN,EAAE;EAEF,MAAM,eAAe,SAAS,QAAQ,MAAM,EAAE,WAAW,EAAE,SAAS,CAAC,CAAC,CAAC;EACvE,IAAI,YAAY;EAChB,IAAI,mBAAmB;EACvB,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,IAAI,aAAa,IAAI,EAAE,OAAO,KAAK;GACzC,aAAa;GACb,oBAAoB,IAAI,EAAE;EAC5B;EACA,MAAM,WACJ,YAAY,IACR,mBAAmB,YACnB,SAAS,QAAQ,GAAG,MAAM,IAAI,EAAE,OAAO,CAAC,IAAI,KAAK,IAAI,GAAG,SAAS,MAAM;EAE7E,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO,QAAQ,WAAW,GAAA,CAAI,QAAQ,CAAC,CAAC;GACxC;GACA;GACA;GACA,SAAS,OAAO,MAAM,WAAW,EAAE;GACnC,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,KAAK,QAAQ,cAAc,OAAO,KAAK,QAAQ;GACxD,WAAW;EACb;CACF,SAAS,KAAK;EACZ,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO;GACP,cAAc;GACd;GACA,UAAU,CAAC;GACX,SAAS;GACT,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,WAAW,CAAC,QAAQ,cAAc,QAAQ,UAAU;GAC7D,WAAW;GACX,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;EACxD;CACF;AACF"}
1
+ {"version":3,"file":"semantic-concept-judge-I36eejJx.js","names":[],"sources":["../src/locked-jsonl-appender.ts","../src/analyst/findings-store.ts","../src/semantic-concept-judge.ts"],"sourcesContent":["/**\n * LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary\n * payloads. The reference-replay store does the same thing for typed\n * `ReferenceReplayRun` rows; this is the generic version used by\n * `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants\n * append-only durable telemetry without rolling its own lock.\n *\n * Locks are per absolute file path (process-local). Cross-process\n * concurrency is NOT addressed — that's an fcntl/flock problem.\n */\n\nimport { appendFileSync, existsSync, mkdirSync } from 'node:fs'\nimport { dirname } from 'node:path'\nimport { Mutex } from './concurrency'\n\nconst mutexes = new Map<string, Mutex>()\n\nfunction getMutex(path: string): Mutex {\n let m = mutexes.get(path)\n if (!m) {\n m = new Mutex()\n mutexes.set(path, m)\n }\n return m\n}\n\nexport class LockedJsonlAppender {\n private readonly mutex: Mutex\n constructor(public readonly path: string) {\n this.mutex = getMutex(path)\n if (!existsSync(dirname(path))) {\n mkdirSync(dirname(path), { recursive: true })\n }\n }\n\n async append(entry: unknown): Promise<void> {\n const line = `${JSON.stringify(entry)}\\n`\n await this.mutex.runExclusive(() => {\n appendFileSync(this.path, line)\n })\n }\n}\n\n/** Reset all internal mutex state — tests only. */\nexport function resetLockedAppendersForTesting(): void {\n mutexes.clear()\n}\n","/**\n * FindingsStore — durable persistence for AnalystFinding rows + a diff\n * helper so we can answer \"what changed since the last run?\" without\n * recomputing analysts.\n *\n * On-disk shape is JSONL: one finding per line, append-only, locked via\n * LockedJsonlAppender. Operators get crash-safety (no partial JSON),\n * cheap reads (sequential parse), and trivial backup (rsync the file).\n *\n * Reads are non-locking: a reader sees a consistent snapshot of all\n * fully-written lines and skips an incomplete trailing line if the\n * writer is mid-append. Cross-process locking is intentionally out of\n * scope (see locked-jsonl-appender.ts).\n *\n * The store is run-scoped: callers pass `runId` on append and on load,\n * which keeps multi-run files cleanly partitioned. The `diffFindings`\n * helper compares two run-id sets using stable `finding_id` semantics —\n * the diff is the cross-run signal the regression dashboard renders.\n */\n\nimport { existsSync, readFileSync } from 'node:fs'\n\nimport { LockedJsonlAppender } from '../locked-jsonl-appender'\nimport type { AnalystFinding } from './types'\n\n/**\n * One persisted row. We attach `run_id` on disk so a single file can\n * hold multiple runs and the diff helper can query without re-walking\n * separate files.\n */\nexport interface PersistedFinding extends AnalystFinding {\n run_id: string\n}\n\nexport class FindingsStore {\n private readonly appender: LockedJsonlAppender\n\n constructor(public readonly path: string) {\n this.appender = new LockedJsonlAppender(path)\n }\n\n async append(runId: string, findings: AnalystFinding[]): Promise<void> {\n for (const f of findings) {\n const row: PersistedFinding = { ...f, run_id: runId }\n await this.appender.append(row)\n }\n }\n\n /** Load every persisted finding. Discards malformed trailing lines silently. */\n loadAll(): PersistedFinding[] {\n if (!existsSync(this.path)) return []\n const raw = readFileSync(this.path, 'utf8')\n if (!raw) return []\n const out: PersistedFinding[] = []\n for (const line of raw.split('\\n')) {\n if (!line) continue\n try {\n out.push(JSON.parse(line) as PersistedFinding)\n } catch {\n // Skip torn trailing line — the lock guarantees no torn lines\n // mid-file, only at EOF when a writer is in-flight.\n }\n }\n return out\n }\n\n /** Filter to a single run. */\n loadRun(runId: string): PersistedFinding[] {\n return this.loadAll().filter((r) => r.run_id === runId)\n }\n}\n\n// ── Cross-run diff ──────────────────────────────────────────────────\n\nexport interface FindingsDiff {\n /** New finding ids in `current` that weren't in `previous`. */\n appeared: PersistedFinding[]\n /** Finding ids in `previous` that aren't in `current`. */\n disappeared: PersistedFinding[]\n /** Same finding id present in both runs and unchanged per the materiality test. */\n persisted: PersistedFinding[]\n /**\n * Same finding id in both runs but at least one non-identity field\n * shifted per `DiffPolicy.isMaterial`. Reported as [previous, current].\n */\n changed: Array<{ previous: PersistedFinding; current: PersistedFinding }>\n}\n\nexport interface DiffPolicy {\n /**\n * Predicate that decides whether two findings (same finding_id) count\n * as a material change. Defaults to {@link defaultIsMaterial}: severity\n * shift, confidence Δ > 0.05, or evidence count change. Compliance /\n * perf consumers MAY supply a stricter predicate (e.g. rationale text\n * diff, metric Δ thresholds).\n */\n isMaterial?: (previous: AnalystFinding, current: AnalystFinding) => boolean\n}\n\n/**\n * Default materiality test. Deliberately narrow so LLM-reword churn\n * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.\n */\nexport function defaultIsMaterial(a: AnalystFinding, b: AnalystFinding): boolean {\n if (a.severity !== b.severity) return true\n if (Math.abs((a.confidence ?? 0) - (b.confidence ?? 0)) > 0.05) return true\n if (a.evidence_refs.length !== b.evidence_refs.length) return true\n return false\n}\n\n/**\n * Diff two findings sets by stable finding_id. Callers typically load\n * the two run-id slices from the same store and pass them in.\n */\nexport function diffFindings(\n previous: PersistedFinding[],\n current: PersistedFinding[],\n policy: DiffPolicy = {},\n): FindingsDiff {\n const isMaterial = policy.isMaterial ?? defaultIsMaterial\n const prevById = new Map(previous.map((f) => [f.finding_id, f]))\n const curById = new Map(current.map((f) => [f.finding_id, f]))\n\n const appeared: PersistedFinding[] = []\n const disappeared: PersistedFinding[] = []\n const persisted: PersistedFinding[] = []\n const changed: FindingsDiff['changed'] = []\n\n for (const [id, cur] of curById) {\n const prev = prevById.get(id)\n if (!prev) {\n appeared.push(cur)\n continue\n }\n if (isMaterial(prev, cur)) {\n changed.push({ previous: prev, current: cur })\n } else {\n persisted.push(cur)\n }\n }\n for (const [id, prev] of prevById) {\n if (!curById.has(id)) disappeared.push(prev)\n }\n return { appeared, disappeared, persisted, changed }\n}\n","/**\n * Semantic concept judge — \"does the built artifact actually implement\n * the features the user asked for?\"\n *\n * Distinct from the domain/code/coherence judges in `judges.ts`:\n * - those judges score free-form conversational agent outputs along\n * quality dimensions (accuracy, depth, etc.)\n * - this judge scores a *built artifact* (served HTML + source files)\n * against an explicit list of expected concepts, returning per-concept\n * {present, score 0-10, evidence, severity}.\n *\n * The judge is strict about distinguishing (a) a working implementation\n * from (b) a keyword-present stub. \"// TODO: mint button\" is NOT present.\n * Only real, functional, wired-up code counts.\n *\n * Use via {@link createSemanticConceptJudge} or directly via\n * {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM\n * or JSON-parse errors so the caller can treat that as \"layer skipped\"\n * rather than \"layer failed\" in a multi-layer pipeline.\n */\n\nimport type { ChatClient } from './analyst/chat-client'\nimport { paidJsonChat } from './chat-json-call'\nimport {\n CostLedger,\n type CostLedgerHandle,\n type CostReceipt,\n type CustomTokenPricing,\n} from './cost-ledger'\nimport type { LlmCallRequest } from './llm-client'\nimport type { Severity } from './multi-layer-verifier'\n\n// ─── Types ──────────────────────────────────────────────────────────────\n\n/**\n * Implementation complexity class for weighted scoring.\n *\n * - `render` (default): the concept is a UI surface that displays static\n * data — render a list, show a counter, lay out a button. Single-file\n * work, no external integration.\n * - `integrate`: the concept requires wiring a real external system —\n * wallet connect (wagmi + RainbowKit + chain config), payment provider\n * (Stripe Elements + intent + webhook), an API client with auth.\n * Multi-file, library-knowledge, runtime correctness matters.\n * - `compute`: the concept requires algorithmic work — solver, simulator,\n * constraint propagation, ML inference. Correctness > UI polish.\n *\n * Default weights (when applied via `weightConcepts: 'complexity'`):\n * render=1.0, integrate=2.0, compute=2.5\n *\n * Cross-vertical scoring without complexity weighting silently inflates\n * the rate of UI-heavy verticals (healthcare, fintech dashboards) vs\n * integration-heavy verticals (DeFi, wallets) — all concepts treated\n * equally even though the agent does 2-3x the work for `integrate`.\n */\nexport type ConceptComplexity = 'render' | 'integrate' | 'compute'\n\nexport interface ConceptSpec {\n name: string\n /** Short hints that help the judge; not used for matching. */\n keywords?: string[]\n /** Optional explicit weight; default 1.0. Overrides complexity-derived weight. */\n weight?: number\n /** Implementation complexity class. Default `render`. */\n complexity?: ConceptComplexity\n}\n\nexport interface ConceptFinding {\n concept: string\n present: boolean\n /** 0..10. 10 = production-ready; 7 = functional thin; 4 = partial; 0 = absent. */\n score: number\n evidence: string\n severity: Severity\n}\n\nexport interface SemanticConceptJudgeInput {\n /** Full natural-language prompt the agent was handed. */\n userRequest: string\n /** Rendered HTML the preview returns (UI artifacts). Optional. */\n servedHtml?: string\n /** Top-level source files from the agent's workdir. */\n sourceFiles: Array<{ path: string; content: string }>\n /** The expected concept list. */\n expectedConcepts: ConceptSpec[]\n /** Free-form metadata (id, difficulty) to inject into the prompt. */\n artifactLabel?: string\n artifactDescription?: string\n}\n\nexport interface SemanticConceptJudgeResult {\n kind: 'semantic-concept'\n version: string\n /** Normalized 0..1 score — mean of per-concept scores / 10. */\n score: number\n presentCount: number\n totalCount: number\n findings: ConceptFinding[]\n summary: string\n durationMs: number\n costUsd: number | null\n /** False on LLM/JSON error — treat as \"skipped / unable to judge\" in pipelines. */\n available: boolean\n error?: string\n}\n\n/**\n * Score-aggregation strategy. `mean` averages 0-10 scores uniformly.\n * `complexity` applies the default weight table (render=1, integrate=2,\n * compute=2.5) unless a concept has an explicit `weight`. `explicit`\n * honors only `weight` (defaulting to 1 for unspecified).\n */\nexport type ConceptWeightStrategy = 'mean' | 'complexity' | 'explicit'\n\nexport const DEFAULT_COMPLEXITY_WEIGHTS: Record<ConceptComplexity, number> = {\n render: 1.0,\n integrate: 2.0,\n compute: 2.5,\n}\n\nexport interface SemanticConceptJudgeOptions {\n /** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */\n model?: string\n /** Per-call timeout. Default 300s. */\n timeoutMs?: number\n /** Provider-enforced output limit. Default 16000. */\n maxTokens?: number\n /** Pipeline budget for the prompt (source blob truncation). Default 45000. */\n maxSourceChars?: number\n /** Per-file cap before inclusion. Default 20000. */\n maxPerFileChars?: number\n /** HTML cap. Default 30000. */\n maxHtmlChars?: number\n /** Caller-owned transport. Required: agent-eval executes no paid model. */\n chat: ChatClient\n /** Endpoint rates used when the transport reports no billed amount. */\n pricing?: CustomTokenPricing\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /**\n * Score aggregation strategy. Default `mean` — uniform average across\n * concepts. Cross-vertical comparisons should use `complexity` to\n * neutralize the integrate-vs-render asymmetry.\n */\n weightConcepts?: ConceptWeightStrategy\n /** Override the default complexity → weight table. */\n complexityWeights?: Partial<Record<ConceptComplexity, number>>\n}\n\n// ─── Prompt assembly ────────────────────────────────────────────────────\n\nexport const SEMANTIC_CONCEPT_JUDGE_VERSION = 'semantic-concept-judge-v1-2026-04-24'\n\nconst DEFAULT_MAX_SOURCE = 45_000\nconst DEFAULT_MAX_HTML = 30_000\nconst DEFAULT_MAX_PER_FILE = 20_000\nconst DEFAULT_TIMEOUT = 300_000\nconst DEFAULT_MAX_TOKENS = 16_000\nconst DEFAULT_MODEL = 'claude-sonnet-4-6'\n\nconst SEMANTIC_SCHEMA = {\n type: 'object',\n additionalProperties: false,\n required: ['summary', 'concepts'],\n properties: {\n summary: { type: 'string', minLength: 20, maxLength: 600 },\n concepts: {\n type: 'array',\n minItems: 1,\n items: {\n type: 'object',\n additionalProperties: false,\n required: ['concept', 'present', 'score', 'evidence', 'severity'],\n properties: {\n concept: { type: 'string', minLength: 1, maxLength: 120 },\n present: { type: 'boolean' },\n score: { type: 'number', minimum: 0, maximum: 10 },\n evidence: { type: 'string', minLength: 5, maxLength: 400 },\n severity: { type: 'string', enum: ['critical', 'major', 'minor', 'info'] },\n },\n },\n },\n },\n}\n\nfunction truncate(body: string, cap: number, label: string): string {\n if (body.length <= cap) return body\n return `${body.slice(0, cap)}\\n… [truncated ${body.length - cap} chars of ${label}]`\n}\n\nfunction buildPrompt(\n input: SemanticConceptJudgeInput,\n opts: { maxPerFileChars: number; maxSourceChars: number; maxHtmlChars: number },\n): string {\n const sourceBlob = input.sourceFiles\n .filter((f) => f.content.length <= opts.maxPerFileChars)\n .map((f) => `--- FILE: ${f.path} ---\\n${f.content}`)\n .join('\\n\\n')\n\n const html = input.servedHtml ?? ''\n\n return `You are a strict code-review judge evaluating whether an agent's 0-to-1 build actually implements the features the user asked for.\n\nYou MUST distinguish:\n (a) WORKING code that implements the concept (rendered UI, wired handler, real API call),\n (b) KEYWORD-PRESENT stub (comments mentioning the concept, variable names, TODOs),\n (c) ABSENT (concept nowhere).\n\nA comment like \"// TODO: add mint button\" is NOT present — score 2-3. Only count a concept as present if there is real functional code: a rendered component, a call handler wired to state or a network call, a computed value actually used.\n\nUSER REQUEST (what the agent was asked to build):\n${input.userRequest}\n\n${input.artifactLabel ? `ARTIFACT METADATA:\\n name: ${input.artifactLabel}\\n description: ${input.artifactDescription ?? ''}\\n\\n` : ''}EXPECTED CONCEPTS (each must be graded independently):\n${input.expectedConcepts\n .map(\n (c, i) =>\n ` ${i + 1}. \"${c.name}\"${c.keywords?.length ? ` — hints: [${c.keywords.slice(0, 6).join(' | ')}]` : ''}`,\n )\n .join('\\n')}\n\n${html ? `SERVED HTML (what the preview returns when hit):\\n${truncate(html, opts.maxHtmlChars, 'HTML')}\\n\\n` : ''}SOURCE FILES (the agent's workdir):\n${truncate(sourceBlob, opts.maxSourceChars, 'source')}\n\nFor EACH concept, return:\n - concept: the concept name as given (match exactly)\n - present: boolean — does a working implementation exist?\n - score: 0-10 — 10 = production-ready; 7 = functional but thin; 4 = partial/stubbed; 2 = keyword-only comment; 0 = absent\n - evidence: cite \"<file>:<line>\" or \"served-html:<selector>\" pointing at the strongest supporting code. If the concept is absent or stubbed, explain what's missing.\n - severity:\n \"info\" when present: true AND score >= 7\n \"minor\" when present: true AND 4 <= score < 7\n \"major\" when present: false OR score < 4\n \"critical\" when the concept is not only absent but a core user flow depends on it\n\nAlso produce a \"summary\" (one sentence, 20-600 chars): overall verdict on whether this is a shippable implementation of the user request vs a keyword-dense placeholder.\n\nBE SKEPTICAL. Keyword matching already passed — your job is to catch what keyword matching misses. If the agent shipped a working build, say so. If it shipped a stub, say so. Don't grade on effort.\n\nReturn STRICT JSON. No prose outside the JSON.`\n}\n\n// ─── Runner ─────────────────────────────────────────────────────────────\n\n/**\n * Run the semantic concept judge. Soft-fails to available=false on\n * LLM/JSON errors — callers in a MultiLayerVerifier pipeline can treat\n * that as \"skip\" rather than \"fail.\"\n */\nexport async function runSemanticConceptJudge(\n input: SemanticConceptJudgeInput,\n options: SemanticConceptJudgeOptions,\n): Promise<SemanticConceptJudgeResult> {\n const start = Date.now()\n const totalCount = input.expectedConcepts.length\n\n if (totalCount === 0) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount: 0,\n findings: [],\n summary: 'no expected concepts declared',\n durationMs: 0,\n costUsd: null,\n available: false,\n error: 'no expected concepts declared',\n }\n }\n\n const opts = {\n chat: options.chat,\n model: options.model ?? options.chat.defaultModel ?? DEFAULT_MODEL,\n timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,\n maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,\n maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,\n maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,\n maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,\n ...(options.pricing ? { pricing: options.pricing } : {}),\n costLedger: options.costLedger ?? new CostLedger(),\n costPhase: options.costPhase ?? 'judge.semantic-concept',\n costTags: options.costTags ?? {},\n signal: options.signal ?? new AbortController().signal,\n weightConcepts: options.weightConcepts ?? 'mean',\n complexityWeights: { ...DEFAULT_COMPLEXITY_WEIGHTS, ...(options.complexityWeights ?? {}) },\n }\n\n // Build a name → weight map for aggregation. Mean strategy keeps every\n // weight at 1 (uniform average). Complexity strategy reads the table\n // and lets an explicit `weight` override. Explicit strategy uses ONLY\n // the spec's `weight` (defaulting to 1).\n const weightForConcept = (spec: ConceptSpec): number => {\n if (opts.weightConcepts === 'mean') return 1\n if (spec.weight != null) return spec.weight\n if (opts.weightConcepts === 'complexity') {\n return opts.complexityWeights[spec.complexity ?? 'render'] ?? 1\n }\n return 1\n }\n const weightByName = new Map<string, number>(\n input.expectedConcepts.map((c) => [c.name, weightForConcept(c)]),\n )\n\n let receipt: CostReceipt | undefined\n try {\n const request = {\n model: opts.model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You are a strict code-review judge. Return strict JSON only. No prose outside the JSON. A keyword in a comment is NOT a working implementation.',\n },\n { role: 'user' as const, content: buildPrompt(input, opts) },\n ],\n jsonSchema: { name: 'semantic_concept_judge', schema: SEMANTIC_SCHEMA },\n temperature: 0,\n maxTokens: opts.maxTokens,\n timeoutMs: opts.timeoutMs,\n } satisfies LlmCallRequest\n const paid = await paidJsonChat<{ summary: string; concepts: ConceptFinding[] }>({\n chat: opts.chat,\n request,\n ledger: opts.costLedger,\n channel: 'judge',\n phase: opts.costPhase,\n actor: 'semantic-concept',\n tags: opts.costTags,\n signal: opts.signal,\n ...(opts.pricing ? { pricing: opts.pricing } : {}),\n })\n receipt = paid.receipt\n if (!paid.succeeded) throw paid.error\n const { value } = paid\n\n if (!value?.concepts || !Array.isArray(value.concepts)) {\n throw new Error('judge returned malformed response — expected array under \"concepts\"')\n }\n\n const findings: ConceptFinding[] = value.concepts.map((c) => ({\n concept: String(c.concept),\n present: Boolean(c.present),\n score: Math.max(0, Math.min(10, Number(c.score ?? 0))),\n evidence: String(c.evidence ?? ''),\n severity: (['critical', 'major', 'minor', 'info'] as const).includes(c.severity)\n ? c.severity\n : 'info',\n }))\n\n const presentCount = findings.filter((f) => f.present && f.score >= 7).length\n let weightSum = 0\n let weightedScoreSum = 0\n for (const f of findings) {\n const w = weightByName.get(f.concept) ?? 1\n weightSum += w\n weightedScoreSum += w * f.score\n }\n const scoreAvg =\n weightSum > 0\n ? weightedScoreSum / weightSum\n : findings.reduce((a, f) => a + f.score, 0) / Math.max(1, findings.length)\n\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: Number((scoreAvg / 10).toFixed(3)),\n presentCount,\n totalCount,\n findings,\n summary: String(value.summary ?? ''),\n durationMs: Date.now() - start,\n costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,\n available: true,\n }\n } catch (err) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount,\n findings: [],\n summary: '',\n durationMs: Date.now() - start,\n costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,\n available: false,\n error: err instanceof Error ? err.message : String(err),\n }\n }\n}\n\n/**\n * Factory: pin LLM options once, return a closure that accepts inputs.\n * Convenient for pipelines that want to share a single LlmClient config.\n */\nexport function createSemanticConceptJudge(\n options: SemanticConceptJudgeOptions,\n): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult> {\n return (input) => runSemanticConceptJudge(input, options)\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAM,0BAAU,IAAI,IAAmB;AAEvC,SAAS,SAAS,MAAqB;CACrC,IAAI,IAAI,QAAQ,IAAI,IAAI;CACxB,IAAI,CAAC,GAAG;EACN,IAAI,IAAI,MAAM;EACd,QAAQ,IAAI,MAAM,CAAC;CACrB;CACA,OAAO;AACT;AAEA,IAAa,sBAAb,MAAiC;CAEH;CAD5B;CACA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,QAAQ,SAAS,IAAI;EAC1B,IAAI,CAAC,WAAW,QAAQ,IAAI,CAAC,GAC3B,UAAU,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAEhD;CAEA,MAAM,OAAO,OAA+B;EAC1C,MAAM,OAAO,GAAG,KAAK,UAAU,KAAK,EAAE;EACtC,MAAM,KAAK,MAAM,mBAAmB;GAClC,eAAe,KAAK,MAAM,IAAI;EAChC,CAAC;CACH;AACF;;;;;;;;;;;;;;;;;;;;;;ACPA,IAAa,gBAAb,MAA2B;CAGG;CAF5B;CAEA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,WAAW,IAAI,oBAAoB,IAAI;CAC9C;CAEA,MAAM,OAAO,OAAe,UAA2C;EACrE,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,MAAwB;IAAE,GAAG;IAAG,QAAQ;GAAM;GACpD,MAAM,KAAK,SAAS,OAAO,GAAG;EAChC;CACF;;CAGA,UAA8B;EAC5B,IAAI,CAAC,WAAW,KAAK,IAAI,GAAG,OAAO,CAAC;EACpC,MAAM,MAAM,aAAa,KAAK,MAAM,MAAM;EAC1C,IAAI,CAAC,KAAK,OAAO,CAAC;EAClB,MAAM,MAA0B,CAAC;EACjC,KAAK,MAAM,QAAQ,IAAI,MAAM,IAAI,GAAG;GAClC,IAAI,CAAC,MAAM;GACX,IAAI;IACF,IAAI,KAAK,KAAK,MAAM,IAAI,CAAqB;GAC/C,QAAQ,CAGR;EACF;EACA,OAAO;CACT;;CAGA,QAAQ,OAAmC;EACzC,OAAO,KAAK,QAAQ,CAAC,CAAC,QAAQ,MAAM,EAAE,WAAW,KAAK;CACxD;AACF;;;;;AAiCA,SAAgB,kBAAkB,GAAmB,GAA4B;CAC/E,IAAI,EAAE,aAAa,EAAE,UAAU,OAAO;CACtC,IAAI,KAAK,KAAK,EAAE,cAAc,MAAM,EAAE,cAAc,EAAE,IAAI,KAAM,OAAO;CACvE,IAAI,EAAE,cAAc,WAAW,EAAE,cAAc,QAAQ,OAAO;CAC9D,OAAO;AACT;;;;;AAMA,SAAgB,aACd,UACA,SACA,SAAqB,CAAC,GACR;CACd,MAAM,aAAa,OAAO,cAAc;CACxC,MAAM,WAAW,IAAI,IAAI,SAAS,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAC/D,MAAM,UAAU,IAAI,IAAI,QAAQ,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAE7D,MAAM,WAA+B,CAAC;CACtC,MAAM,cAAkC,CAAC;CACzC,MAAM,YAAgC,CAAC;CACvC,MAAM,UAAmC,CAAC;CAE1C,KAAK,MAAM,CAAC,IAAI,QAAQ,SAAS;EAC/B,MAAM,OAAO,SAAS,IAAI,EAAE;EAC5B,IAAI,CAAC,MAAM;GACT,SAAS,KAAK,GAAG;GACjB;EACF;EACA,IAAI,WAAW,MAAM,GAAG,GACtB,QAAQ,KAAK;GAAE,UAAU;GAAM,SAAS;EAAI,CAAC;OAE7C,UAAU,KAAK,GAAG;CAEtB;CACA,KAAK,MAAM,CAAC,IAAI,SAAS,UACvB,IAAI,CAAC,QAAQ,IAAI,EAAE,GAAG,YAAY,KAAK,IAAI;CAE7C,OAAO;EAAE;EAAU;EAAa;EAAW;CAAQ;AACrD;;;AC9BA,MAAa,6BAAgE;CAC3E,QAAQ;CACR,WAAW;CACX,SAAS;AACX;AAmCA,MAAa,iCAAiC;AAE9C,MAAM,qBAAqB;AAC3B,MAAM,mBAAmB;AACzB,MAAM,uBAAuB;AAC7B,MAAM,kBAAkB;AACxB,MAAM,qBAAqB;AAC3B,MAAM,gBAAgB;AAEtB,MAAM,kBAAkB;CACtB,MAAM;CACN,sBAAsB;CACtB,UAAU,CAAC,WAAW,UAAU;CAChC,YAAY;EACV,SAAS;GAAE,MAAM;GAAU,WAAW;GAAI,WAAW;EAAI;EACzD,UAAU;GACR,MAAM;GACN,UAAU;GACV,OAAO;IACL,MAAM;IACN,sBAAsB;IACtB,UAAU;KAAC;KAAW;KAAW;KAAS;KAAY;IAAU;IAChE,YAAY;KACV,SAAS;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACxD,SAAS,EAAE,MAAM,UAAU;KAC3B,OAAO;MAAE,MAAM;MAAU,SAAS;MAAG,SAAS;KAAG;KACjD,UAAU;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACzD,UAAU;MAAE,MAAM;MAAU,MAAM;OAAC;OAAY;OAAS;OAAS;MAAM;KAAE;IAC3E;GACF;EACF;CACF;AACF;AAEA,SAAS,SAAS,MAAc,KAAa,OAAuB;CAClE,IAAI,KAAK,UAAU,KAAK,OAAO;CAC/B,OAAO,GAAG,KAAK,MAAM,GAAG,GAAG,EAAE,iBAAiB,KAAK,SAAS,IAAI,YAAY,MAAM;AACpF;AAEA,SAAS,YACP,OACA,MACQ;CACR,MAAM,aAAa,MAAM,YACtB,QAAQ,MAAM,EAAE,QAAQ,UAAU,KAAK,eAAe,CAAC,CACvD,KAAK,MAAM,aAAa,EAAE,KAAK,QAAQ,EAAE,SAAS,CAAC,CACnD,KAAK,MAAM;CAEd,MAAM,OAAO,MAAM,cAAc;CAEjC,OAAO;;;;;;;;;;EAUP,MAAM,YAAY;;EAElB,MAAM,gBAAgB,+BAA+B,MAAM,cAAc,mBAAmB,MAAM,uBAAuB,GAAG,QAAQ,GAAG;EACvI,MAAM,iBACL,KACE,GAAG,MACF,KAAK,IAAI,EAAE,KAAK,EAAE,KAAK,GAAG,EAAE,UAAU,SAAS,cAAc,EAAE,SAAS,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,KAAK,EAAE,KAAK,IACzG,CAAC,CACA,KAAK,IAAI,EAAE;;EAEZ,OAAO,qDAAqD,SAAS,MAAM,KAAK,cAAc,MAAM,EAAE,QAAQ,GAAG;EACjH,SAAS,YAAY,KAAK,gBAAgB,QAAQ,EAAE;;;;;;;;;;;;;;;;;;AAkBtD;;;;;;AASA,eAAsB,wBACpB,OACA,SACqC;CACrC,MAAM,QAAQ,KAAK,IAAI;CACvB,MAAM,aAAa,MAAM,iBAAiB;CAE1C,IAAI,eAAe,GACjB,OAAO;EACL,MAAM;EACN,SAAS;EACT,OAAO;EACP,cAAc;EACd,YAAY;EACZ,UAAU,CAAC;EACX,SAAS;EACT,YAAY;EACZ,SAAS;EACT,WAAW;EACX,OAAO;CACT;CAGF,MAAM,OAAO;EACX,MAAM,QAAQ;EACd,OAAO,QAAQ,SAAS,QAAQ,KAAK,gBAAgB;EACrD,WAAW,QAAQ,aAAa;EAChC,WAAW,QAAQ,aAAa;EAChC,gBAAgB,QAAQ,kBAAkB;EAC1C,iBAAiB,QAAQ,mBAAmB;EAC5C,cAAc,QAAQ,gBAAgB;EACtC,GAAI,QAAQ,UAAU,EAAE,SAAS,QAAQ,QAAQ,IAAI,CAAC;EACtD,YAAY,QAAQ,cAAc,IAAI,WAAW;EACjD,WAAW,QAAQ,aAAa;EAChC,UAAU,QAAQ,YAAY,CAAC;EAC/B,QAAQ,QAAQ,UAAU,IAAI,gBAAgB,CAAC,CAAC;EAChD,gBAAgB,QAAQ,kBAAkB;EAC1C,mBAAmB;GAAE,GAAG;GAA4B,GAAI,QAAQ,qBAAqB,CAAC;EAAG;CAC3F;CAMA,MAAM,oBAAoB,SAA8B;EACtD,IAAI,KAAK,mBAAmB,QAAQ,OAAO;EAC3C,IAAI,KAAK,UAAU,MAAM,OAAO,KAAK;EACrC,IAAI,KAAK,mBAAmB,cAC1B,OAAO,KAAK,kBAAkB,KAAK,cAAc,aAAa;EAEhE,OAAO;CACT;CACA,MAAM,eAAe,IAAI,IACvB,MAAM,iBAAiB,KAAK,MAAM,CAAC,EAAE,MAAM,iBAAiB,CAAC,CAAC,CAAC,CACjE;CAEA,IAAI;CACJ,IAAI;EACF,MAAM,UAAU;GACd,OAAO,KAAK;GACZ,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IAAE,MAAM;IAAiB,SAAS,YAAY,OAAO,IAAI;GAAE,CAC7D;GACA,YAAY;IAAE,MAAM;IAA0B,QAAQ;GAAgB;GACtE,aAAa;GACb,WAAW,KAAK;GAChB,WAAW,KAAK;EAClB;EACA,MAAM,OAAO,MAAM,aAA8D;GAC/E,MAAM,KAAK;GACX;GACA,QAAQ,KAAK;GACb,SAAS;GACT,OAAO,KAAK;GACZ,OAAO;GACP,MAAM,KAAK;GACX,QAAQ,KAAK;GACb,GAAI,KAAK,UAAU,EAAE,SAAS,KAAK,QAAQ,IAAI,CAAC;EAClD,CAAC;EACD,UAAU,KAAK;EACf,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;EAChC,MAAM,EAAE,UAAU;EAElB,IAAI,CAAC,OAAO,YAAY,CAAC,MAAM,QAAQ,MAAM,QAAQ,GACnD,MAAM,IAAI,MAAM,uEAAqE;EAGvF,MAAM,WAA6B,MAAM,SAAS,KAAK,OAAO;GAC5D,SAAS,OAAO,EAAE,OAAO;GACzB,SAAS,QAAQ,EAAE,OAAO;GAC1B,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,IAAI,OAAO,EAAE,SAAS,CAAC,CAAC,CAAC;GACrD,UAAU,OAAO,EAAE,YAAY,EAAE;GACjC,UAAW;IAAC;IAAY;IAAS;IAAS;GAAM,CAAC,CAAW,SAAS,EAAE,QAAQ,IAC3E,EAAE,WACF;EACN,EAAE;EAEF,MAAM,eAAe,SAAS,QAAQ,MAAM,EAAE,WAAW,EAAE,SAAS,CAAC,CAAC,CAAC;EACvE,IAAI,YAAY;EAChB,IAAI,mBAAmB;EACvB,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,IAAI,aAAa,IAAI,EAAE,OAAO,KAAK;GACzC,aAAa;GACb,oBAAoB,IAAI,EAAE;EAC5B;EACA,MAAM,WACJ,YAAY,IACR,mBAAmB,YACnB,SAAS,QAAQ,GAAG,MAAM,IAAI,EAAE,OAAO,CAAC,IAAI,KAAK,IAAI,GAAG,SAAS,MAAM;EAE7E,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO,QAAQ,WAAW,GAAA,CAAI,QAAQ,CAAC,CAAC;GACxC;GACA;GACA;GACA,SAAS,OAAO,MAAM,WAAW,EAAE;GACnC,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,KAAK,QAAQ,cAAc,OAAO,KAAK,QAAQ;GACxD,WAAW;EACb;CACF,SAAS,KAAK;EACZ,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO;GACP,cAAc;GACd;GACA,UAAU,CAAC;GACX,SAAS;GACT,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,WAAW,CAAC,QAAQ,cAAc,QAAQ,UAAU;GAC7D,WAAW;GACX,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;EACxD;CACF;AACF"}
@@ -1,7 +1,7 @@
1
- import { r as manifestContentDigest } from "./pre-registration-KN9jkh58.js";
2
- import { c as mulberry32 } from "./internal-BDHPCnjk.js";
1
+ import { r as manifestContentDigest } from "./pre-registration-D94b7Of5.js";
2
+ import { t as mulberry32 } from "./random-Dn5fPWkt.js";
3
3
  import { t as eProcess } from "./sequential-eprocess-D1jKoihe.js";
4
- import { o as pairHoldout } from "./power-preflight-DEw-uC7q.js";
4
+ import { o as pairHoldout } from "./power-preflight-CFXm0Vjo.js";
5
5
  //#region src/campaign/gates/sequential.ts
6
6
  /**
7
7
  * Anytime-valid sequential promotion gate — an e-process (betting
@@ -322,4 +322,4 @@ function sequentialDecide(options = {}) {
322
322
  //#endregion
323
323
  export { sequentialPairedGate as n, sequentialDecide as t };
324
324
 
325
- //# sourceMappingURL=sequential-rYW-Ophm.js.map
325
+ //# sourceMappingURL=sequential-B51qAYE4.js.map