@tangle-network/agent-eval 0.144.5 → 0.144.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (257) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
  6. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
  7. package/dist/analyst/index.d.ts +471 -89
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +24 -5
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  12. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  13. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  14. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  15. package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
  16. package/dist/benchmark-command-BCafwNrf.js.map +1 -0
  17. package/dist/benchmarks/index.d.ts +1 -1
  18. package/dist/benchmarks/index.js +1 -1
  19. package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
  20. package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
  21. package/dist/campaign/index.d.ts +8 -6
  22. package/dist/campaign/index.js +5 -3
  23. package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
  24. package/dist/campaign-Tdy3h62h.js.map +1 -0
  25. package/dist/cli.js +2 -2
  26. package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
  27. package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  28. package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
  29. package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  30. package/dist/contract/index.d.ts +11 -10
  31. package/dist/contract/index.d.ts.map +1 -1
  32. package/dist/contract/index.js +7 -6
  33. package/dist/contract/index.js.map +1 -1
  34. package/dist/control.d.ts +2 -2
  35. package/dist/control.js +1 -1
  36. package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
  37. package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
  38. package/dist/counterfactual-CWPTrMH7.js +126 -0
  39. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  40. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  41. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  42. package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
  43. package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
  44. package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  45. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  46. package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
  47. package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  48. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  49. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  50. package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
  51. package/dist/engine-nB64f48I.d.ts.map +1 -0
  52. package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
  53. package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
  54. package/dist/errors-D-LKuDhb.js.map +1 -1
  55. package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
  56. package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  57. package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  58. package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  59. package/dist/exec-BLtYZdWo.js +49 -0
  60. package/dist/exec-BLtYZdWo.js.map +1 -0
  61. package/dist/experiment/index.d.ts +802 -0
  62. package/dist/experiment/index.d.ts.map +1 -0
  63. package/dist/experiment/index.js +1108 -0
  64. package/dist/experiment/index.js.map +1 -0
  65. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  66. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  67. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  68. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  69. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
  70. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  71. package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
  72. package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  73. package/dist/fuzz.d.ts +2 -2
  74. package/dist/hosted/index.d.ts +3 -3
  75. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  76. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  77. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  78. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  79. package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
  80. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  81. package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
  82. package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
  83. package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
  84. package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  85. package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
  86. package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  87. package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
  88. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  89. package/dist/index.d.ts +252 -451
  90. package/dist/index.d.ts.map +1 -1
  91. package/dist/index.js +271 -751
  92. package/dist/index.js.map +1 -1
  93. package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
  94. package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  95. package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
  96. package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  97. package/dist/integrity-MLzHOfV9.js +141 -0
  98. package/dist/integrity-MLzHOfV9.js.map +1 -0
  99. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  100. package/dist/ledger-core/index.d.ts +1 -1
  101. package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
  102. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  103. package/dist/matrix/index.d.ts +2 -2
  104. package/dist/meta-eval/index.d.ts +2 -2
  105. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  106. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  107. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  108. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  109. package/dist/multishot/index.d.ts +3 -3
  110. package/dist/openapi.json +1 -1
  111. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  112. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  113. package/dist/pipelines/index.d.ts +2 -1
  114. package/dist/pipelines/index.d.ts.map +1 -1
  115. package/dist/pre-registration-DakwTRXk.js +96 -0
  116. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  117. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  118. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  119. package/dist/prime-protocol-BfSalTfR.js +453 -0
  120. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  121. package/dist/profile-cell.d.ts +1 -1
  122. package/dist/profile-cell.js +242 -1
  123. package/dist/profile-cell.js.map +1 -0
  124. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  125. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  126. package/dist/promotion-policy-CrLrmys8.js +682 -0
  127. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  128. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  129. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  130. package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
  131. package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  132. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  133. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  134. package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
  135. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  136. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  137. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  138. package/dist/reporting.d.ts +4 -4
  139. package/dist/reporting.js +2 -2
  140. package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
  141. package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  142. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  143. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  144. package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
  145. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  146. package/dist/rl.d.ts +19 -9
  147. package/dist/rl.d.ts.map +1 -1
  148. package/dist/rl.js +16 -7
  149. package/dist/rl.js.map +1 -1
  150. package/dist/rollout/index.d.ts +2 -2
  151. package/dist/rollout/index.js +2 -2
  152. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  153. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  154. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
  155. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
  156. package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
  157. package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  158. package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
  159. package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
  160. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  161. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  162. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  163. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  164. package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  165. package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  166. package/dist/sequential-D-BLJBKU.js +299 -0
  167. package/dist/sequential-D-BLJBKU.js.map +1 -0
  168. package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
  169. package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
  170. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  171. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  172. package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
  173. package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  174. package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
  175. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  176. package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
  177. package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
  178. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  179. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  180. package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  181. package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  182. package/dist/steps-BArUxhna.d.ts +51 -0
  183. package/dist/steps-BArUxhna.d.ts.map +1 -0
  184. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  185. package/dist/store-DNe_Uv1Q.js.map +1 -0
  186. package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
  187. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  188. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  189. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  190. package/dist/supervisor-run/index.d.ts +2 -2
  191. package/dist/supervisor-run/index.js +1 -1
  192. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  193. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  194. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  195. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  196. package/dist/trace-repair/index.d.ts +2102 -0
  197. package/dist/trace-repair/index.d.ts.map +1 -0
  198. package/dist/trace-repair/index.js +3878 -0
  199. package/dist/trace-repair/index.js.map +1 -0
  200. package/dist/traces.d.ts +6 -6
  201. package/dist/traces.js +3 -2
  202. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  203. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  204. package/dist/trajectory-replay/index.d.ts +781 -0
  205. package/dist/trajectory-replay/index.d.ts.map +1 -0
  206. package/dist/trajectory-replay/index.js +2103 -0
  207. package/dist/trajectory-replay/index.js.map +1 -0
  208. package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
  209. package/dist/types-D216SgwM.d.ts.map +1 -0
  210. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  211. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  212. package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
  213. package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  214. package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
  215. package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  216. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  217. package/dist/verdict-DExhxfgR.d.ts +201 -0
  218. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  219. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  220. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  221. package/dist/wire/index.d.ts +3 -3
  222. package/dist/wire/index.js +1 -1
  223. package/docs/building-doctrine.md +15 -0
  224. package/docs/charter.md +112 -0
  225. package/docs/experiment.md +104 -0
  226. package/docs/prime-analyst.md +1 -0
  227. package/docs/trace-analysis.md +26 -0
  228. package/docs/trace-repair-admission.md +194 -0
  229. package/docs/trace-repair-analyst-arms.md +121 -0
  230. package/docs/trace-repair-continuation.md +107 -0
  231. package/docs/trace-repair-grader.md +163 -0
  232. package/docs/trajectory-replay.md +110 -0
  233. package/docs/verification-strategies.md +103 -0
  234. package/package.json +21 -4
  235. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  236. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  237. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  238. package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
  239. package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
  240. package/dist/campaign-ClpnD7Ug.js.map +0 -1
  241. package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
  242. package/dist/index-CpxZSlB7.d.ts.map +0 -1
  243. package/dist/index-CsuAo2-J.d.ts.map +0 -1
  244. package/dist/integrity-fdt8XPAv.js.map +0 -1
  245. package/dist/llm-client-D3EoChAU.js.map +0 -1
  246. package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
  247. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  248. package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
  249. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  250. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  251. package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
  252. package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
  253. package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
  254. package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
  255. package/dist/types-BjMFz88h.d.ts.map +0 -1
  256. package/dist/verdict-Dps8_okt.d.ts +0 -37
  257. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
package/dist/index.js CHANGED
@@ -1,12 +1,13 @@
1
1
  import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
2
- import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-BciUCZoh.js";
2
+ import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-qgWLA6E9.js";
3
3
  import { a as LimitExceededError, c as ValidationError, i as JudgeError, l as VerificationError, n as CaptureIntegrityError, o as NotFoundError, r as ConfigError, s as ReplayError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
4
- import { _ as verifyManifest, a as assertRunAgentProfileCell, c as groupRunsByAgentProfileCell, d as validateAgentProfileCell, f as verifyAgentProfileCell, g as signManifest, h as hashJson, i as agentProfileCellKey, l as requireAgentProfileCell, m as evaluateHypothesis, n as AgentProfileCellValidationError, o as buildAgentInterfaceProfileCell, p as canonicalize, r as agentProfileCellHashMaterial, s as buildAgentProfileCell, t as AGENT_PROFILE_KINDS, u as toAgentProfileJson } from "./agent-profile-cell-CbfBm2g6.js";
4
+ import { a as verifyManifest, i as signManifest, n as evaluateHypothesis, r as hashJson, t as canonicalize } from "./pre-registration-DakwTRXk.js";
5
+ import { AGENT_PROFILE_KINDS, AgentProfileCellValidationError, agentProfileCellHashMaterial, agentProfileCellKey, assertRunAgentProfileCell, buildAgentInterfaceProfileCell, buildAgentProfileCell, groupRunsByAgentProfileCell, requireAgentProfileCell, toAgentProfileJson, validateAgentProfileCell, verifyAgentProfileCell } from "./profile-cell.js";
5
6
  import { a as estimateTokens, i as estimateCost, n as MetricsCollector, o as isModelPriced, r as TokenCounter, s as resolveModelPricing, t as MODEL_PRICING } from "./metrics-C9YY1OcL.js";
6
7
  import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-DMFxsLKr.js";
8
+ import { C as servedModelAcceptable, E as judgeFamily, S as normalizeModelId, T as assertCrossFamily, _ as ServedCrossFamilyError, a as assertLlmRoute, b as assertServedModels, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, g as PROBE_MAX_TOKENS, h as ModelSubstitutionError, i as LlmRouteAssertionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError, v as assertCrossFamilyServed, w as CrossFamilyError, x as checkServedModel, y as assertServedModel } from "./llm-client-DzvMUsS_.js";
7
9
  import { a as providerFromBaseUrl, i as defaultProviderRedactor, n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
8
- import { a as assertLlmRoute, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, i as LlmRouteAssertionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-D3EoChAU.js";
9
- import { C as createChatClient, S as computeTraceMetrics, _ as CONTROL_INTEGRITY_ANALYST, a as analystFindingDigest, c as completedAnalystReviewQuality, d as validateAnalystReviewDecisions, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, l as readAnalystReview, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, o as analystRunDigest, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, s as assertUniqueFindingIds, t as buildDefaultAnalystRegistry, u as snapshotAnalystRun, v as ControlIntegrityAnalyst, y as emitControlIntegrityFindings } from "./default-registry-RLNNoeEP.js";
10
+ import { C as createChatClient, S as computeTraceMetrics, _ as CONTROL_INTEGRITY_ANALYST, a as analystFindingDigest, c as completedAnalystReviewQuality, d as validateAnalystReviewDecisions, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, l as readAnalystReview, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, o as analystRunDigest, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, s as assertUniqueFindingIds, t as buildDefaultAnalystRegistry, u as snapshotAnalystRun, v as ControlIntegrityAnalyst, y as emitControlIntegrityFindings } from "./default-registry-BaQXW1Ow.js";
10
11
  import { INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OUTPUT_VALUE, RUN_COST_ATTR_KEYS, SPAN_KIND_ATTR_KEYS, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, applyLlmSpanOtlpAttributes, asNumber, contextInputTokens, firstNumberAttr } from "./trace-attributes.js";
11
12
  import { C as TraceNotFoundError, G as resolveTraceAnalystLimits, J as extractOtlpAttributes, K as asString, Q as readOtlpStatus, S as TraceFileTooLargeError, W as DEFAULT_TRACE_ANALYST_LIMITS, X as inferOtlpKind, Y as firstStringAttr, Z as projectOtlpFlatLine, _ as TraceAnalysisLimitError, b as TraceFileMalformedError, c as traceAnalystFunctionGroup, et as stringField, g as SpanNotFoundError, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst, it as traceSpanKindToOpenInferenceKind, l as createBoundedTraceAnalysisStore, m as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, n as renderPriorFindings, nt as classifyOtlpSpanRole, o as TRACE_ANALYST_TOOL_NAMESPACE, p as DEFAULT_TRACE_ANALYST_BUDGETS, r as renderUpstreamFindings, rt as isOtlpModelCall, s as buildTraceAnalysisToolDescriptors, t as createTraceAnalyst, tt as applyToolSpanOtlpAttributes, v as TraceAnalysisStoreContractError, x as TraceFileMissingError, y as TraceAnalysisValidationError } from "./kind-factory-BHIgPmzS.js";
12
13
  import { a as computeFindingId, o as makeFinding, s as makeProposalFinding } from "./usage-receipt-EVI8B8Xu.js";
@@ -14,42 +15,46 @@ import { c as SUPERVISOR_RUN_INTEGRITY_SCHEMA, n as supervisorRunRolloutLines, t
14
15
  import { c as assertMintedLines, f as isRolloutLine, i as ROLLOUT_SCHEMA, m as validateRolloutLine, p as isTrainableSplit, s as assertMinted, u as assertRolloutLine } from "./schema-C6DW4ZHR.js";
15
16
  import { a as scoreOrigin, n as observedScore, o as trainingReward, r as observedSplitScore, s as trainingScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
16
17
  import { a as clamp01, i as aggregateRunScore, r as DEFAULT_RUN_SCORE_WEIGHTS } from "./proposal-findings-2GIUo1et.js";
17
- import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-DKRtp2sY.js";
18
- import { _ as cachedJudge, b as fileVerdictCache, v as canonicalJson, x as inMemoryVerdictCache, y as contentHash } from "./single-run-lock-D5iN0Xzb.js";
19
- import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-19FQEMBK.js";
18
+ import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-Bmrq6yqU.js";
19
+ import { a as inMemoryVerdictCache, i as fileVerdictCache, n as canonicalJson, r as contentHash, t as cachedJudge } from "./verdict-cache-BCcOh0kF.js";
20
+ import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BiN49gK6.js";
20
21
  import { f as Mutex, p as mapConcurrent } from "./ledger-core-DXZIqu17.js";
21
- import { At as summarizeAgentReceiptIntegrity, B as surfaceContentHash, Ct as decidePairedPromotion, D as runCanaries, Dt as BackendIntegrityError, Et as pairedDeltaTest, G as DEFAULT_MUTATION_PRIMITIVES, K as buildReflectionPrompt, Mt as JudgeParseError, Ot as assertRealAgentReceipts, Tt as minimumPairsForPairedDeltaTest, _t as crowdingDistance, at as REFERENCE_EQUIVALENCE_JUDGE_VERSION, bt as paretoFrontierWithCrowding, ct as DEFAULT_RED_TEAM_CORPUS, dt as scoreRedTeamOutput, ft as toolNamesForRun, gt as llmJudge, ht as hashScenarios, it as REFERENCE_EQUIVALENCE_INPUT_LIMITS, jt as summarizeBackendIntegrity, kt as assertRealBackend, lt as redTeamDataset, mt as HoldoutLockedError, ot as createReferenceEquivalenceJudge, pt as Dataset, q as parseReflectionResponse, st as runReferenceEquivalenceJudge, ut as redTeamReport, vt as dominates, wt as pairedDecisionShape, xt as scalarScore, yt as paretoFrontier } from "./skillopt-optimization-method-C4FX42dy.js";
22
+ import { $ as runReferenceEquivalenceJudge, L as DEFAULT_MUTATION_PRIMITIVES, M as surfaceContentHash, Q as createReferenceEquivalenceJudge, R as buildReflectionPrompt, X as REFERENCE_EQUIVALENCE_INPUT_LIMITS, Z as REFERENCE_EQUIVALENCE_JUDGE_VERSION, _t as assertRealBackend, at as Dataset, bt as JudgeParseError, ct as llmJudge, dt as paretoFrontier, et as DEFAULT_RED_TEAM_CORPUS, ft as paretoFrontierWithCrowding, gt as assertRealAgentReceipts, ht as BackendIntegrityError, it as toolNamesForRun, lt as crowdingDistance, nt as redTeamReport, ot as HoldoutLockedError, pt as scalarScore, rt as scoreRedTeamOutput, st as hashScenarios, tt as redTeamDataset, ut as dominates, vt as summarizeAgentReceiptIntegrity, y as runCanaries, yt as summarizeBackendIntegrity, z as parseReflectionResponse } from "./skillopt-optimization-method-CQwZ-ZX8.js";
22
23
  import { $ as selfPreference, A as pairedRiskDifference, B as requiredSampleSize, C as mulberry32, D as pairedCohensDz, E as pairedBootstrap, F as partialCredit, G as wilson, H as weightedComposite, I as passAtK, J as normalCdf, K as studentTCdf, L as pearsonR, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, P as pairedTTest, Q as positionalBias, R as ranks, S as mcnemarRequiredN, T as pairedBinaryScale, U as weightedMean, V as spearmanR, W as wilcoxonSignedRank, X as calibrateJudgeContinuous, Y as calibrateJudge, Z as continuousAgreement, _ as interpretCliffs, a as MANN_WHITNEY_EXACT_MAX_WORK, b as mcnemar, c as bonferroni, d as confidenceInterval, et as verbosityBias, f as corpusInterRaterAgreement, g as interRaterReliability, h as holm, i as MANN_WHITNEY_EXACT_MAX_STATES, j as pairedRiskDifferenceExact, k as pairedMde, l as cliffsDelta, m as eProcess, n as DECISION_PAIRED_DELTA_STATISTIC, o as WILCOXON_EXACT_MAX_N, p as corpusInterRaterAgreementFromJudgeScores, r as DEFAULT_PERMUTATIONS, s as benjaminiHochberg, t as BOOTSTRAP_GATE_MIN_N, u as cohensD, v as isBinaryOutcomeVector, w as normalizeScores, x as mcnemarPower, y as mannWhitneyU, z as requiredPairedSampleSize } from "./statistics-ByxzSiOM.js";
23
24
  import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-iZ08VFMN.js";
25
+ import { a as improvementVerdict, i as gitProvenanceReader, n as computeExperimentStats, o as inMemoryExperimentStore, r as fileExperimentStore, s as clusteredPairedBinary, t as ExperimentTracker } from "./experiment-tracker-CnRICnMl.js";
24
26
  import { n as llmSpanFromProvider, t as TraceEmitter } from "./emitter-CPBAhxum.js";
25
27
  import { a as isRetrievalSpan, i as isLlmSpan, n as TRACE_SCHEMA_VERSION, o as isSandboxSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
26
- import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-CWN8-VsV.js";
27
- import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-Sl0xfkFv.js";
28
+ import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-DqOw5X6_.js";
29
+ import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-Dy39cbFF.js";
30
+ import { d as pairedDecisionShape, f as minimumPairsForPairedDeltaTest, p as pairedDeltaTest, u as decidePairedPromotion } from "./promotion-policy-CrLrmys8.js";
28
31
  import { i as toRewardRows, r as toJsonl, s as toSftRows } from "./exporters-q9iL-2Jf.js";
29
- import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-C2fD1cf4.js";
32
+ import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-2ECTXb2N.js";
30
33
  import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
31
- import { n as unmintableReasons, t as mintRolloutRows } from "./mint-DD-0oQTA.js";
32
- import { C as rollupSupervisorRuns, D as isUnavailable, E as SUPERVISOR_RUN_SCHEMA, O as showMeasured, b as readClaudeCodeSupervisorRun, c as writeSupervisorRunReport, d as readRuntimeSupervisorRun, f as runtimeSupervisorRunReader, h as renderSupervisorRunMarkdown, m as renderSupervisorRunHeadline, t as analyzeSupervisorRun, u as isRuntimeSupervisorRunDir, x as analyzeSupervisorRunSources, y as claudeCodeSupervisorRunReader } from "./supervisor-run-DiyQVczd.js";
33
- import { A as TRACE_ANALYST_ACTOR_DESCRIPTION, C as domainEvidencePattern, D as tokenizeDomainWords, E as scoreTraceInsightReadiness, O as traceAnalystOnRunComplete, S as describeTraceInsightScope, T as planTraceInsightQuestions, _ as otlpToTraceRunRecords, a as convertTraceStoresToOtlp, b as buildTraceInsightPrompt, c as otelRunCompleteHook, d as captureFetchToRawSink, f as ToolTraceMissingError, g as otlpToRunRecords, h as otlpRowsToTraceRunRecords, i as iterateRawCalls, j as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, k as analyzeTraces, l as OTEL_AGENT_EVAL_SCOPE, m as otlpRowsToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as toolSpansToTraceAnalysisStore, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as flattenOtlpExportToNdjson, w as inferDomainKeywords, x as defaultTraceInsightPanel, y as buildTraceInsightContext } from "./replay-GW61ezMW.js";
34
+ import { n as unmintableReasons, t as mintRolloutRows } from "./mint-CGEkzPLf.js";
35
+ import { C as rollupSupervisorRuns, D as isUnavailable, E as SUPERVISOR_RUN_SCHEMA, O as showMeasured, b as readClaudeCodeSupervisorRun, c as writeSupervisorRunReport, d as readRuntimeSupervisorRun, f as runtimeSupervisorRunReader, h as renderSupervisorRunMarkdown, m as renderSupervisorRunHeadline, t as analyzeSupervisorRun, u as isRuntimeSupervisorRunDir, x as analyzeSupervisorRunSources, y as claudeCodeSupervisorRunReader } from "./supervisor-run-D_sokXcO.js";
36
+ import { A as TRACE_ANALYST_ACTOR_DESCRIPTION, C as domainEvidencePattern, D as tokenizeDomainWords, E as scoreTraceInsightReadiness, O as traceAnalystOnRunComplete, S as describeTraceInsightScope, T as planTraceInsightQuestions, _ as otlpToTraceRunRecords, a as convertTraceStoresToOtlp, b as buildTraceInsightPrompt, c as otelRunCompleteHook, d as captureFetchToRawSink, f as ToolTraceMissingError, g as otlpToRunRecords, h as otlpRowsToTraceRunRecords, i as iterateRawCalls, j as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, k as analyzeTraces, l as OTEL_AGENT_EVAL_SCOPE, m as otlpRowsToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as toolSpansToTraceAnalysisStore, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as flattenOtlpExportToNdjson, w as inferDomainKeywords, x as defaultTraceInsightPanel, y as buildTraceInsightContext } from "./replay-DyBLaKFc.js";
34
37
  import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-CKtTpRhv.js";
35
38
  import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BW27f3XW.js";
36
- import { B as agentProfileModelId, G as createLlmCorrectnessChecker, H as harnessAxisOf, I as CODING_HARNESSES, J as verifyCompletion, K as createTokenRecallChecker, L as HARNESS_NATIVE_MODEL, R as agentProfileHash, U as extractProducedState, V as expandProfileAxes, W as completionVerdict, q as parseCorrectnessResponse, z as agentProfileId } from "./campaign-ClpnD7Ug.js";
39
+ import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-Tdy3h62h.js";
37
40
  import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-C-GocmIW.js";
38
41
  import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-Di7eEQ79.js";
39
- import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-DWIvOAGk.js";
40
- import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-9A5y7EsK.js";
42
+ import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-BNNK7irB.js";
43
+ import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-Lf-5I7xh.js";
41
44
  import { a as composeParsers, c as vitestTestParser, i as SubprocessSandboxDriver, n as DockerSandboxDriver, o as jestTestParser, r as SandboxHarness, s as pytestTestParser, t as runTestGradedScenario } from "./test-graded-scenario-JHcKQNpq.js";
42
- import { a as InMemoryTraceStore, i as FileSystemTraceStore, n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-fdt8XPAv.js";
45
+ import { n as InMemoryTraceStore, t as FileSystemTraceStore } from "./store-DNe_Uv1Q.js";
46
+ import { n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-MLzHOfV9.js";
43
47
  import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
44
48
  import { n as DEFAULT_RULES, r as classifyFailure, t as computeToolUseMetrics } from "./tool-use-metrics-DEGMKycK.js";
45
49
  import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
46
- import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-CWbsj0t4.js";
47
- import { t as runEvalCampaign } from "./eval-campaign-lZcDIwQM.js";
50
+ import { n as runCounterfactual, t as attributeCounterfactuals } from "./counterfactual-CWPTrMH7.js";
51
+ import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-CDSolHq7.js";
52
+ import { t as runEvalCampaign } from "./eval-campaign-DNjCvAm-.js";
48
53
  import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
49
54
  import { accessSync, appendFileSync, constants, cpSync, existsSync, mkdirSync, promises, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
50
55
  import { basename, delimiter, dirname, extname, isAbsolute, join, relative, resolve } from "node:path";
51
56
  import { createHash } from "node:crypto";
52
- import { execSync, spawnSync } from "node:child_process";
57
+ import { spawnSync } from "node:child_process";
53
58
  import { readFile } from "node:fs/promises";
54
59
  import { cpus } from "node:os";
55
60
  import { gzipSync } from "node:zlib";
@@ -423,6 +428,10 @@ async function executeScenario(chat, scenario, config) {
423
428
  receiptFromError: costReceiptFromLlmError
424
429
  });
425
430
  if (!paid.succeeded) throw paid.error;
431
+ assertServedModel(model, paid.value.servedModel, {
432
+ allowUnreported: true,
433
+ context: `executeScenario "${scenario.id}" turn ${i}`
434
+ });
426
435
  const rawContent = paid.value.content;
427
436
  if (typeof rawContent !== "string") throw new CaptureIntegrityError(`chat response for scenario "${scenario.id}" turn ${i} is malformed: expected content to be a string, got ${rawContent === null ? "null" : typeof rawContent}`);
428
437
  const content = rawContent;
@@ -1100,235 +1109,6 @@ async function runE2EWorkflow(client, name, workflow) {
1100
1109
  };
1101
1110
  }
1102
1111
  //#endregion
1103
- //#region src/clustered-paired-binary.ts
1104
- /**
1105
- * Paired binary comparison for work items nested inside independent clusters.
1106
- *
1107
- * Pairing is delegated to {@link pairArms}; this module adds the cluster-aware
1108
- * estimands and inference that task-level McNemar/bootstrap utilities cannot
1109
- * provide. Callers keep their own row shape through accessors, and every
1110
- * matched or unpaired result returns the original row object unchanged.
1111
- */
1112
- const DEFAULT_BOOTSTRAP_RESAMPLES = 1e4;
1113
- const DEFAULT_SIGN_FLIP_RESAMPLES = 1e5;
1114
- const MAX_RESAMPLES = 1e6;
1115
- const DEFAULT_EXACT_CLUSTER_LIMIT = 20;
1116
- const SIGN_FLIP_SEED_SALT = 2654435769;
1117
- /**
1118
- * Compare binary outcomes on matched work items while respecting independent
1119
- * clusters. The confidence interval resamples whole clusters and recomputes the
1120
- * task-weighted risk difference. The sign-flip test flips whole-cluster outcome
1121
- * totals and tests that same task-weighted estimand.
1122
- */
1123
- function clusteredPairedBinary(rows, options) {
1124
- const config = validateOptions(options);
1125
- const paired = pairArms(projectSelectedRows(rows, options), {
1126
- baselineArm: options.baselineArm,
1127
- treatmentArm: options.treatmentArm
1128
- });
1129
- const matchedPairs = paired.pairs.map((pair) => {
1130
- const baseline = pair.baseline;
1131
- const treatment = pair.treatment;
1132
- if (baseline.clusterKey !== treatment.clusterKey) throw new ValidationError(`clusteredPairedBinary: pairKey '${pair.pairKey}' rep ${pair.repIndex} crosses clusters ('${baseline.clusterKey}' vs '${treatment.clusterKey}')`);
1133
- return {
1134
- pairKey: pair.pairKey,
1135
- repIndex: pair.repIndex,
1136
- clusterKey: baseline.clusterKey,
1137
- baseline: baseline.original,
1138
- treatment: treatment.original,
1139
- baselinePass: baseline.pass,
1140
- treatmentPass: treatment.pass
1141
- };
1142
- });
1143
- const unpairedBaseline = paired.unpairedBaseline.map((row) => row.original);
1144
- const unpairedTreatment = paired.unpairedTreatment.map((row) => row.original);
1145
- if (matchedPairs.length === 0) return {
1146
- matchedPairs,
1147
- unpairedBaseline,
1148
- unpairedTreatment,
1149
- statistics: null
1150
- };
1151
- const clusters = summarizeClusters(matchedPairs);
1152
- const b10 = clusters.reduce((sum, cluster) => sum + cluster.b10, 0);
1153
- const b01 = clusters.reduce((sum, cluster) => sum + cluster.b01, 0);
1154
- const taskWeightedRiskDifference = (b10 - b01) / matchedPairs.length;
1155
- const equalClusterMean = mean$4(clusters.map((cluster) => cluster.meanDifference));
1156
- const bootstrap = clusters.length < 2 ? null : clusterBootstrap(clusters, config);
1157
- const signFlip = clusterSignFlip(clusters, config);
1158
- return {
1159
- matchedPairs,
1160
- unpairedBaseline,
1161
- unpairedTreatment,
1162
- statistics: {
1163
- nPairs: matchedPairs.length,
1164
- nClusters: clusters.length,
1165
- b10,
1166
- b01,
1167
- taskWeightedRiskDifference,
1168
- equalClusterMean,
1169
- clusters,
1170
- bootstrap,
1171
- signFlip
1172
- }
1173
- };
1174
- }
1175
- function validateOptions(options) {
1176
- assertNonEmptyString("baselineArm", options.baselineArm);
1177
- assertNonEmptyString("treatmentArm", options.treatmentArm);
1178
- if (options.baselineArm === options.treatmentArm) throw new ValidationError(`clusteredPairedBinary: baselineArm and treatmentArm are both '${options.baselineArm}'`);
1179
- if (!Number.isInteger(options.seed)) throw new ValidationError(`clusteredPairedBinary: seed must be an integer, got ${options.seed}`);
1180
- const confidence = options.confidence ?? .95;
1181
- if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new ValidationError(`clusteredPairedBinary: confidence must be in (0,1), got ${confidence}`);
1182
- const bootstrapResamples = options.bootstrapResamples ?? DEFAULT_BOOTSTRAP_RESAMPLES;
1183
- assertResampleCount("bootstrapResamples", bootstrapResamples);
1184
- const rawMinimumBootstrapResamples = 2 / (1 - confidence);
1185
- const minimumBootstrapResamples = Math.ceil(rawMinimumBootstrapResamples - Number.EPSILON * Math.max(1, rawMinimumBootstrapResamples) * 8);
1186
- if (bootstrapResamples < minimumBootstrapResamples) throw new ValidationError(`clusteredPairedBinary: bootstrapResamples must be at least ${minimumBootstrapResamples} for confidence ${confidence} so both interval tails are represented, got ${bootstrapResamples}`);
1187
- const signFlipResamples = options.signFlipResamples ?? DEFAULT_SIGN_FLIP_RESAMPLES;
1188
- assertResampleCount("signFlipResamples", signFlipResamples);
1189
- const exactClusterLimit = options.exactClusterLimit ?? DEFAULT_EXACT_CLUSTER_LIMIT;
1190
- if (!Number.isInteger(exactClusterLimit) || exactClusterLimit < 0 || exactClusterLimit > DEFAULT_EXACT_CLUSTER_LIMIT) throw new ValidationError(`clusteredPairedBinary: exactClusterLimit must be an integer in [0,${DEFAULT_EXACT_CLUSTER_LIMIT}], got ${exactClusterLimit}`);
1191
- const alternative = options.alternative ?? "two-sided";
1192
- if (alternative !== "two-sided" && alternative !== "greater" && alternative !== "less") throw new ValidationError(`clusteredPairedBinary: alternative must be 'two-sided', 'greater', or 'less', got ${String(alternative)}`);
1193
- return {
1194
- seed: options.seed,
1195
- confidence,
1196
- bootstrapResamples,
1197
- alternative,
1198
- exactClusterLimit,
1199
- signFlipResamples
1200
- };
1201
- }
1202
- function assertResampleCount(name, value) {
1203
- if (!Number.isInteger(value) || value <= 0) throw new ValidationError(`clusteredPairedBinary: ${name} must be a positive integer, got ${value}`);
1204
- if (value > MAX_RESAMPLES) throw new ValidationError(`clusteredPairedBinary: ${name} must not exceed ${MAX_RESAMPLES}, got ${value}`);
1205
- }
1206
- function projectSelectedRows(rows, options) {
1207
- const projected = [];
1208
- for (const original of rows) {
1209
- const arm = options.arm(original);
1210
- assertNonEmptyString("arm", arm);
1211
- if (arm !== options.baselineArm && arm !== options.treatmentArm) continue;
1212
- const pairKey = options.pairKey(original);
1213
- const clusterKey = options.clusterKey(original);
1214
- const pass = options.pass(original);
1215
- const repKey = options.repKey?.(original);
1216
- assertNonEmptyString("pairKey", pairKey);
1217
- assertNonEmptyString("clusterKey", clusterKey);
1218
- if (typeof pass !== "boolean") throw new ValidationError(`clusteredPairedBinary: pass accessor must return boolean for pairKey '${pairKey}'`);
1219
- if (repKey !== void 0) assertNonEmptyString("repKey", repKey);
1220
- projected.push({
1221
- pairKey,
1222
- clusterKey,
1223
- arm,
1224
- pass,
1225
- repKey,
1226
- original
1227
- });
1228
- }
1229
- return projected;
1230
- }
1231
- function assertNonEmptyString(name, value) {
1232
- if (typeof value !== "string" || value.trim().length === 0) throw new ValidationError(`clusteredPairedBinary: ${name} accessor must return a non-empty string`);
1233
- }
1234
- function summarizeClusters(pairs) {
1235
- const byCluster = /* @__PURE__ */ new Map();
1236
- for (const pair of pairs) {
1237
- const summary = byCluster.get(pair.clusterKey) ?? {
1238
- nPairs: 0,
1239
- b10: 0,
1240
- b01: 0
1241
- };
1242
- summary.nPairs++;
1243
- if (pair.treatmentPass && !pair.baselinePass) summary.b10++;
1244
- else if (pair.baselinePass && !pair.treatmentPass) summary.b01++;
1245
- byCluster.set(pair.clusterKey, summary);
1246
- }
1247
- return [...byCluster.entries()].sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0).map(([clusterKey, summary]) => ({
1248
- clusterKey,
1249
- ...summary,
1250
- meanDifference: (summary.b10 - summary.b01) / summary.nPairs
1251
- }));
1252
- }
1253
- function clusterBootstrap(clusters, config) {
1254
- const rng = mulberry32(config.seed);
1255
- const samples = new Array(config.bootstrapResamples);
1256
- for (let draw = 0; draw < config.bootstrapResamples; draw++) {
1257
- let differenceSum = 0;
1258
- let pairCount = 0;
1259
- for (let index = 0; index < clusters.length; index++) {
1260
- const cluster = clusters[Math.floor(rng() * clusters.length)];
1261
- differenceSum += cluster.b10 - cluster.b01;
1262
- pairCount += cluster.nPairs;
1263
- }
1264
- samples[draw] = differenceSum / pairCount;
1265
- }
1266
- samples.sort((a, b) => a - b);
1267
- const alpha = 1 - config.confidence;
1268
- const lowerIndex = Math.floor(alpha / 2 * config.bootstrapResamples);
1269
- const upperIndex = Math.min(config.bootstrapResamples - 1, Math.ceil((1 - alpha / 2) * config.bootstrapResamples) - 1);
1270
- return {
1271
- statistic: "task-weighted-risk-difference",
1272
- lower: samples[lowerIndex],
1273
- upper: samples[Math.max(lowerIndex, upperIndex)],
1274
- confidence: config.confidence,
1275
- resamples: config.bootstrapResamples,
1276
- seed: config.seed
1277
- };
1278
- }
1279
- function clusterSignFlip(clusters, config) {
1280
- const clusterTotals = clusters.map((cluster) => cluster.b10 - cluster.b01);
1281
- const nonZero = clusterTotals.filter((delta) => delta !== 0);
1282
- const totalPairs = clusters.reduce((sum, cluster) => sum + cluster.nPairs, 0);
1283
- const statistic = clusterTotals.reduce((sum, delta) => sum + delta, 0) / totalPairs;
1284
- if (nonZero.length <= config.exactClusterLimit) {
1285
- const assignments = 2 ** nonZero.length;
1286
- let extreme = 0;
1287
- for (let mask = 0; mask < assignments; mask++) {
1288
- let sum = 0;
1289
- for (let index = 0; index < nonZero.length; index++) sum += (mask & 2 ** index ? 1 : -1) * nonZero[index];
1290
- if (isExtreme(sum / totalPairs, statistic, config.alternative)) extreme++;
1291
- }
1292
- return {
1293
- statistic,
1294
- pValue: extreme / assignments,
1295
- alternative: config.alternative,
1296
- method: "exact",
1297
- assignments,
1298
- nClusters: clusters.length,
1299
- nNonZeroClusters: nonZero.length,
1300
- seed: null
1301
- };
1302
- }
1303
- const signFlipSeed = (config.seed ^ SIGN_FLIP_SEED_SALT) >>> 0;
1304
- const rng = mulberry32(signFlipSeed);
1305
- let extreme = 0;
1306
- for (let draw = 0; draw < config.signFlipResamples; draw++) {
1307
- let sum = 0;
1308
- for (const delta of nonZero) sum += (rng() < .5 ? -1 : 1) * delta;
1309
- if (isExtreme(sum / totalPairs, statistic, config.alternative)) extreme++;
1310
- }
1311
- return {
1312
- statistic,
1313
- pValue: (extreme + 1) / (config.signFlipResamples + 1),
1314
- alternative: config.alternative,
1315
- method: "monte-carlo",
1316
- assignments: config.signFlipResamples,
1317
- nClusters: clusters.length,
1318
- nNonZeroClusters: nonZero.length,
1319
- seed: signFlipSeed
1320
- };
1321
- }
1322
- function isExtreme(candidate, observed, alternative) {
1323
- const tolerance = Number.EPSILON * Math.max(1, Math.abs(observed)) * 16;
1324
- if (alternative === "greater") return candidate >= observed - tolerance;
1325
- if (alternative === "less") return candidate <= observed + tolerance;
1326
- return Math.abs(candidate) >= Math.abs(observed) - tolerance;
1327
- }
1328
- function mean$4(values) {
1329
- return values.reduce((sum, value) => sum + value, 0) / values.length;
1330
- }
1331
- //#endregion
1332
1112
  //#region src/convergence.ts
1333
1113
  /**
1334
1114
  * ConvergenceTracker — tracks completion percentage over turns.
@@ -1606,6 +1386,10 @@ async function decideNextUserTurn(chat, opts) {
1606
1386
  receiptFromError: costReceiptFromLlmError
1607
1387
  });
1608
1388
  if (!paid.succeeded) throw paid.error;
1389
+ assertServedModel(model, paid.value.servedModel, {
1390
+ allowUnreported: true,
1391
+ context: "decideNextUserTurn"
1392
+ });
1609
1393
  return paid.value.content.trim();
1610
1394
  }
1611
1395
  //#endregion
@@ -2132,14 +1916,28 @@ function canonicalize$1(value) {
2132
1916
  * - membership (free): GET `{baseUrl}/models` once; a model is `listed` when
2133
1917
  * its id is in the served set.
2134
1918
  * - probe (spends a tiny number of tokens): POST `{baseUrl}/chat/completions`
2135
- * per model with a 1-message, 5-token request; `served` is whether the
2136
- * router returns 2xx, with the HTTP `status` and the body's `error.message`
2137
- * captured in `detail`.
1919
+ * per model with a 1-message, `PROBE_MAX_TOKENS`-token request; `served` is
1920
+ * whether the router reached a provider, with the HTTP `status` and the
1921
+ * body's `error.message` captured in `detail`, and `servedModel` recording
1922
+ * WHICH model answered.
1923
+ *
1924
+ * A 2xx is not proof the requested model answered — a gateway can accept one
1925
+ * id and route to another. The probe therefore compares the echoed id against
1926
+ * the requested one and reports `substitution`; `assertModelsServed` fails on
1927
+ * a substituted id exactly as it fails on a dead one, because a campaign that
1928
+ * runs on a silently-swapped model produces per-model numbers about a model it
1929
+ * never called.
2138
1930
  *
2139
1931
  * A default model the router cannot serve is a config bug. Gate a campaign on
2140
- * `assertModelsServed` and it surfaces every dead id with its status + detail
2141
- * instead of silently producing a stub run.
1932
+ * `assertModelsServed` and it surfaces every dead or substituted id with its
1933
+ * status + detail instead of silently producing a stub or mislabelled run.
1934
+ */
1935
+ /**
1936
+ * Provider signature for "the budget ran out before an answer". The model is
1937
+ * alive — a provider took the request and consumed the budget — so this must
1938
+ * never be scored as a dead id.
2142
1939
  */
1940
+ const REASONING_BUDGET_EXHAUSTED = /reasoning[\s_-]?budget[\s_-]?exhausted/i;
2143
1941
  function stripSlash$1(url) {
2144
1942
  return url.replace(/\/+$/, "");
2145
1943
  }
@@ -2158,13 +1956,19 @@ function errorMessage(body) {
2158
1956
  * fallbacks.
2159
1957
  *
2160
1958
  * The membership check (one GET) always runs. When `probe` is true, each model
2161
- * additionally gets a 1-token chat probe so a model that is listed but
1959
+ * additionally gets a small chat probe so a model that is listed but
2162
1960
  * unconfigured (a 401 `model_not_found` from the router) is caught.
2163
1961
  */
2164
1962
  async function preflightModels(opts) {
2165
1963
  const fetchImpl = opts.fetchImpl ?? fetch;
2166
1964
  const baseUrl = stripSlash$1(opts.baseUrl);
2167
1965
  const authHeaders = { authorization: `Bearer ${opts.apiKey}` };
1966
+ const maxTokens = opts.probeMaxTokens ?? 64;
1967
+ if (!Number.isInteger(maxTokens) || maxTokens <= 0) return {
1968
+ succeeded: false,
1969
+ value: null,
1970
+ error: `preflightModels: probeMaxTokens must be a positive integer, got ${maxTokens}`
1971
+ };
2168
1972
  let served;
2169
1973
  try {
2170
1974
  const res = await fetchImpl(`${baseUrl}/models`, {
@@ -2198,7 +2002,9 @@ async function preflightModels(opts) {
2198
2002
  listed,
2199
2003
  served: null,
2200
2004
  status: null,
2201
- detail: null
2005
+ detail: null,
2006
+ budgetExhausted: false,
2007
+ substitution: null
2202
2008
  });
2203
2009
  continue;
2204
2010
  }
@@ -2215,17 +2021,29 @@ async function preflightModels(opts) {
2215
2021
  role: "user",
2216
2022
  content: "ping"
2217
2023
  }],
2218
- max_tokens: 5
2024
+ max_tokens: maxTokens
2219
2025
  })
2220
2026
  });
2221
2027
  let detail = null;
2222
- if (!res.ok) detail = errorMessage(await res.json().catch(() => null));
2028
+ let substitution = null;
2029
+ let budgetExhausted = false;
2030
+ const body = await res.json().catch(() => null);
2031
+ if (res.ok) {
2032
+ const echoed = body?.model;
2033
+ substitution = checkServedModel(model, typeof echoed === "string" && echoed.trim() !== "" ? echoed : null);
2034
+ } else {
2035
+ detail = errorMessage(body);
2036
+ budgetExhausted = detail !== null && REASONING_BUDGET_EXHAUSTED.test(detail);
2037
+ if (budgetExhausted) substitution = checkServedModel(model, null);
2038
+ }
2223
2039
  results.push({
2224
2040
  model,
2225
2041
  listed,
2226
- served: res.ok,
2042
+ served: res.ok || budgetExhausted,
2227
2043
  status: res.status,
2228
- detail
2044
+ detail,
2045
+ budgetExhausted,
2046
+ substitution
2229
2047
  });
2230
2048
  } catch (err) {
2231
2049
  return {
@@ -2254,20 +2072,28 @@ function describeFailure(r) {
2254
2072
  const probeNote = r.served === false ? ` (probe ${r.status}${r.detail ? `: ${r.detail}` : ""})` : "";
2255
2073
  return `${r.model}: not in /models${probeNote}`;
2256
2074
  }
2257
- return `${r.model}: listed but probe ${r.status}${r.detail ? ` — ${r.detail}` : ""}`;
2075
+ if (r.served === false) return `${r.model}: listed but probe ${r.status}${r.detail ? ` — ${r.detail}` : ""}`;
2076
+ if (r.budgetExhausted) return `${r.model}: alive but the probe ran out of reasoning budget (status ${r.status}${r.detail ? `: ${r.detail}` : ""}) — it echoed no model id, so identity is unproven. Raise probeMaxTokens, or pass allowUnreported to accept reachability without identity.`;
2077
+ const s = r.substitution;
2078
+ if (s?.verdict === "unreported") return `${r.model}: probe answered without echoing a model id — identity unproven`;
2079
+ return `${r.model}: probe answered by ${s?.served} (${s?.verdict})`;
2258
2080
  }
2259
2081
  /**
2260
- * Throw `ModelsUnreachableError` naming EVERY model that is unlisted or (when
2261
- * probed) failed its probe — with status + detail per model. A model is dead
2262
- * if it is unlisted, or if `served === false`. Callers gate a campaign on this
2263
- * before spending tokens. When the network call itself fails the underlying
2264
- * outcome error is rethrown there is no partial silent pass.
2082
+ * Throw `ModelsUnreachableError` naming EVERY model that is unusable for a
2083
+ * measured run — with status, detail, and served identity per model. A model
2084
+ * is unusable when it is unlisted, when its probe fails, or when its probe is
2085
+ * answered by a different model than the one requested. Callers gate a
2086
+ * campaign on this before spending tokens. When the network call itself fails
2087
+ * the underlying outcome error is rethrown — there is no partial silent pass.
2088
+ *
2089
+ * Substitution is only detectable with `probe: true`; a membership-only check
2090
+ * proves an id is in the catalogue, never that the catalogue entry answers.
2265
2091
  */
2266
2092
  async function assertModelsServed(opts) {
2267
2093
  const outcome = await preflightModels(opts);
2268
2094
  if (!outcome.succeeded || outcome.value === null) throw new ConfigError(outcome.error ?? "assertModelsServed: preflight failed without an error message");
2269
- const dead = outcome.value.filter((r) => !r.listed || r.served === false);
2270
- if (dead.length > 0) throw new ModelsUnreachableError(`assertModelsServed: ${dead.length}/${outcome.value.length} model(s) unreachable on the router — ${dead.map(describeFailure).join("; ")}`, outcome.value);
2095
+ const bad = outcome.value.filter((r) => !r.listed || r.served === false || r.substitution !== null && !servedModelAcceptable(r.substitution, opts));
2096
+ if (bad.length > 0) throw new ModelsUnreachableError(`assertModelsServed: ${bad.length}/${outcome.value.length} model(s) unusable on the router — ${bad.map(describeFailure).join("; ")}`, outcome.value);
2271
2097
  return outcome.value;
2272
2098
  }
2273
2099
  //#endregion
@@ -2346,95 +2172,6 @@ function assertSingleBackend(agent, judge, opts = {}) {
2346
2172
  return report;
2347
2173
  }
2348
2174
  //#endregion
2349
- //#region src/judge-families.ts
2350
- /** Explicit `provider/...` prefix → family (models.dev / OpenRouter style). */
2351
- const PROVIDER_PREFIX = {
2352
- anthropic: "anthropic",
2353
- openai: "openai",
2354
- "azure-openai": "openai",
2355
- google: "google",
2356
- "google-vertex": "google",
2357
- meta: "meta",
2358
- "meta-llama": "meta",
2359
- mistral: "mistral",
2360
- mistralai: "mistral",
2361
- deepseek: "deepseek",
2362
- xai: "xai",
2363
- qwen: "qwen",
2364
- alibaba: "qwen",
2365
- cohere: "cohere",
2366
- amazon: "amazon",
2367
- bedrock: "amazon",
2368
- moonshot: "moonshot",
2369
- moonshotai: "moonshot",
2370
- kimi: "moonshot",
2371
- "kimi-code": "moonshot",
2372
- zhipu: "zhipu",
2373
- zhipuai: "zhipu",
2374
- zai: "zhipu",
2375
- "z-ai": "zhipu",
2376
- glm: "zhipu"
2377
- };
2378
- /** Fallback model-name patterns when there's no recognised provider prefix. */
2379
- const NAME_PATTERNS = [
2380
- [/claude/i, "anthropic"],
2381
- [/\b(gpt|davinci|babbage)\b|^o[134]\b|[-/]o[134]\b|gpt-/i, "openai"],
2382
- [/gemini|palm|gemma|bison/i, "google"],
2383
- [/llama/i, "meta"],
2384
- [/mi(s|x)tral|codestral|magistral/i, "mistral"],
2385
- [/deepseek/i, "deepseek"],
2386
- [/grok/i, "xai"],
2387
- [/qwen/i, "qwen"],
2388
- [/command-?(r|a)?/i, "cohere"],
2389
- [/\b(nova|titan)\b/i, "amazon"],
2390
- [/\bkimi\b|moonshot/i, "moonshot"],
2391
- [/\bglm\b|zhipu|\bz-?ai\b/i, "zhipu"]
2392
- ];
2393
- /**
2394
- * Classify a model id into its provider family. Strips a `@snapshot` suffix
2395
- * and prefers an explicit `provider/...` prefix; otherwise matches the model
2396
- * name. Returns `unknown` when nothing matches (callers decide whether that's
2397
- * acceptable — `assertCrossFamily` counts it as its own family).
2398
- */
2399
- function judgeFamily(modelId) {
2400
- const id = modelId.trim().split("@")[0].toLowerCase();
2401
- const slash = id.indexOf("/");
2402
- if (slash > 0) {
2403
- const prefix = id.slice(0, slash);
2404
- const mapped = PROVIDER_PREFIX[prefix];
2405
- if (mapped) return mapped;
2406
- }
2407
- for (const [pattern, family] of NAME_PATTERNS) if (pattern.test(id)) return family;
2408
- return "unknown";
2409
- }
2410
- var CrossFamilyError = class extends Error {
2411
- families;
2412
- models;
2413
- constructor(message, families, models) {
2414
- super(message);
2415
- this.families = families;
2416
- this.models = models;
2417
- this.name = "CrossFamilyError";
2418
- }
2419
- };
2420
- /**
2421
- * Throw unless the judge models span at least `minFamilies` distinct provider
2422
- * families. Pass the model ids backing your judge ensemble. Fail-loud by
2423
- * design — a correlated single-family ensemble silently inflates agreement.
2424
- */
2425
- function assertCrossFamily(models, opts = {}) {
2426
- const minFamilies = opts.minFamilies ?? 2;
2427
- const families = /* @__PURE__ */ new Set();
2428
- for (const m of models) {
2429
- const f = judgeFamily(m);
2430
- if (f === "unknown" && !opts.allowUnknown) continue;
2431
- families.add(f);
2432
- }
2433
- const list = [...families].sort();
2434
- if (list.length < minFamilies) throw new CrossFamilyError(`judge ensemble spans ${list.length} provider famil${list.length === 1 ? "y" : "ies"} (${list.join(", ") || "none"}) but ${minFamilies} required — a single-family ensemble is correlated bias, not independent signal`, list, models);
2435
- return list;
2436
- }
2437
- //#endregion
2438
2175
  //#region src/knowledge/readiness.ts
2439
2176
  function scoreKnowledgeReadiness(options) {
2440
2177
  const now = options.now ?? /* @__PURE__ */ new Date();
@@ -5115,269 +4852,6 @@ var EvalTraceStore = class {
5115
4852
  }
5116
4853
  };
5117
4854
  //#endregion
5118
- //#region src/experiment-tracker.ts
5119
- /**
5120
- * Experiment tracker — git-provenanced experiment log with N-rep stats and a
5121
- * KEEP / REGRESSION / NOISE verdict against a parent.
5122
- *
5123
- * Every loop the fleet runs reduces to the same question: "I ran the candidate
5124
- * N times — is the median measurably better than the parent, or is the delta
5125
- * inside the noise band?" The hand-rolled copies bake a fixed score scale
5126
- * (percentage points), a fixed store path (`.evolve/experiments-v2.json`), and
5127
- * `execSync('git …')` straight into the module. This is the canonical version:
5128
- * provenance and persistence are injected, thresholds are configurable, and the
5129
- * stats + verdict are pure functions you can unit-test without a git repo or a
5130
- * filesystem.
5131
- *
5132
- * Stats per experiment: median / mean / min / max / iqr / stddev / passRate /
5133
- * n, plus a `stable` flag (`iqr < iqrUnstableAbove && stddev < stddevUnstableAbove`).
5134
- *
5135
- * Verdict against a parent (both must have `n >= minRepsForVerdict`):
5136
- * - NOISE — the candidate is too unstable to judge (`!stable`)
5137
- * - KEEP — `medianDelta > keepThreshold`
5138
- * - REGRESSION — `medianDelta < -regressionThreshold`
5139
- * - NOISE — otherwise (delta inside the band)
5140
- * With no parent (or insufficient reps) the verdict is the neutral ITERATE.
5141
- */
5142
- const DEFAULTS = {
5143
- keepThreshold: 5,
5144
- regressionThreshold: 5,
5145
- iqrUnstableAbove: 10,
5146
- stddevUnstableAbove: Number.POSITIVE_INFINITY,
5147
- minRepsForVerdict: 3
5148
- };
5149
- function resolveThresholds(t) {
5150
- const r = {
5151
- ...DEFAULTS,
5152
- ...t ?? {}
5153
- };
5154
- if (r.keepThreshold < 0) throw new ValidationError(`experiment-tracker: keepThreshold must be >= 0, got ${r.keepThreshold}`);
5155
- if (r.regressionThreshold < 0) throw new ValidationError(`experiment-tracker: regressionThreshold must be >= 0, got ${r.regressionThreshold}`);
5156
- if (r.minRepsForVerdict < 1) throw new ValidationError(`experiment-tracker: minRepsForVerdict must be >= 1, got ${r.minRepsForVerdict}`);
5157
- return r;
5158
- }
5159
- function median$1(sorted) {
5160
- const n = sorted.length;
5161
- if (n === 0) return 0;
5162
- const mid = Math.floor(n / 2);
5163
- return n % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
5164
- }
5165
- /** Population standard deviation (÷n). 0 for fewer than 2 values. */
5166
- function stddev(values, mean) {
5167
- if (values.length < 2) return 0;
5168
- const variance = values.reduce((acc, v) => acc + (v - mean) ** 2, 0) / values.length;
5169
- return Math.sqrt(variance);
5170
- }
5171
- /**
5172
- * Compute the N-rep statistics for a set of reps. Pure — no I/O. The `stable`
5173
- * flag is the trust gate the verdict depends on: a sample whose spread exceeds
5174
- * the configured bounds can't distinguish a real delta from run-to-run noise.
5175
- */
5176
- function computeExperimentStats(reps, thresholds) {
5177
- const t = resolveThresholds(thresholds);
5178
- const n = reps.length;
5179
- if (n === 0) return {
5180
- median: 0,
5181
- mean: 0,
5182
- min: 0,
5183
- max: 0,
5184
- iqr: 0,
5185
- stddev: 0,
5186
- passRate: null,
5187
- n: 0,
5188
- stable: false
5189
- };
5190
- const scores = reps.map((r) => {
5191
- if (!Number.isFinite(r.score)) throw new ValidationError(`experiment-tracker: rep ${r.rep} has non-finite score ${r.score}`);
5192
- return r.score;
5193
- });
5194
- const sorted = [...scores].sort((a, b) => a - b);
5195
- const mean = scores.reduce((s, v) => s + v, 0) / n;
5196
- const sd = stddev(scores, mean);
5197
- const spread = iqr(scores);
5198
- const rated = reps.filter((r) => typeof r.passed === "boolean");
5199
- const passRate = rated.length === 0 ? null : rated.filter((r) => r.passed).length / rated.length;
5200
- const stable = spread < t.iqrUnstableAbove && sd < t.stddevUnstableAbove;
5201
- return {
5202
- median: median$1(sorted),
5203
- mean,
5204
- min: sorted[0],
5205
- max: sorted[n - 1],
5206
- iqr: spread,
5207
- stddev: sd,
5208
- passRate,
5209
- n,
5210
- stable
5211
- };
5212
- }
5213
- /**
5214
- * Verdict for a candidate against its parent. Pure — operates on already-computed
5215
- * stats. KEEP/REGRESSION require both sides to have `>= minRepsForVerdict` reps
5216
- * AND the candidate to be `stable`; otherwise the result is NOISE (unstable) or
5217
- * ITERATE (not enough reps / no parent).
5218
- */
5219
- function improvementVerdict(candidate, parent, thresholds) {
5220
- const t = resolveThresholds(thresholds);
5221
- if (!parent) return {
5222
- verdict: "ITERATE",
5223
- medianDelta: null,
5224
- reason: "no parent experiment to compare against"
5225
- };
5226
- if (candidate.n < t.minRepsForVerdict || parent.n < t.minRepsForVerdict) return {
5227
- verdict: "ITERATE",
5228
- medianDelta: null,
5229
- reason: `need >= ${t.minRepsForVerdict} reps on both sides (candidate n=${candidate.n}, parent n=${parent.n})`
5230
- };
5231
- if (!candidate.stable) return {
5232
- verdict: "NOISE",
5233
- medianDelta: candidate.median - parent.median,
5234
- reason: `candidate unstable (iqr=${candidate.iqr}, stddev=${candidate.stddev.toFixed(2)})`
5235
- };
5236
- const medianDelta = candidate.median - parent.median;
5237
- if (medianDelta > t.keepThreshold) return {
5238
- verdict: "KEEP",
5239
- medianDelta,
5240
- reason: `median +${medianDelta} > +${t.keepThreshold}`
5241
- };
5242
- if (medianDelta < -t.regressionThreshold) return {
5243
- verdict: "REGRESSION",
5244
- medianDelta,
5245
- reason: `median ${medianDelta} < -${t.regressionThreshold}`
5246
- };
5247
- return {
5248
- verdict: "NOISE",
5249
- medianDelta,
5250
- reason: `median delta ${medianDelta} inside noise band [-${t.regressionThreshold}, +${t.keepThreshold}]`
5251
- };
5252
- }
5253
- /**
5254
- * Default provenance reader: `git rev-parse HEAD`, the subject line, and the
5255
- * files changed vs `HEAD~1`. Fail-loud — a tracker that silently logs
5256
- * `commit: 'unknown'` corrupts the provenance the whole point of the log is to
5257
- * carry. When the working tree genuinely has no parent commit, pass an override.
5258
- */
5259
- const gitProvenanceReader = () => {
5260
- const run = (cmd) => execSync(cmd, { encoding: "utf8" }).trim();
5261
- const commit = run("git rev-parse --short HEAD");
5262
- const message = run("git log -1 --format=%s");
5263
- const changedRaw = run("git diff --name-only HEAD~1");
5264
- return {
5265
- commit,
5266
- message,
5267
- changedFiles: changedRaw.length === 0 ? [] : changedRaw.split("\n").filter(Boolean)
5268
- };
5269
- };
5270
- /** In-memory store — the default when no persistence is wanted (tests, ephemeral
5271
- * runs). State lives on the instance. */
5272
- function inMemoryExperimentStore(initial = []) {
5273
- let state = initial.map((e) => structuredClone(e));
5274
- return {
5275
- async load() {
5276
- return state.map((e) => structuredClone(e));
5277
- },
5278
- async save(experiments) {
5279
- state = experiments.map((e) => structuredClone(e));
5280
- }
5281
- };
5282
- }
5283
- /** Filesystem store — a single JSON array at `path`, created on first save. */
5284
- function fileExperimentStore(path) {
5285
- return {
5286
- async load() {
5287
- const fs = await import("node:fs/promises");
5288
- try {
5289
- const raw = await fs.readFile(path, "utf8");
5290
- const parsed = JSON.parse(raw);
5291
- if (!Array.isArray(parsed)) throw new ValidationError(`experiment-tracker: store at ${path} is not a JSON array`);
5292
- return parsed;
5293
- } catch (err) {
5294
- if (err.code === "ENOENT") return [];
5295
- throw err;
5296
- }
5297
- },
5298
- async save(experiments) {
5299
- const fs = await import("node:fs/promises");
5300
- const pathMod = await import("node:path");
5301
- await fs.mkdir(pathMod.dirname(path), { recursive: true });
5302
- await fs.writeFile(path, JSON.stringify(experiments, null, 2), "utf8");
5303
- }
5304
- };
5305
- }
5306
- /**
5307
- * Stateful tracker over an `ExperimentStore`. Create an experiment (provenance
5308
- * is captured once), append reps as they complete (stats + verdict recompute on
5309
- * every append), and read the log back for a dashboard. All persistence and git
5310
- * access flow through the injected seams, so the tracker is fully testable
5311
- * without a repo or disk.
5312
- */
5313
- var ExperimentTracker = class {
5314
- store;
5315
- provenanceReader;
5316
- thresholds;
5317
- now;
5318
- constructor(options = {}) {
5319
- this.store = options.store ?? inMemoryExperimentStore();
5320
- this.provenanceReader = options.provenanceReader ?? gitProvenanceReader;
5321
- this.thresholds = resolveThresholds(options.thresholds);
5322
- this.now = options.now ?? Date.now;
5323
- }
5324
- async create(input) {
5325
- const experiments = await this.store.load();
5326
- if (experiments.some((e) => e.id === input.id)) throw new ValidationError(`experiment-tracker: experiment id "${input.id}" already exists`);
5327
- if (input.parentId && !experiments.some((e) => e.id === input.parentId)) throw new ValidationError(`experiment-tracker: parent experiment "${input.parentId}" not found`);
5328
- const provenance = input.provenance ?? await this.provenanceReader();
5329
- const experiment = {
5330
- id: input.id,
5331
- label: input.label,
5332
- provenance,
5333
- parentId: input.parentId,
5334
- changeSummary: input.changeSummary,
5335
- reps: [],
5336
- stats: computeExperimentStats([], this.thresholds),
5337
- verdict: "ITERATE",
5338
- createdAt: new Date(this.now()).toISOString()
5339
- };
5340
- experiments.push(experiment);
5341
- await this.store.save(experiments);
5342
- return structuredClone(experiment);
5343
- }
5344
- /** Append a rep (its `rep` index defaults to the current rep count) and
5345
- * recompute stats + verdict. Returns the updated experiment. */
5346
- async addRep(experimentId, rep) {
5347
- const experiments = await this.store.load();
5348
- const exp = experiments.find((e) => e.id === experimentId);
5349
- if (!exp) throw new ValidationError(`experiment-tracker: experiment "${experimentId}" not found`);
5350
- const fullRep = {
5351
- rep: rep.rep ?? exp.reps.length,
5352
- score: rep.score,
5353
- passed: rep.passed,
5354
- metrics: rep.metrics,
5355
- timestamp: rep.timestamp ?? new Date(this.now()).toISOString()
5356
- };
5357
- exp.reps.push(fullRep);
5358
- exp.stats = computeExperimentStats(exp.reps, this.thresholds);
5359
- const parent = exp.parentId ? experiments.find((e) => e.id === exp.parentId) : void 0;
5360
- exp.verdict = improvementVerdict(exp.stats, parent?.stats ?? null, this.thresholds).verdict;
5361
- await this.store.save(experiments);
5362
- return structuredClone(exp);
5363
- }
5364
- async get(experimentId) {
5365
- const found = (await this.store.load()).find((e) => e.id === experimentId);
5366
- return found ? structuredClone(found) : void 0;
5367
- }
5368
- async list() {
5369
- return this.store.load();
5370
- }
5371
- /** Full verdict (not just the enum) for an experiment vs its parent. */
5372
- async verdictFor(experimentId) {
5373
- const experiments = await this.store.load();
5374
- const exp = experiments.find((e) => e.id === experimentId);
5375
- if (!exp) throw new ValidationError(`experiment-tracker: experiment "${experimentId}" not found`);
5376
- const parent = exp.parentId ? experiments.find((e) => e.id === exp.parentId) : void 0;
5377
- return improvementVerdict(exp.stats, parent?.stats ?? null, this.thresholds);
5378
- }
5379
- };
5380
- //#endregion
5381
4855
  //#region src/leaderboard.ts
5382
4856
  function mean$1(xs) {
5383
4857
  return xs.length ? xs.reduce((a, b) => a + b, 0) / xs.length : null;
@@ -6296,6 +5770,162 @@ function statusAdvanced(key, progression) {
6296
5770
  };
6297
5771
  }
6298
5772
  //#endregion
5773
+ //#region src/verification-strategy.ts
5774
+ /**
5775
+ * The family registry. `Record` over the union keeps it exhaustive: adding
5776
+ * a member to `VerificationStrategySource` without a profile here fails to
5777
+ * compile. The failure mode travels with the taxonomy so a reader of a
5778
+ * certification can surface it without this package's docs at hand.
5779
+ */
5780
+ const VERIFICATION_STRATEGIES = {
5781
+ compile: {
5782
+ determinism: "deterministic",
5783
+ failureMode: "code that compiles is not code that is correct"
5784
+ },
5785
+ test: {
5786
+ determinism: "deterministic",
5787
+ failureMode: "assumes an answer key; certifies nothing outside suite coverage, and a stubbed integration reports green"
5788
+ },
5789
+ schema: {
5790
+ determinism: "deterministic",
5791
+ failureMode: "shape is not meaning; a well-formed wrong answer passes"
5792
+ },
5793
+ sandbox: {
5794
+ determinism: "deterministic",
5795
+ failureMode: "an exit code compresses the run to one bit; a faked success exits 0"
5796
+ },
5797
+ judge: {
5798
+ determinism: "probabilistic",
5799
+ failureMode: "drifts across model versions and is Goodhart-gameable by the graded policy"
5800
+ },
5801
+ composite: {
5802
+ determinism: "inherited",
5803
+ failureMode: "scalar collapse: the blend hides which member carried the score"
5804
+ },
5805
+ "proof-kernel": {
5806
+ determinism: "deterministic",
5807
+ failureMode: "the formalization gap: the kernel certifies the formal statement, never that it matches the informal claim"
5808
+ },
5809
+ invariant: {
5810
+ determinism: "deterministic",
5811
+ failureMode: "weak invariants pass everything; a set uncalibrated by seeded bugs is a rubber stamp"
5812
+ },
5813
+ replication: {
5814
+ determinism: "deterministic",
5815
+ failureMode: "re-runs the method, so it catches drift and nondeterminism, never an error the method itself carries"
5816
+ },
5817
+ agreement: {
5818
+ determinism: "probabilistic",
5819
+ failureMode: "the shared blind spot: derivers with common corpora or priors agree for the same wrong reason"
5820
+ }
5821
+ };
5822
+ /** Every family member, derived from the registry so it cannot drift. */
5823
+ const VERIFICATION_STRATEGY_SOURCES = Object.keys(VERIFICATION_STRATEGIES);
5824
+ //#endregion
5825
+ //#region src/equivalence-check.ts
5826
+ /** A refused equivalence check. `code` names the exact refusal for programmatic handling. */
5827
+ var EquivalenceProtocolError = class extends Error {
5828
+ code;
5829
+ constructor(code, message) {
5830
+ super(message);
5831
+ this.name = "EquivalenceProtocolError";
5832
+ this.code = code;
5833
+ }
5834
+ };
5835
+ /**
5836
+ * Validate and freeze a two-arm blind equivalence check spec.
5837
+ *
5838
+ * The literal types already refuse a wide design at compile time; the
5839
+ * runtime checks hold the same line for untyped callers. There is no
5840
+ * escape hatch: `arms: 3` or `blind: false` throws, never downgrades.
5841
+ */
5842
+ function defineEquivalenceCheck(spec) {
5843
+ if (spec.arms !== 2) throw new EquivalenceProtocolError("arm-count", `equivalence check requires exactly 2 arms, received ${String(spec.arms)}`);
5844
+ if (spec.blind !== true) throw new EquivalenceProtocolError("not-blind", "equivalence check requires blind: true — a non-blind run is not a weaker check, it is no check");
5845
+ if (typeof spec.artifact !== "string" || spec.artifact.trim() === "") throw new EquivalenceProtocolError("empty-field", "spec.artifact must identify the claim under verification");
5846
+ if (!VERIFICATION_STRATEGY_SOURCES.includes(spec.source)) throw new EquivalenceProtocolError("unknown-source", `spec.source '${String(spec.source)}' is not a verification-strategy member`);
5847
+ return Object.freeze({ spec: Object.freeze({ ...spec }) });
5848
+ }
5849
+ /**
5850
+ * Assemble the equivalence record, refusing every asymmetry.
5851
+ *
5852
+ * Refusals (each throws `EquivalenceProtocolError`):
5853
+ * - an arm whose `blindness.toOtherArms` is false — it saw the other
5854
+ * statement, so nothing was independently derived;
5855
+ * - an arm whose `blindness.toOutcome` is false — it could steer its
5856
+ * statement toward (or away from) the known result;
5857
+ * - duplicate arm ids, empty statements, empty derivations;
5858
+ * - an obligation whose fields contradict its status (see
5859
+ * `EquivalenceObligation`).
5860
+ */
5861
+ function buildEquivalenceRecord(definition, arms, obligation) {
5862
+ assertArms(arms);
5863
+ assertObligation(obligation);
5864
+ return Object.freeze({
5865
+ spec: definition.spec,
5866
+ arms: Object.freeze([Object.freeze({ ...arms[0] }), Object.freeze({ ...arms[1] })]),
5867
+ obligation: Object.freeze({ ...obligation })
5868
+ });
5869
+ }
5870
+ /**
5871
+ * Discharge the obligation through an injected checker and return the
5872
+ * record.
5873
+ *
5874
+ * Order matters: every arm refusal fires BEFORE the checker runs — an
5875
+ * invalid check must not spend. A checker whose `strategy` differs from
5876
+ * `spec.source` is refused for the same reason: a judge cannot silently
5877
+ * discharge a proof-kernel obligation.
5878
+ *
5879
+ * A checker failure (`succeeded: false`) is not thrown: it becomes an
5880
+ * `'unresolved'` obligation carrying the full error text, which is the
5881
+ * honest record of an undischarged check.
5882
+ */
5883
+ async function runEquivalenceCheck(definition, arms, checker) {
5884
+ assertArms(arms);
5885
+ if (checker.strategy !== definition.spec.source) throw new EquivalenceProtocolError("checker-strategy-mismatch", `spec.source is '${definition.spec.source}' but the bound checker declares '${checker.strategy}'`);
5886
+ const outcome = await checker.check({
5887
+ artifact: definition.spec.artifact,
5888
+ statements: [arms[0].statement, arms[1].statement]
5889
+ });
5890
+ if (!outcome.succeeded) return buildEquivalenceRecord(definition, arms, {
5891
+ status: "unresolved",
5892
+ unresolvedReason: outcome.error,
5893
+ checker: checker.identity
5894
+ });
5895
+ const { status, separatingWitness, evidenceDigest } = outcome.value;
5896
+ return buildEquivalenceRecord(definition, arms, {
5897
+ status,
5898
+ ...separatingWitness === void 0 ? {} : { separatingWitness },
5899
+ checker: checker.identity,
5900
+ evidenceDigest
5901
+ });
5902
+ }
5903
+ function assertArms(arms) {
5904
+ if (arms.length !== 2) throw new EquivalenceProtocolError("arm-count", `equivalence record requires exactly 2 arms, received ${arms.length}`);
5905
+ if (arms[0].armId === arms[1].armId) throw new EquivalenceProtocolError("duplicate-arm-id", `both arms declare armId '${arms[0].armId}' — two labels for one derivation is one arm`);
5906
+ for (const arm of arms) {
5907
+ if (arm.statement.trim() === "") throw new EquivalenceProtocolError("empty-field", `arm '${arm.armId}' committed an empty statement`);
5908
+ if (arm.derivedFrom.trim() === "") throw new EquivalenceProtocolError("empty-field", `arm '${arm.armId}' declares no derivation provenance`);
5909
+ if (arm.blindness.toOtherArms !== true) throw new EquivalenceProtocolError("arm-saw-other", `arm '${arm.armId}' saw another arm's statement — the derivation is not independent and the check is invalid`);
5910
+ if (arm.blindness.toOutcome !== true) throw new EquivalenceProtocolError("arm-saw-outcome", `arm '${arm.armId}' saw the outcome before committing — the check is invalid`);
5911
+ }
5912
+ }
5913
+ function assertObligation(obligation) {
5914
+ const { status, separatingWitness, unresolvedReason, evidenceDigest } = obligation;
5915
+ if (status === "refuted-with-separating-witness") {
5916
+ if (typeof separatingWitness !== "string" || separatingWitness.trim() === "") throw new EquivalenceProtocolError("witness-missing", "a refuted equivalence must carry the separating witness — a refutation without one is an assertion");
5917
+ if (typeof evidenceDigest !== "string" || evidenceDigest.trim() === "") throw new EquivalenceProtocolError("evidence-missing", "a refuted equivalence must carry its evidence digest");
5918
+ return;
5919
+ }
5920
+ if (status === "proved") {
5921
+ if (separatingWitness !== void 0) throw new EquivalenceProtocolError("witness-on-proved", "a proved equivalence cannot carry a separating witness — the two claims contradict");
5922
+ if (typeof evidenceDigest !== "string" || evidenceDigest.trim() === "") throw new EquivalenceProtocolError("evidence-missing", "a proved equivalence must carry its evidence digest");
5923
+ return;
5924
+ }
5925
+ if (separatingWitness !== void 0) throw new EquivalenceProtocolError("witness-on-unresolved", "an unresolved obligation cannot carry a separating witness — a witness in hand is a refutation");
5926
+ if (typeof unresolvedReason !== "string" || unresolvedReason.trim() === "") throw new EquivalenceProtocolError("reason-missing", "an unresolved obligation must say why — 'unresolved' with no reason erases the diagnostic");
5927
+ }
5928
+ //#endregion
6299
5929
  //#region src/ui-finding.ts
6300
5930
  /** Frozen tuple of lenses for validation + iteration. */
6301
5931
  const UI_LENSES = [
@@ -7496,126 +7126,6 @@ async function promptBisect(options) {
7496
7126
  };
7497
7127
  }
7498
7128
  //#endregion
7499
- //#region src/counterfactual.ts
7500
- /**
7501
- * Counterfactual replay — "what would have happened if we'd changed
7502
- * exactly one thing at turn N?"
7503
- *
7504
- * The framework does NOT drive the agent — it sets up the replay
7505
- * context (prior spans, prior state, mutation spec) and records the
7506
- * resulting divergence. Consumers supply an `executeFrom(ctx)` callback
7507
- * that runs their agent starting from turn N with the mutation applied.
7508
- *
7509
- * Counterfactual runs are recorded as a new Run with `layer='meta'` and
7510
- * `parentRunId = originalRunId`, so downstream diff + correlation
7511
- * pipelines see them natively.
7512
- */
7513
- async function runCounterfactual(store, originalRunId, mutation, runner) {
7514
- const originalRun = await store.getRun(originalRunId);
7515
- if (!originalRun) throw new NotFoundError(`counterfactual: run ${originalRunId} not found`);
7516
- const trajectory = await buildTrajectory(store, originalRunId);
7517
- if (mutation.at < 0 || mutation.at >= trajectory.steps.length) throw new ValidationError(`counterfactual: mutation.at=${mutation.at} out of range [0, ${trajectory.steps.length})`);
7518
- const targetStep = trajectory.steps[mutation.at];
7519
- const mutatedStep = applyMutation(targetStep, mutation);
7520
- const cfEmitter = new TraceEmitter(store);
7521
- await cfEmitter.startRun({
7522
- scenarioId: originalRun.scenarioId,
7523
- variantId: originalRun.variantId ? `${originalRun.variantId}+cf:${mutation.kind}@${mutation.at}` : `cf:${mutation.kind}@${mutation.at}`,
7524
- projectId: originalRun.projectId,
7525
- parentRunId: originalRunId,
7526
- layer: "meta",
7527
- tags: {
7528
- counterfactual: "true",
7529
- mutationKind: mutation.kind,
7530
- mutationAt: String(mutation.at)
7531
- }
7532
- });
7533
- await runner.executeFrom({
7534
- originalRunId,
7535
- originalTrajectory: trajectory,
7536
- prefix: trajectory.steps.slice(0, mutation.at),
7537
- mutation,
7538
- mutatedStep
7539
- }, cfEmitter);
7540
- const counterfactual = await store.getRun(cfEmitter.runId);
7541
- const delta = {
7542
- originalOutcomeScore: originalRun.outcome?.score ?? null,
7543
- counterfactualOutcomeScore: counterfactual?.outcome?.score ?? null,
7544
- deltaScore: originalRun.outcome?.score !== void 0 && counterfactual?.outcome?.score !== void 0 ? counterfactual.outcome.score - originalRun.outcome.score : null
7545
- };
7546
- return {
7547
- counterfactualRunId: cfEmitter.runId,
7548
- originalRunId,
7549
- mutation,
7550
- delta
7551
- };
7552
- }
7553
- function applyMutation(step, mutation) {
7554
- if (mutation.kind === "swap-model" && step.span.kind === "llm") {
7555
- const llm = step.span;
7556
- return {
7557
- ...step,
7558
- span: {
7559
- ...llm,
7560
- model: mutation.newModel
7561
- }
7562
- };
7563
- }
7564
- if (mutation.kind === "swap-tool-result" && step.span.kind === "tool") {
7565
- const tool = step.span;
7566
- return {
7567
- ...step,
7568
- span: {
7569
- ...tool,
7570
- result: mutation.newResult
7571
- }
7572
- };
7573
- }
7574
- if (mutation.kind === "inject-system-message" && step.span.kind === "llm") {
7575
- const llm = step.span;
7576
- return {
7577
- ...step,
7578
- span: {
7579
- ...llm,
7580
- messages: [{
7581
- role: "system",
7582
- content: mutation.content
7583
- }, ...llm.messages]
7584
- }
7585
- };
7586
- }
7587
- if (mutation.kind === "custom") return mutation.apply(step);
7588
- return step;
7589
- }
7590
- /**
7591
- * Aggregate a batch of counterfactuals into a simple attribution table:
7592
- * which mutation kinds move outcomes most? (Useful when you run a grid
7593
- * over the same trajectory — swap-model at every llm span, swap-tool
7594
- * at every tool span — and want a ranked summary.)
7595
- */
7596
- function attributeCounterfactuals(results) {
7597
- const grouped = /* @__PURE__ */ new Map();
7598
- for (const r of results) {
7599
- const arr = grouped.get(r.mutation.kind) ?? [];
7600
- arr.push(r);
7601
- grouped.set(r.mutation.kind, arr);
7602
- }
7603
- const out = [];
7604
- for (const [kind, items] of grouped) {
7605
- const deltas = items.map((i) => i.delta.deltaScore).filter((d) => typeof d === "number");
7606
- if (deltas.length === 0) continue;
7607
- const meanAbs = deltas.reduce((a, b) => a + Math.abs(b), 0) / deltas.length;
7608
- const meanSigned = deltas.reduce((a, b) => a + b, 0) / deltas.length;
7609
- out.push({
7610
- mutationKind: kind,
7611
- n: deltas.length,
7612
- meanAbsDelta: meanAbs,
7613
- meanSignedDelta: meanSigned
7614
- });
7615
- }
7616
- return out.sort((a, b) => b.meanAbsDelta - a.meanAbsDelta);
7617
- }
7618
- //#endregion
7619
7129
  //#region src/cross-trace-diff.ts
7620
7130
  async function crossTraceDiff(store, runA, runB, options = {}) {
7621
7131
  const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)]);
@@ -11470,10 +10980,20 @@ function attachCostToReport(report, ledger) {
11470
10980
  /**
11471
10981
  * Tier presets — plain data, swap or spread freely.
11472
10982
  *
11473
- * `economy` uses the fleet-policy ids: every id resolves through the
11474
- * substrate's family pricing (no costUnknown axis) and the judge trio spans
11475
- * three provider families (moonshot / deepseek / openai), so it passes
11476
- * `assertCrossFamily` as-is.
10983
+ * A preset names REQUESTED ids, and a requested id is NOT a guarantee of
10984
+ * family, of provider, or of liveness. A routing gateway can accept any id
10985
+ * below and answer from a different model on HTTP 200, with only the response
10986
+ * body's `model` field betraying the swap. The served assertion is the
10987
+ * guarantee: gate a run on `assertModelsServed({ probe: true })` and assert
10988
+ * `assertServedModel` per call. `assertCrossFamily` over these ids proves the
10989
+ * configuration is diverse; only `assertCrossFamilyServed` over the ids that
10990
+ * ANSWERED proves the run was.
10991
+ *
10992
+ * `economy` names ids a live router probe answered from the provider their
10993
+ * name implies, so the judge trio spanned three provider families (deepseek /
10994
+ * zhipu / google) as configured and, at that probe, as served. A preset is
10995
+ * only as good as its last probe: ids go dead and start resolving elsewhere
10996
+ * without notice, so re-probe rather than trusting this list.
11477
10997
  *
11478
10998
  * `frontier` is deliberately EMPTY: entitled frontier ids vary per router
11479
10999
  * account, and a hardcoded claude/gpt-5 id 401s on keys that lack it. Supply
@@ -11482,14 +11002,14 @@ function attachCostToReport(report, ledger) {
11482
11002
  */
11483
11003
  const seatPresets = {
11484
11004
  economy: {
11485
- worker: "kimi-k2.6",
11005
+ worker: "zai/glm-5.2",
11486
11006
  judges: [
11487
- "kimi-k2.6",
11488
11007
  "deepseek-v4-pro",
11489
- "gpt-4.1-mini"
11008
+ "zai/glm-5.2",
11009
+ "google/gemini-2.5-flash"
11490
11010
  ],
11491
- analyst: "gpt-4.1-mini",
11492
- reflection: "gpt-4.1-mini",
11011
+ analyst: "zai/glm-5.2",
11012
+ reflection: "zai/glm-5.2",
11493
11013
  verifier: "deepseek-v4-pro"
11494
11014
  },
11495
11015
  frontier: {}
@@ -12464,6 +11984,6 @@ function assertProductBenchmarkRun(runDir) {
12464
11984
  return report;
12465
11985
  }
12466
11986
  //#endregion
12467
- export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CONTROL_INTEGRITY_ANALYST, CallbackResearcher, CaptureIntegrityError, ConfigError, ControlIntegrityAnalyst, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExactAnalystRunExecutionError, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_BRIEFS, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LimitExceededError, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_INTEGRITY_SCHEMA, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYSIS_LIMITS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TOOL_NAMESPACE, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, ToolTraceMissingError, TraceAnalysisLimitError, TraceAnalysisStoreContractError, TraceAnalysisValidationError, TraceContractBuilder, TraceEmitter, TraceFileMalformedError, TraceFileMissingError, TraceFileTooLargeError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, WORKER_DRIVER_DOCTRINE, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunIntegrity, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertExactRegistryRunOpts, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalysisToolDescriptors, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDefaultReviewer, createDspyRlmTraceEngine, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalyst, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, defineTraceAnalyst, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, emitControlIntegrityFindings, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isRuntimeSupervisorRunDir, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpTextToTraceAnalysisStore, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, readRuntimeSupervisorRun, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resetDeprecationWarnings, resolveModelPricing, resolveSeat, resolveTraceAnalystLimits, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runTraceAnalyst, runsForScenario, runtimeSupervisorRunReader, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, warnDeprecatedOnce, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
11987
+ export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CONTROL_INTEGRITY_ANALYST, CallbackResearcher, CaptureIntegrityError, ConfigError, ControlIntegrityAnalyst, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, EvalTraceStore, ExactAnalystRunExecutionError, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_BRIEFS, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LimitExceededError, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelSubstitutionError, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PROBE_MAX_TOKENS, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_INTEGRITY_SCHEMA, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, ServedCrossFamilyError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYSIS_LIMITS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TOOL_NAMESPACE, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, ToolTraceMissingError, TraceAnalysisLimitError, TraceAnalysisStoreContractError, TraceAnalysisValidationError, TraceContractBuilder, TraceEmitter, TraceFileMalformedError, TraceFileMissingError, TraceFileTooLargeError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, WORKER_DRIVER_DOCTRINE, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunIntegrity, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertExactRegistryRunOpts, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildEquivalenceRecord, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalysisToolDescriptors, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkServedModel, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDefaultReviewer, createDspyRlmTraceEngine, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalyst, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, defineEquivalenceCheck, defineTraceAnalyst, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, emitControlIntegrityFindings, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isRuntimeSupervisorRunDir, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeModelId, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpTextToTraceAnalysisStore, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, readRuntimeSupervisorRun, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resetDeprecationWarnings, resolveModelPricing, resolveSeat, resolveTraceAnalystLimits, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEquivalenceCheck, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runTraceAnalyst, runsForScenario, runtimeSupervisorRunReader, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, servedModelAcceptable, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, warnDeprecatedOnce, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
12468
11988
 
12469
11989
  //# sourceMappingURL=index.js.map