@tangle-network/agent-eval 0.144.6 → 0.144.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (239) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/analyst/index.d.ts +470 -88
  6. package/dist/analyst/index.d.ts.map +1 -1
  7. package/dist/analyst/index.js +24 -5
  8. package/dist/analyst/index.js.map +1 -1
  9. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  10. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  11. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  12. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  13. package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BCafwNrf.js} +662 -605
  14. package/dist/benchmark-command-BCafwNrf.js.map +1 -0
  15. package/dist/benchmarks/index.d.ts +1 -1
  16. package/dist/benchmarks/index.js +1 -1
  17. package/dist/{benchmarks-BEOkuvIg.js → benchmarks-CDSolHq7.js} +4 -4
  18. package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-CDSolHq7.js.map} +1 -1
  19. package/dist/campaign/index.d.ts +7 -5
  20. package/dist/campaign/index.js +5 -3
  21. package/dist/{campaign-CXsdyym7.js → campaign-Tdy3h62h.js} +17 -301
  22. package/dist/campaign-Tdy3h62h.js.map +1 -0
  23. package/dist/cli.js +2 -2
  24. package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
  25. package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  26. package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
  27. package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  28. package/dist/contract/index.d.ts +9 -8
  29. package/dist/contract/index.d.ts.map +1 -1
  30. package/dist/contract/index.js +7 -6
  31. package/dist/contract/index.js.map +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/control.js +1 -1
  34. package/dist/counterfactual-CWPTrMH7.js +126 -0
  35. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  36. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  37. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  38. package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  39. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  40. package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
  41. package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  42. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  43. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  44. package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
  45. package/dist/engine-nB64f48I.d.ts.map +1 -0
  46. package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
  47. package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  48. package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  49. package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  50. package/dist/exec-BLtYZdWo.js +49 -0
  51. package/dist/exec-BLtYZdWo.js.map +1 -0
  52. package/dist/experiment/index.d.ts +802 -0
  53. package/dist/experiment/index.d.ts.map +1 -0
  54. package/dist/experiment/index.js +1108 -0
  55. package/dist/experiment/index.js.map +1 -0
  56. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  57. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  58. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  59. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  60. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
  61. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  62. package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
  63. package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  64. package/dist/fuzz.d.ts +1 -1
  65. package/dist/hosted/index.d.ts +3 -3
  66. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  67. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  68. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  69. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  70. package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
  71. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  72. package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
  73. package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  74. package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
  75. package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  76. package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
  77. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  78. package/dist/index.d.ts +214 -404
  79. package/dist/index.d.ts.map +1 -1
  80. package/dist/index.js +233 -649
  81. package/dist/index.js.map +1 -1
  82. package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
  83. package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  84. package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
  85. package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  86. package/dist/integrity-MLzHOfV9.js +141 -0
  87. package/dist/integrity-MLzHOfV9.js.map +1 -0
  88. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  89. package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
  90. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  91. package/dist/matrix/index.d.ts +2 -2
  92. package/dist/meta-eval/index.d.ts +1 -1
  93. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  94. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  95. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  96. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  97. package/dist/multishot/index.d.ts +2 -2
  98. package/dist/openapi.json +1 -1
  99. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  100. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  101. package/dist/pipelines/index.d.ts +2 -1
  102. package/dist/pipelines/index.d.ts.map +1 -1
  103. package/dist/pre-registration-DakwTRXk.js +96 -0
  104. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  105. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  106. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  107. package/dist/prime-protocol-BfSalTfR.js +453 -0
  108. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  109. package/dist/profile-cell.js +242 -1
  110. package/dist/profile-cell.js.map +1 -0
  111. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  112. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  113. package/dist/promotion-policy-CrLrmys8.js +682 -0
  114. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  115. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  116. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  117. package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
  118. package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  119. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  120. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  121. package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
  122. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  123. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  124. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  125. package/dist/reporting.d.ts +3 -3
  126. package/dist/reporting.js +2 -2
  127. package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
  128. package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  130. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  131. package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
  132. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  133. package/dist/rl.d.ts +17 -7
  134. package/dist/rl.d.ts.map +1 -1
  135. package/dist/rl.js +16 -7
  136. package/dist/rl.js.map +1 -1
  137. package/dist/rollout/index.d.ts +2 -2
  138. package/dist/rollout/index.js +2 -2
  139. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  140. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  141. package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
  142. package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  143. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  144. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  145. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  146. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  147. package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  148. package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  149. package/dist/sequential-D-BLJBKU.js +299 -0
  150. package/dist/sequential-D-BLJBKU.js.map +1 -0
  151. package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
  152. package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
  153. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  154. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  155. package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
  156. package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  157. package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
  158. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  159. package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
  160. package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
  161. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  162. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  163. package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  164. package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  165. package/dist/steps-BArUxhna.d.ts +51 -0
  166. package/dist/steps-BArUxhna.d.ts.map +1 -0
  167. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  168. package/dist/store-DNe_Uv1Q.js.map +1 -0
  169. package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
  170. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  171. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  172. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  173. package/dist/supervisor-run/index.d.ts +2 -2
  174. package/dist/supervisor-run/index.js +1 -1
  175. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  176. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  177. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  178. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  179. package/dist/trace-repair/index.d.ts +2102 -0
  180. package/dist/trace-repair/index.d.ts.map +1 -0
  181. package/dist/trace-repair/index.js +3878 -0
  182. package/dist/trace-repair/index.js.map +1 -0
  183. package/dist/traces.d.ts +6 -6
  184. package/dist/traces.js +3 -2
  185. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  186. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  187. package/dist/trajectory-replay/index.d.ts +781 -0
  188. package/dist/trajectory-replay/index.d.ts.map +1 -0
  189. package/dist/trajectory-replay/index.js +2103 -0
  190. package/dist/trajectory-replay/index.js.map +1 -0
  191. package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
  192. package/dist/types-D216SgwM.d.ts.map +1 -0
  193. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  194. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  195. package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
  196. package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  197. package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
  198. package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  199. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  200. package/dist/verdict-DExhxfgR.d.ts +201 -0
  201. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  202. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  203. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  204. package/dist/wire/index.d.ts +2 -2
  205. package/dist/wire/index.js +1 -1
  206. package/docs/charter.md +112 -0
  207. package/docs/experiment.md +104 -0
  208. package/docs/prime-analyst.md +1 -0
  209. package/docs/trace-analysis.md +26 -0
  210. package/docs/trace-repair-admission.md +194 -0
  211. package/docs/trace-repair-analyst-arms.md +121 -0
  212. package/docs/trace-repair-continuation.md +107 -0
  213. package/docs/trace-repair-grader.md +163 -0
  214. package/docs/trajectory-replay.md +110 -0
  215. package/docs/verification-strategies.md +103 -0
  216. package/package.json +19 -2
  217. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  218. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  219. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  220. package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
  221. package/dist/benchmark-command-CQd78YHt.js.map +0 -1
  222. package/dist/campaign-CXsdyym7.js.map +0 -1
  223. package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
  224. package/dist/index-4XwggC10.d.ts.map +0 -1
  225. package/dist/index-BIL5vxxt.d.ts.map +0 -1
  226. package/dist/integrity-fdt8XPAv.js.map +0 -1
  227. package/dist/llm-client-Dv5BiKLE.js.map +0 -1
  228. package/dist/replay-Krvb114g.d.ts.map +0 -1
  229. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  230. package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
  231. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  232. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  233. package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
  234. package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
  235. package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
  236. package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
  237. package/dist/types-XMVEdrE_.d.ts.map +0 -1
  238. package/dist/verdict-Dps8_okt.d.ts +0 -37
  239. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
package/dist/index.js CHANGED
@@ -1,12 +1,13 @@
1
1
  import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
2
- import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-BciUCZoh.js";
2
+ import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-qgWLA6E9.js";
3
3
  import { a as LimitExceededError, c as ValidationError, i as JudgeError, l as VerificationError, n as CaptureIntegrityError, o as NotFoundError, r as ConfigError, s as ReplayError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
4
- import { _ as verifyManifest, a as assertRunAgentProfileCell, c as groupRunsByAgentProfileCell, d as validateAgentProfileCell, f as verifyAgentProfileCell, g as signManifest, h as hashJson, i as agentProfileCellKey, l as requireAgentProfileCell, m as evaluateHypothesis, n as AgentProfileCellValidationError, o as buildAgentInterfaceProfileCell, p as canonicalize, r as agentProfileCellHashMaterial, s as buildAgentProfileCell, t as AGENT_PROFILE_KINDS, u as toAgentProfileJson } from "./agent-profile-cell-CbfBm2g6.js";
4
+ import { a as verifyManifest, i as signManifest, n as evaluateHypothesis, r as hashJson, t as canonicalize } from "./pre-registration-DakwTRXk.js";
5
+ import { AGENT_PROFILE_KINDS, AgentProfileCellValidationError, agentProfileCellHashMaterial, agentProfileCellKey, assertRunAgentProfileCell, buildAgentInterfaceProfileCell, buildAgentProfileCell, groupRunsByAgentProfileCell, requireAgentProfileCell, toAgentProfileJson, validateAgentProfileCell, verifyAgentProfileCell } from "./profile-cell.js";
5
6
  import { a as estimateTokens, i as estimateCost, n as MetricsCollector, o as isModelPriced, r as TokenCounter, s as resolveModelPricing, t as MODEL_PRICING } from "./metrics-C9YY1OcL.js";
6
7
  import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-DMFxsLKr.js";
7
- import { C as CrossFamilyError, S as servedModelAcceptable, T as judgeFamily, _ as assertCrossFamilyServed, a as assertLlmRoute, b as checkServedModel, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, g as ServedCrossFamilyError, h as ModelSubstitutionError, i as LlmRouteAssertionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError, v as assertServedModel, w as assertCrossFamily, x as normalizeModelId, y as assertServedModels } from "./llm-client-Dv5BiKLE.js";
8
+ import { C as servedModelAcceptable, E as judgeFamily, S as normalizeModelId, T as assertCrossFamily, _ as ServedCrossFamilyError, a as assertLlmRoute, b as assertServedModels, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, g as PROBE_MAX_TOKENS, h as ModelSubstitutionError, i as LlmRouteAssertionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError, v as assertCrossFamilyServed, w as CrossFamilyError, x as checkServedModel, y as assertServedModel } from "./llm-client-DzvMUsS_.js";
8
9
  import { a as providerFromBaseUrl, i as defaultProviderRedactor, n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
9
- import { C as createChatClient, S as computeTraceMetrics, _ as CONTROL_INTEGRITY_ANALYST, a as analystFindingDigest, c as completedAnalystReviewQuality, d as validateAnalystReviewDecisions, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, l as readAnalystReview, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, o as analystRunDigest, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, s as assertUniqueFindingIds, t as buildDefaultAnalystRegistry, u as snapshotAnalystRun, v as ControlIntegrityAnalyst, y as emitControlIntegrityFindings } from "./default-registry-Dta70shL.js";
10
+ import { C as createChatClient, S as computeTraceMetrics, _ as CONTROL_INTEGRITY_ANALYST, a as analystFindingDigest, c as completedAnalystReviewQuality, d as validateAnalystReviewDecisions, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, l as readAnalystReview, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, o as analystRunDigest, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, s as assertUniqueFindingIds, t as buildDefaultAnalystRegistry, u as snapshotAnalystRun, v as ControlIntegrityAnalyst, y as emitControlIntegrityFindings } from "./default-registry-BaQXW1Ow.js";
10
11
  import { INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OUTPUT_VALUE, RUN_COST_ATTR_KEYS, SPAN_KIND_ATTR_KEYS, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, applyLlmSpanOtlpAttributes, asNumber, contextInputTokens, firstNumberAttr } from "./trace-attributes.js";
11
12
  import { C as TraceNotFoundError, G as resolveTraceAnalystLimits, J as extractOtlpAttributes, K as asString, Q as readOtlpStatus, S as TraceFileTooLargeError, W as DEFAULT_TRACE_ANALYST_LIMITS, X as inferOtlpKind, Y as firstStringAttr, Z as projectOtlpFlatLine, _ as TraceAnalysisLimitError, b as TraceFileMalformedError, c as traceAnalystFunctionGroup, et as stringField, g as SpanNotFoundError, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst, it as traceSpanKindToOpenInferenceKind, l as createBoundedTraceAnalysisStore, m as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, n as renderPriorFindings, nt as classifyOtlpSpanRole, o as TRACE_ANALYST_TOOL_NAMESPACE, p as DEFAULT_TRACE_ANALYST_BUDGETS, r as renderUpstreamFindings, rt as isOtlpModelCall, s as buildTraceAnalysisToolDescriptors, t as createTraceAnalyst, tt as applyToolSpanOtlpAttributes, v as TraceAnalysisStoreContractError, x as TraceFileMissingError, y as TraceAnalysisValidationError } from "./kind-factory-BHIgPmzS.js";
12
13
  import { a as computeFindingId, o as makeFinding, s as makeProposalFinding } from "./usage-receipt-EVI8B8Xu.js";
@@ -14,42 +15,46 @@ import { c as SUPERVISOR_RUN_INTEGRITY_SCHEMA, n as supervisorRunRolloutLines, t
14
15
  import { c as assertMintedLines, f as isRolloutLine, i as ROLLOUT_SCHEMA, m as validateRolloutLine, p as isTrainableSplit, s as assertMinted, u as assertRolloutLine } from "./schema-C6DW4ZHR.js";
15
16
  import { a as scoreOrigin, n as observedScore, o as trainingReward, r as observedSplitScore, s as trainingScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
16
17
  import { a as clamp01, i as aggregateRunScore, r as DEFAULT_RUN_SCORE_WEIGHTS } from "./proposal-findings-2GIUo1et.js";
17
- import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-DwF6n05O.js";
18
- import { _ as cachedJudge, b as fileVerdictCache, v as canonicalJson, x as inMemoryVerdictCache, y as contentHash } from "./single-run-lock-D5iN0Xzb.js";
19
- import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-19FQEMBK.js";
18
+ import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-Bmrq6yqU.js";
19
+ import { a as inMemoryVerdictCache, i as fileVerdictCache, n as canonicalJson, r as contentHash, t as cachedJudge } from "./verdict-cache-BCcOh0kF.js";
20
+ import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BiN49gK6.js";
20
21
  import { f as Mutex, p as mapConcurrent } from "./ledger-core-DXZIqu17.js";
21
- import { At as summarizeAgentReceiptIntegrity, B as surfaceContentHash, Ct as decidePairedPromotion, D as runCanaries, Dt as BackendIntegrityError, Et as pairedDeltaTest, G as DEFAULT_MUTATION_PRIMITIVES, K as buildReflectionPrompt, Mt as JudgeParseError, Ot as assertRealAgentReceipts, Tt as minimumPairsForPairedDeltaTest, _t as crowdingDistance, at as REFERENCE_EQUIVALENCE_JUDGE_VERSION, bt as paretoFrontierWithCrowding, ct as DEFAULT_RED_TEAM_CORPUS, dt as scoreRedTeamOutput, ft as toolNamesForRun, gt as llmJudge, ht as hashScenarios, it as REFERENCE_EQUIVALENCE_INPUT_LIMITS, jt as summarizeBackendIntegrity, kt as assertRealBackend, lt as redTeamDataset, mt as HoldoutLockedError, ot as createReferenceEquivalenceJudge, pt as Dataset, q as parseReflectionResponse, st as runReferenceEquivalenceJudge, ut as redTeamReport, vt as dominates, wt as pairedDecisionShape, xt as scalarScore, yt as paretoFrontier } from "./skillopt-optimization-method-Bfb-vBKe.js";
22
+ import { $ as runReferenceEquivalenceJudge, L as DEFAULT_MUTATION_PRIMITIVES, M as surfaceContentHash, Q as createReferenceEquivalenceJudge, R as buildReflectionPrompt, X as REFERENCE_EQUIVALENCE_INPUT_LIMITS, Z as REFERENCE_EQUIVALENCE_JUDGE_VERSION, _t as assertRealBackend, at as Dataset, bt as JudgeParseError, ct as llmJudge, dt as paretoFrontier, et as DEFAULT_RED_TEAM_CORPUS, ft as paretoFrontierWithCrowding, gt as assertRealAgentReceipts, ht as BackendIntegrityError, it as toolNamesForRun, lt as crowdingDistance, nt as redTeamReport, ot as HoldoutLockedError, pt as scalarScore, rt as scoreRedTeamOutput, st as hashScenarios, tt as redTeamDataset, ut as dominates, vt as summarizeAgentReceiptIntegrity, y as runCanaries, yt as summarizeBackendIntegrity, z as parseReflectionResponse } from "./skillopt-optimization-method-CQwZ-ZX8.js";
22
23
  import { $ as selfPreference, A as pairedRiskDifference, B as requiredSampleSize, C as mulberry32, D as pairedCohensDz, E as pairedBootstrap, F as partialCredit, G as wilson, H as weightedComposite, I as passAtK, J as normalCdf, K as studentTCdf, L as pearsonR, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, P as pairedTTest, Q as positionalBias, R as ranks, S as mcnemarRequiredN, T as pairedBinaryScale, U as weightedMean, V as spearmanR, W as wilcoxonSignedRank, X as calibrateJudgeContinuous, Y as calibrateJudge, Z as continuousAgreement, _ as interpretCliffs, a as MANN_WHITNEY_EXACT_MAX_WORK, b as mcnemar, c as bonferroni, d as confidenceInterval, et as verbosityBias, f as corpusInterRaterAgreement, g as interRaterReliability, h as holm, i as MANN_WHITNEY_EXACT_MAX_STATES, j as pairedRiskDifferenceExact, k as pairedMde, l as cliffsDelta, m as eProcess, n as DECISION_PAIRED_DELTA_STATISTIC, o as WILCOXON_EXACT_MAX_N, p as corpusInterRaterAgreementFromJudgeScores, r as DEFAULT_PERMUTATIONS, s as benjaminiHochberg, t as BOOTSTRAP_GATE_MIN_N, u as cohensD, v as isBinaryOutcomeVector, w as normalizeScores, x as mcnemarPower, y as mannWhitneyU, z as requiredPairedSampleSize } from "./statistics-ByxzSiOM.js";
23
24
  import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-iZ08VFMN.js";
25
+ import { a as improvementVerdict, i as gitProvenanceReader, n as computeExperimentStats, o as inMemoryExperimentStore, r as fileExperimentStore, s as clusteredPairedBinary, t as ExperimentTracker } from "./experiment-tracker-CnRICnMl.js";
24
26
  import { n as llmSpanFromProvider, t as TraceEmitter } from "./emitter-CPBAhxum.js";
25
27
  import { a as isRetrievalSpan, i as isLlmSpan, n as TRACE_SCHEMA_VERSION, o as isSandboxSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
26
- import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-CWN8-VsV.js";
27
- import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-Sl0xfkFv.js";
28
+ import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-DqOw5X6_.js";
29
+ import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-Dy39cbFF.js";
30
+ import { d as pairedDecisionShape, f as minimumPairsForPairedDeltaTest, p as pairedDeltaTest, u as decidePairedPromotion } from "./promotion-policy-CrLrmys8.js";
28
31
  import { i as toRewardRows, r as toJsonl, s as toSftRows } from "./exporters-q9iL-2Jf.js";
29
- import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-C2fD1cf4.js";
32
+ import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-2ECTXb2N.js";
30
33
  import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
31
- import { n as unmintableReasons, t as mintRolloutRows } from "./mint-DD-0oQTA.js";
32
- import { C as rollupSupervisorRuns, D as isUnavailable, E as SUPERVISOR_RUN_SCHEMA, O as showMeasured, b as readClaudeCodeSupervisorRun, c as writeSupervisorRunReport, d as readRuntimeSupervisorRun, f as runtimeSupervisorRunReader, h as renderSupervisorRunMarkdown, m as renderSupervisorRunHeadline, t as analyzeSupervisorRun, u as isRuntimeSupervisorRunDir, x as analyzeSupervisorRunSources, y as claudeCodeSupervisorRunReader } from "./supervisor-run-DiyQVczd.js";
33
- import { A as TRACE_ANALYST_ACTOR_DESCRIPTION, C as domainEvidencePattern, D as tokenizeDomainWords, E as scoreTraceInsightReadiness, O as traceAnalystOnRunComplete, S as describeTraceInsightScope, T as planTraceInsightQuestions, _ as otlpToTraceRunRecords, a as convertTraceStoresToOtlp, b as buildTraceInsightPrompt, c as otelRunCompleteHook, d as captureFetchToRawSink, f as ToolTraceMissingError, g as otlpToRunRecords, h as otlpRowsToTraceRunRecords, i as iterateRawCalls, j as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, k as analyzeTraces, l as OTEL_AGENT_EVAL_SCOPE, m as otlpRowsToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as toolSpansToTraceAnalysisStore, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as flattenOtlpExportToNdjson, w as inferDomainKeywords, x as defaultTraceInsightPanel, y as buildTraceInsightContext } from "./replay-GW61ezMW.js";
34
+ import { n as unmintableReasons, t as mintRolloutRows } from "./mint-CGEkzPLf.js";
35
+ import { C as rollupSupervisorRuns, D as isUnavailable, E as SUPERVISOR_RUN_SCHEMA, O as showMeasured, b as readClaudeCodeSupervisorRun, c as writeSupervisorRunReport, d as readRuntimeSupervisorRun, f as runtimeSupervisorRunReader, h as renderSupervisorRunMarkdown, m as renderSupervisorRunHeadline, t as analyzeSupervisorRun, u as isRuntimeSupervisorRunDir, x as analyzeSupervisorRunSources, y as claudeCodeSupervisorRunReader } from "./supervisor-run-D_sokXcO.js";
36
+ import { A as TRACE_ANALYST_ACTOR_DESCRIPTION, C as domainEvidencePattern, D as tokenizeDomainWords, E as scoreTraceInsightReadiness, O as traceAnalystOnRunComplete, S as describeTraceInsightScope, T as planTraceInsightQuestions, _ as otlpToTraceRunRecords, a as convertTraceStoresToOtlp, b as buildTraceInsightPrompt, c as otelRunCompleteHook, d as captureFetchToRawSink, f as ToolTraceMissingError, g as otlpToRunRecords, h as otlpRowsToTraceRunRecords, i as iterateRawCalls, j as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, k as analyzeTraces, l as OTEL_AGENT_EVAL_SCOPE, m as otlpRowsToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as toolSpansToTraceAnalysisStore, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as flattenOtlpExportToNdjson, w as inferDomainKeywords, x as defaultTraceInsightPanel, y as buildTraceInsightContext } from "./replay-DyBLaKFc.js";
34
37
  import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-CKtTpRhv.js";
35
38
  import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BW27f3XW.js";
36
- import { B as agentProfileModelId, G as createLlmCorrectnessChecker, H as harnessAxisOf, I as CODING_HARNESSES, J as verifyCompletion, K as createTokenRecallChecker, L as HARNESS_NATIVE_MODEL, R as agentProfileHash, U as extractProducedState, V as expandProfileAxes, W as completionVerdict, q as parseCorrectnessResponse, z as agentProfileId } from "./campaign-CXsdyym7.js";
39
+ import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-Tdy3h62h.js";
37
40
  import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-C-GocmIW.js";
38
41
  import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-Di7eEQ79.js";
39
- import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-DWIvOAGk.js";
40
- import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-9A5y7EsK.js";
42
+ import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-BNNK7irB.js";
43
+ import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-Lf-5I7xh.js";
41
44
  import { a as composeParsers, c as vitestTestParser, i as SubprocessSandboxDriver, n as DockerSandboxDriver, o as jestTestParser, r as SandboxHarness, s as pytestTestParser, t as runTestGradedScenario } from "./test-graded-scenario-JHcKQNpq.js";
42
- import { a as InMemoryTraceStore, i as FileSystemTraceStore, n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-fdt8XPAv.js";
45
+ import { n as InMemoryTraceStore, t as FileSystemTraceStore } from "./store-DNe_Uv1Q.js";
46
+ import { n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-MLzHOfV9.js";
43
47
  import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
44
48
  import { n as DEFAULT_RULES, r as classifyFailure, t as computeToolUseMetrics } from "./tool-use-metrics-DEGMKycK.js";
45
49
  import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
46
- import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-BEOkuvIg.js";
47
- import { t as runEvalCampaign } from "./eval-campaign-CfLQQs9B.js";
50
+ import { n as runCounterfactual, t as attributeCounterfactuals } from "./counterfactual-CWPTrMH7.js";
51
+ import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-CDSolHq7.js";
52
+ import { t as runEvalCampaign } from "./eval-campaign-DNjCvAm-.js";
48
53
  import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
49
54
  import { accessSync, appendFileSync, constants, cpSync, existsSync, mkdirSync, promises, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
50
55
  import { basename, delimiter, dirname, extname, isAbsolute, join, relative, resolve } from "node:path";
51
56
  import { createHash } from "node:crypto";
52
- import { execSync, spawnSync } from "node:child_process";
57
+ import { spawnSync } from "node:child_process";
53
58
  import { readFile } from "node:fs/promises";
54
59
  import { cpus } from "node:os";
55
60
  import { gzipSync } from "node:zlib";
@@ -423,6 +428,10 @@ async function executeScenario(chat, scenario, config) {
423
428
  receiptFromError: costReceiptFromLlmError
424
429
  });
425
430
  if (!paid.succeeded) throw paid.error;
431
+ assertServedModel(model, paid.value.servedModel, {
432
+ allowUnreported: true,
433
+ context: `executeScenario "${scenario.id}" turn ${i}`
434
+ });
426
435
  const rawContent = paid.value.content;
427
436
  if (typeof rawContent !== "string") throw new CaptureIntegrityError(`chat response for scenario "${scenario.id}" turn ${i} is malformed: expected content to be a string, got ${rawContent === null ? "null" : typeof rawContent}`);
428
437
  const content = rawContent;
@@ -1100,235 +1109,6 @@ async function runE2EWorkflow(client, name, workflow) {
1100
1109
  };
1101
1110
  }
1102
1111
  //#endregion
1103
- //#region src/clustered-paired-binary.ts
1104
- /**
1105
- * Paired binary comparison for work items nested inside independent clusters.
1106
- *
1107
- * Pairing is delegated to {@link pairArms}; this module adds the cluster-aware
1108
- * estimands and inference that task-level McNemar/bootstrap utilities cannot
1109
- * provide. Callers keep their own row shape through accessors, and every
1110
- * matched or unpaired result returns the original row object unchanged.
1111
- */
1112
- const DEFAULT_BOOTSTRAP_RESAMPLES = 1e4;
1113
- const DEFAULT_SIGN_FLIP_RESAMPLES = 1e5;
1114
- const MAX_RESAMPLES = 1e6;
1115
- const DEFAULT_EXACT_CLUSTER_LIMIT = 20;
1116
- const SIGN_FLIP_SEED_SALT = 2654435769;
1117
- /**
1118
- * Compare binary outcomes on matched work items while respecting independent
1119
- * clusters. The confidence interval resamples whole clusters and recomputes the
1120
- * task-weighted risk difference. The sign-flip test flips whole-cluster outcome
1121
- * totals and tests that same task-weighted estimand.
1122
- */
1123
- function clusteredPairedBinary(rows, options) {
1124
- const config = validateOptions(options);
1125
- const paired = pairArms(projectSelectedRows(rows, options), {
1126
- baselineArm: options.baselineArm,
1127
- treatmentArm: options.treatmentArm
1128
- });
1129
- const matchedPairs = paired.pairs.map((pair) => {
1130
- const baseline = pair.baseline;
1131
- const treatment = pair.treatment;
1132
- if (baseline.clusterKey !== treatment.clusterKey) throw new ValidationError(`clusteredPairedBinary: pairKey '${pair.pairKey}' rep ${pair.repIndex} crosses clusters ('${baseline.clusterKey}' vs '${treatment.clusterKey}')`);
1133
- return {
1134
- pairKey: pair.pairKey,
1135
- repIndex: pair.repIndex,
1136
- clusterKey: baseline.clusterKey,
1137
- baseline: baseline.original,
1138
- treatment: treatment.original,
1139
- baselinePass: baseline.pass,
1140
- treatmentPass: treatment.pass
1141
- };
1142
- });
1143
- const unpairedBaseline = paired.unpairedBaseline.map((row) => row.original);
1144
- const unpairedTreatment = paired.unpairedTreatment.map((row) => row.original);
1145
- if (matchedPairs.length === 0) return {
1146
- matchedPairs,
1147
- unpairedBaseline,
1148
- unpairedTreatment,
1149
- statistics: null
1150
- };
1151
- const clusters = summarizeClusters(matchedPairs);
1152
- const b10 = clusters.reduce((sum, cluster) => sum + cluster.b10, 0);
1153
- const b01 = clusters.reduce((sum, cluster) => sum + cluster.b01, 0);
1154
- const taskWeightedRiskDifference = (b10 - b01) / matchedPairs.length;
1155
- const equalClusterMean = mean$4(clusters.map((cluster) => cluster.meanDifference));
1156
- const bootstrap = clusters.length < 2 ? null : clusterBootstrap(clusters, config);
1157
- const signFlip = clusterSignFlip(clusters, config);
1158
- return {
1159
- matchedPairs,
1160
- unpairedBaseline,
1161
- unpairedTreatment,
1162
- statistics: {
1163
- nPairs: matchedPairs.length,
1164
- nClusters: clusters.length,
1165
- b10,
1166
- b01,
1167
- taskWeightedRiskDifference,
1168
- equalClusterMean,
1169
- clusters,
1170
- bootstrap,
1171
- signFlip
1172
- }
1173
- };
1174
- }
1175
- function validateOptions(options) {
1176
- assertNonEmptyString("baselineArm", options.baselineArm);
1177
- assertNonEmptyString("treatmentArm", options.treatmentArm);
1178
- if (options.baselineArm === options.treatmentArm) throw new ValidationError(`clusteredPairedBinary: baselineArm and treatmentArm are both '${options.baselineArm}'`);
1179
- if (!Number.isInteger(options.seed)) throw new ValidationError(`clusteredPairedBinary: seed must be an integer, got ${options.seed}`);
1180
- const confidence = options.confidence ?? .95;
1181
- if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new ValidationError(`clusteredPairedBinary: confidence must be in (0,1), got ${confidence}`);
1182
- const bootstrapResamples = options.bootstrapResamples ?? DEFAULT_BOOTSTRAP_RESAMPLES;
1183
- assertResampleCount("bootstrapResamples", bootstrapResamples);
1184
- const rawMinimumBootstrapResamples = 2 / (1 - confidence);
1185
- const minimumBootstrapResamples = Math.ceil(rawMinimumBootstrapResamples - Number.EPSILON * Math.max(1, rawMinimumBootstrapResamples) * 8);
1186
- if (bootstrapResamples < minimumBootstrapResamples) throw new ValidationError(`clusteredPairedBinary: bootstrapResamples must be at least ${minimumBootstrapResamples} for confidence ${confidence} so both interval tails are represented, got ${bootstrapResamples}`);
1187
- const signFlipResamples = options.signFlipResamples ?? DEFAULT_SIGN_FLIP_RESAMPLES;
1188
- assertResampleCount("signFlipResamples", signFlipResamples);
1189
- const exactClusterLimit = options.exactClusterLimit ?? DEFAULT_EXACT_CLUSTER_LIMIT;
1190
- if (!Number.isInteger(exactClusterLimit) || exactClusterLimit < 0 || exactClusterLimit > DEFAULT_EXACT_CLUSTER_LIMIT) throw new ValidationError(`clusteredPairedBinary: exactClusterLimit must be an integer in [0,${DEFAULT_EXACT_CLUSTER_LIMIT}], got ${exactClusterLimit}`);
1191
- const alternative = options.alternative ?? "two-sided";
1192
- if (alternative !== "two-sided" && alternative !== "greater" && alternative !== "less") throw new ValidationError(`clusteredPairedBinary: alternative must be 'two-sided', 'greater', or 'less', got ${String(alternative)}`);
1193
- return {
1194
- seed: options.seed,
1195
- confidence,
1196
- bootstrapResamples,
1197
- alternative,
1198
- exactClusterLimit,
1199
- signFlipResamples
1200
- };
1201
- }
1202
- function assertResampleCount(name, value) {
1203
- if (!Number.isInteger(value) || value <= 0) throw new ValidationError(`clusteredPairedBinary: ${name} must be a positive integer, got ${value}`);
1204
- if (value > MAX_RESAMPLES) throw new ValidationError(`clusteredPairedBinary: ${name} must not exceed ${MAX_RESAMPLES}, got ${value}`);
1205
- }
1206
- function projectSelectedRows(rows, options) {
1207
- const projected = [];
1208
- for (const original of rows) {
1209
- const arm = options.arm(original);
1210
- assertNonEmptyString("arm", arm);
1211
- if (arm !== options.baselineArm && arm !== options.treatmentArm) continue;
1212
- const pairKey = options.pairKey(original);
1213
- const clusterKey = options.clusterKey(original);
1214
- const pass = options.pass(original);
1215
- const repKey = options.repKey?.(original);
1216
- assertNonEmptyString("pairKey", pairKey);
1217
- assertNonEmptyString("clusterKey", clusterKey);
1218
- if (typeof pass !== "boolean") throw new ValidationError(`clusteredPairedBinary: pass accessor must return boolean for pairKey '${pairKey}'`);
1219
- if (repKey !== void 0) assertNonEmptyString("repKey", repKey);
1220
- projected.push({
1221
- pairKey,
1222
- clusterKey,
1223
- arm,
1224
- pass,
1225
- repKey,
1226
- original
1227
- });
1228
- }
1229
- return projected;
1230
- }
1231
- function assertNonEmptyString(name, value) {
1232
- if (typeof value !== "string" || value.trim().length === 0) throw new ValidationError(`clusteredPairedBinary: ${name} accessor must return a non-empty string`);
1233
- }
1234
- function summarizeClusters(pairs) {
1235
- const byCluster = /* @__PURE__ */ new Map();
1236
- for (const pair of pairs) {
1237
- const summary = byCluster.get(pair.clusterKey) ?? {
1238
- nPairs: 0,
1239
- b10: 0,
1240
- b01: 0
1241
- };
1242
- summary.nPairs++;
1243
- if (pair.treatmentPass && !pair.baselinePass) summary.b10++;
1244
- else if (pair.baselinePass && !pair.treatmentPass) summary.b01++;
1245
- byCluster.set(pair.clusterKey, summary);
1246
- }
1247
- return [...byCluster.entries()].sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0).map(([clusterKey, summary]) => ({
1248
- clusterKey,
1249
- ...summary,
1250
- meanDifference: (summary.b10 - summary.b01) / summary.nPairs
1251
- }));
1252
- }
1253
- function clusterBootstrap(clusters, config) {
1254
- const rng = mulberry32(config.seed);
1255
- const samples = new Array(config.bootstrapResamples);
1256
- for (let draw = 0; draw < config.bootstrapResamples; draw++) {
1257
- let differenceSum = 0;
1258
- let pairCount = 0;
1259
- for (let index = 0; index < clusters.length; index++) {
1260
- const cluster = clusters[Math.floor(rng() * clusters.length)];
1261
- differenceSum += cluster.b10 - cluster.b01;
1262
- pairCount += cluster.nPairs;
1263
- }
1264
- samples[draw] = differenceSum / pairCount;
1265
- }
1266
- samples.sort((a, b) => a - b);
1267
- const alpha = 1 - config.confidence;
1268
- const lowerIndex = Math.floor(alpha / 2 * config.bootstrapResamples);
1269
- const upperIndex = Math.min(config.bootstrapResamples - 1, Math.ceil((1 - alpha / 2) * config.bootstrapResamples) - 1);
1270
- return {
1271
- statistic: "task-weighted-risk-difference",
1272
- lower: samples[lowerIndex],
1273
- upper: samples[Math.max(lowerIndex, upperIndex)],
1274
- confidence: config.confidence,
1275
- resamples: config.bootstrapResamples,
1276
- seed: config.seed
1277
- };
1278
- }
1279
- function clusterSignFlip(clusters, config) {
1280
- const clusterTotals = clusters.map((cluster) => cluster.b10 - cluster.b01);
1281
- const nonZero = clusterTotals.filter((delta) => delta !== 0);
1282
- const totalPairs = clusters.reduce((sum, cluster) => sum + cluster.nPairs, 0);
1283
- const statistic = clusterTotals.reduce((sum, delta) => sum + delta, 0) / totalPairs;
1284
- if (nonZero.length <= config.exactClusterLimit) {
1285
- const assignments = 2 ** nonZero.length;
1286
- let extreme = 0;
1287
- for (let mask = 0; mask < assignments; mask++) {
1288
- let sum = 0;
1289
- for (let index = 0; index < nonZero.length; index++) sum += (mask & 2 ** index ? 1 : -1) * nonZero[index];
1290
- if (isExtreme(sum / totalPairs, statistic, config.alternative)) extreme++;
1291
- }
1292
- return {
1293
- statistic,
1294
- pValue: extreme / assignments,
1295
- alternative: config.alternative,
1296
- method: "exact",
1297
- assignments,
1298
- nClusters: clusters.length,
1299
- nNonZeroClusters: nonZero.length,
1300
- seed: null
1301
- };
1302
- }
1303
- const signFlipSeed = (config.seed ^ SIGN_FLIP_SEED_SALT) >>> 0;
1304
- const rng = mulberry32(signFlipSeed);
1305
- let extreme = 0;
1306
- for (let draw = 0; draw < config.signFlipResamples; draw++) {
1307
- let sum = 0;
1308
- for (const delta of nonZero) sum += (rng() < .5 ? -1 : 1) * delta;
1309
- if (isExtreme(sum / totalPairs, statistic, config.alternative)) extreme++;
1310
- }
1311
- return {
1312
- statistic,
1313
- pValue: (extreme + 1) / (config.signFlipResamples + 1),
1314
- alternative: config.alternative,
1315
- method: "monte-carlo",
1316
- assignments: config.signFlipResamples,
1317
- nClusters: clusters.length,
1318
- nNonZeroClusters: nonZero.length,
1319
- seed: signFlipSeed
1320
- };
1321
- }
1322
- function isExtreme(candidate, observed, alternative) {
1323
- const tolerance = Number.EPSILON * Math.max(1, Math.abs(observed)) * 16;
1324
- if (alternative === "greater") return candidate >= observed - tolerance;
1325
- if (alternative === "less") return candidate <= observed + tolerance;
1326
- return Math.abs(candidate) >= Math.abs(observed) - tolerance;
1327
- }
1328
- function mean$4(values) {
1329
- return values.reduce((sum, value) => sum + value, 0) / values.length;
1330
- }
1331
- //#endregion
1332
1112
  //#region src/convergence.ts
1333
1113
  /**
1334
1114
  * ConvergenceTracker — tracks completion percentage over turns.
@@ -1606,6 +1386,10 @@ async function decideNextUserTurn(chat, opts) {
1606
1386
  receiptFromError: costReceiptFromLlmError
1607
1387
  });
1608
1388
  if (!paid.succeeded) throw paid.error;
1389
+ assertServedModel(model, paid.value.servedModel, {
1390
+ allowUnreported: true,
1391
+ context: "decideNextUserTurn"
1392
+ });
1609
1393
  return paid.value.content.trim();
1610
1394
  }
1611
1395
  //#endregion
@@ -2132,9 +1916,10 @@ function canonicalize$1(value) {
2132
1916
  * - membership (free): GET `{baseUrl}/models` once; a model is `listed` when
2133
1917
  * its id is in the served set.
2134
1918
  * - probe (spends a tiny number of tokens): POST `{baseUrl}/chat/completions`
2135
- * per model with a 1-message, 5-token request; `served` is whether the
2136
- * router returns 2xx, with the HTTP `status` and the body's `error.message`
2137
- * captured in `detail`, and `servedModel` recording WHICH model answered.
1919
+ * per model with a 1-message, `PROBE_MAX_TOKENS`-token request; `served` is
1920
+ * whether the router reached a provider, with the HTTP `status` and the
1921
+ * body's `error.message` captured in `detail`, and `servedModel` recording
1922
+ * WHICH model answered.
2138
1923
  *
2139
1924
  * A 2xx is not proof the requested model answered — a gateway can accept one
2140
1925
  * id and route to another. The probe therefore compares the echoed id against
@@ -2147,6 +1932,12 @@ function canonicalize$1(value) {
2147
1932
  * `assertModelsServed` and it surfaces every dead or substituted id with its
2148
1933
  * status + detail instead of silently producing a stub or mislabelled run.
2149
1934
  */
1935
+ /**
1936
+ * Provider signature for "the budget ran out before an answer". The model is
1937
+ * alive — a provider took the request and consumed the budget — so this must
1938
+ * never be scored as a dead id.
1939
+ */
1940
+ const REASONING_BUDGET_EXHAUSTED = /reasoning[\s_-]?budget[\s_-]?exhausted/i;
2150
1941
  function stripSlash$1(url) {
2151
1942
  return url.replace(/\/+$/, "");
2152
1943
  }
@@ -2165,13 +1956,19 @@ function errorMessage(body) {
2165
1956
  * fallbacks.
2166
1957
  *
2167
1958
  * The membership check (one GET) always runs. When `probe` is true, each model
2168
- * additionally gets a 1-token chat probe so a model that is listed but
1959
+ * additionally gets a small chat probe so a model that is listed but
2169
1960
  * unconfigured (a 401 `model_not_found` from the router) is caught.
2170
1961
  */
2171
1962
  async function preflightModels(opts) {
2172
1963
  const fetchImpl = opts.fetchImpl ?? fetch;
2173
1964
  const baseUrl = stripSlash$1(opts.baseUrl);
2174
1965
  const authHeaders = { authorization: `Bearer ${opts.apiKey}` };
1966
+ const maxTokens = opts.probeMaxTokens ?? 64;
1967
+ if (!Number.isInteger(maxTokens) || maxTokens <= 0) return {
1968
+ succeeded: false,
1969
+ value: null,
1970
+ error: `preflightModels: probeMaxTokens must be a positive integer, got ${maxTokens}`
1971
+ };
2175
1972
  let served;
2176
1973
  try {
2177
1974
  const res = await fetchImpl(`${baseUrl}/models`, {
@@ -2206,6 +2003,7 @@ async function preflightModels(opts) {
2206
2003
  served: null,
2207
2004
  status: null,
2208
2005
  detail: null,
2006
+ budgetExhausted: false,
2209
2007
  substitution: null
2210
2008
  });
2211
2009
  continue;
@@ -2223,22 +2021,28 @@ async function preflightModels(opts) {
2223
2021
  role: "user",
2224
2022
  content: "ping"
2225
2023
  }],
2226
- max_tokens: 5
2024
+ max_tokens: maxTokens
2227
2025
  })
2228
2026
  });
2229
2027
  let detail = null;
2230
2028
  let substitution = null;
2029
+ let budgetExhausted = false;
2231
2030
  const body = await res.json().catch(() => null);
2232
2031
  if (res.ok) {
2233
2032
  const echoed = body?.model;
2234
2033
  substitution = checkServedModel(model, typeof echoed === "string" && echoed.trim() !== "" ? echoed : null);
2235
- } else detail = errorMessage(body);
2034
+ } else {
2035
+ detail = errorMessage(body);
2036
+ budgetExhausted = detail !== null && REASONING_BUDGET_EXHAUSTED.test(detail);
2037
+ if (budgetExhausted) substitution = checkServedModel(model, null);
2038
+ }
2236
2039
  results.push({
2237
2040
  model,
2238
2041
  listed,
2239
- served: res.ok,
2042
+ served: res.ok || budgetExhausted,
2240
2043
  status: res.status,
2241
2044
  detail,
2045
+ budgetExhausted,
2242
2046
  substitution
2243
2047
  });
2244
2048
  } catch (err) {
@@ -2269,6 +2073,7 @@ function describeFailure(r) {
2269
2073
  return `${r.model}: not in /models${probeNote}`;
2270
2074
  }
2271
2075
  if (r.served === false) return `${r.model}: listed but probe ${r.status}${r.detail ? ` — ${r.detail}` : ""}`;
2076
+ if (r.budgetExhausted) return `${r.model}: alive but the probe ran out of reasoning budget (status ${r.status}${r.detail ? `: ${r.detail}` : ""}) — it echoed no model id, so identity is unproven. Raise probeMaxTokens, or pass allowUnreported to accept reachability without identity.`;
2272
2077
  const s = r.substitution;
2273
2078
  if (s?.verdict === "unreported") return `${r.model}: probe answered without echoing a model id — identity unproven`;
2274
2079
  return `${r.model}: probe answered by ${s?.served} (${s?.verdict})`;
@@ -5047,269 +4852,6 @@ var EvalTraceStore = class {
5047
4852
  }
5048
4853
  };
5049
4854
  //#endregion
5050
- //#region src/experiment-tracker.ts
5051
- /**
5052
- * Experiment tracker — git-provenanced experiment log with N-rep stats and a
5053
- * KEEP / REGRESSION / NOISE verdict against a parent.
5054
- *
5055
- * Every loop the fleet runs reduces to the same question: "I ran the candidate
5056
- * N times — is the median measurably better than the parent, or is the delta
5057
- * inside the noise band?" The hand-rolled copies bake a fixed score scale
5058
- * (percentage points), a fixed store path (`.evolve/experiments-v2.json`), and
5059
- * `execSync('git …')` straight into the module. This is the canonical version:
5060
- * provenance and persistence are injected, thresholds are configurable, and the
5061
- * stats + verdict are pure functions you can unit-test without a git repo or a
5062
- * filesystem.
5063
- *
5064
- * Stats per experiment: median / mean / min / max / iqr / stddev / passRate /
5065
- * n, plus a `stable` flag (`iqr < iqrUnstableAbove && stddev < stddevUnstableAbove`).
5066
- *
5067
- * Verdict against a parent (both must have `n >= minRepsForVerdict`):
5068
- * - NOISE — the candidate is too unstable to judge (`!stable`)
5069
- * - KEEP — `medianDelta > keepThreshold`
5070
- * - REGRESSION — `medianDelta < -regressionThreshold`
5071
- * - NOISE — otherwise (delta inside the band)
5072
- * With no parent (or insufficient reps) the verdict is the neutral ITERATE.
5073
- */
5074
- const DEFAULTS = {
5075
- keepThreshold: 5,
5076
- regressionThreshold: 5,
5077
- iqrUnstableAbove: 10,
5078
- stddevUnstableAbove: Number.POSITIVE_INFINITY,
5079
- minRepsForVerdict: 3
5080
- };
5081
- function resolveThresholds(t) {
5082
- const r = {
5083
- ...DEFAULTS,
5084
- ...t ?? {}
5085
- };
5086
- if (r.keepThreshold < 0) throw new ValidationError(`experiment-tracker: keepThreshold must be >= 0, got ${r.keepThreshold}`);
5087
- if (r.regressionThreshold < 0) throw new ValidationError(`experiment-tracker: regressionThreshold must be >= 0, got ${r.regressionThreshold}`);
5088
- if (r.minRepsForVerdict < 1) throw new ValidationError(`experiment-tracker: minRepsForVerdict must be >= 1, got ${r.minRepsForVerdict}`);
5089
- return r;
5090
- }
5091
- function median$1(sorted) {
5092
- const n = sorted.length;
5093
- if (n === 0) return 0;
5094
- const mid = Math.floor(n / 2);
5095
- return n % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
5096
- }
5097
- /** Population standard deviation (÷n). 0 for fewer than 2 values. */
5098
- function stddev(values, mean) {
5099
- if (values.length < 2) return 0;
5100
- const variance = values.reduce((acc, v) => acc + (v - mean) ** 2, 0) / values.length;
5101
- return Math.sqrt(variance);
5102
- }
5103
- /**
5104
- * Compute the N-rep statistics for a set of reps. Pure — no I/O. The `stable`
5105
- * flag is the trust gate the verdict depends on: a sample whose spread exceeds
5106
- * the configured bounds can't distinguish a real delta from run-to-run noise.
5107
- */
5108
- function computeExperimentStats(reps, thresholds) {
5109
- const t = resolveThresholds(thresholds);
5110
- const n = reps.length;
5111
- if (n === 0) return {
5112
- median: 0,
5113
- mean: 0,
5114
- min: 0,
5115
- max: 0,
5116
- iqr: 0,
5117
- stddev: 0,
5118
- passRate: null,
5119
- n: 0,
5120
- stable: false
5121
- };
5122
- const scores = reps.map((r) => {
5123
- if (!Number.isFinite(r.score)) throw new ValidationError(`experiment-tracker: rep ${r.rep} has non-finite score ${r.score}`);
5124
- return r.score;
5125
- });
5126
- const sorted = [...scores].sort((a, b) => a - b);
5127
- const mean = scores.reduce((s, v) => s + v, 0) / n;
5128
- const sd = stddev(scores, mean);
5129
- const spread = iqr(scores);
5130
- const rated = reps.filter((r) => typeof r.passed === "boolean");
5131
- const passRate = rated.length === 0 ? null : rated.filter((r) => r.passed).length / rated.length;
5132
- const stable = spread < t.iqrUnstableAbove && sd < t.stddevUnstableAbove;
5133
- return {
5134
- median: median$1(sorted),
5135
- mean,
5136
- min: sorted[0],
5137
- max: sorted[n - 1],
5138
- iqr: spread,
5139
- stddev: sd,
5140
- passRate,
5141
- n,
5142
- stable
5143
- };
5144
- }
5145
- /**
5146
- * Verdict for a candidate against its parent. Pure — operates on already-computed
5147
- * stats. KEEP/REGRESSION require both sides to have `>= minRepsForVerdict` reps
5148
- * AND the candidate to be `stable`; otherwise the result is NOISE (unstable) or
5149
- * ITERATE (not enough reps / no parent).
5150
- */
5151
- function improvementVerdict(candidate, parent, thresholds) {
5152
- const t = resolveThresholds(thresholds);
5153
- if (!parent) return {
5154
- verdict: "ITERATE",
5155
- medianDelta: null,
5156
- reason: "no parent experiment to compare against"
5157
- };
5158
- if (candidate.n < t.minRepsForVerdict || parent.n < t.minRepsForVerdict) return {
5159
- verdict: "ITERATE",
5160
- medianDelta: null,
5161
- reason: `need >= ${t.minRepsForVerdict} reps on both sides (candidate n=${candidate.n}, parent n=${parent.n})`
5162
- };
5163
- if (!candidate.stable) return {
5164
- verdict: "NOISE",
5165
- medianDelta: candidate.median - parent.median,
5166
- reason: `candidate unstable (iqr=${candidate.iqr}, stddev=${candidate.stddev.toFixed(2)})`
5167
- };
5168
- const medianDelta = candidate.median - parent.median;
5169
- if (medianDelta > t.keepThreshold) return {
5170
- verdict: "KEEP",
5171
- medianDelta,
5172
- reason: `median +${medianDelta} > +${t.keepThreshold}`
5173
- };
5174
- if (medianDelta < -t.regressionThreshold) return {
5175
- verdict: "REGRESSION",
5176
- medianDelta,
5177
- reason: `median ${medianDelta} < -${t.regressionThreshold}`
5178
- };
5179
- return {
5180
- verdict: "NOISE",
5181
- medianDelta,
5182
- reason: `median delta ${medianDelta} inside noise band [-${t.regressionThreshold}, +${t.keepThreshold}]`
5183
- };
5184
- }
5185
- /**
5186
- * Default provenance reader: `git rev-parse HEAD`, the subject line, and the
5187
- * files changed vs `HEAD~1`. Fail-loud — a tracker that silently logs
5188
- * `commit: 'unknown'` corrupts the provenance the whole point of the log is to
5189
- * carry. When the working tree genuinely has no parent commit, pass an override.
5190
- */
5191
- const gitProvenanceReader = () => {
5192
- const run = (cmd) => execSync(cmd, { encoding: "utf8" }).trim();
5193
- const commit = run("git rev-parse --short HEAD");
5194
- const message = run("git log -1 --format=%s");
5195
- const changedRaw = run("git diff --name-only HEAD~1");
5196
- return {
5197
- commit,
5198
- message,
5199
- changedFiles: changedRaw.length === 0 ? [] : changedRaw.split("\n").filter(Boolean)
5200
- };
5201
- };
5202
- /** In-memory store — the default when no persistence is wanted (tests, ephemeral
5203
- * runs). State lives on the instance. */
5204
- function inMemoryExperimentStore(initial = []) {
5205
- let state = initial.map((e) => structuredClone(e));
5206
- return {
5207
- async load() {
5208
- return state.map((e) => structuredClone(e));
5209
- },
5210
- async save(experiments) {
5211
- state = experiments.map((e) => structuredClone(e));
5212
- }
5213
- };
5214
- }
5215
- /** Filesystem store — a single JSON array at `path`, created on first save. */
5216
- function fileExperimentStore(path) {
5217
- return {
5218
- async load() {
5219
- const fs = await import("node:fs/promises");
5220
- try {
5221
- const raw = await fs.readFile(path, "utf8");
5222
- const parsed = JSON.parse(raw);
5223
- if (!Array.isArray(parsed)) throw new ValidationError(`experiment-tracker: store at ${path} is not a JSON array`);
5224
- return parsed;
5225
- } catch (err) {
5226
- if (err.code === "ENOENT") return [];
5227
- throw err;
5228
- }
5229
- },
5230
- async save(experiments) {
5231
- const fs = await import("node:fs/promises");
5232
- const pathMod = await import("node:path");
5233
- await fs.mkdir(pathMod.dirname(path), { recursive: true });
5234
- await fs.writeFile(path, JSON.stringify(experiments, null, 2), "utf8");
5235
- }
5236
- };
5237
- }
5238
- /**
5239
- * Stateful tracker over an `ExperimentStore`. Create an experiment (provenance
5240
- * is captured once), append reps as they complete (stats + verdict recompute on
5241
- * every append), and read the log back for a dashboard. All persistence and git
5242
- * access flow through the injected seams, so the tracker is fully testable
5243
- * without a repo or disk.
5244
- */
5245
- var ExperimentTracker = class {
5246
- store;
5247
- provenanceReader;
5248
- thresholds;
5249
- now;
5250
- constructor(options = {}) {
5251
- this.store = options.store ?? inMemoryExperimentStore();
5252
- this.provenanceReader = options.provenanceReader ?? gitProvenanceReader;
5253
- this.thresholds = resolveThresholds(options.thresholds);
5254
- this.now = options.now ?? Date.now;
5255
- }
5256
- async create(input) {
5257
- const experiments = await this.store.load();
5258
- if (experiments.some((e) => e.id === input.id)) throw new ValidationError(`experiment-tracker: experiment id "${input.id}" already exists`);
5259
- if (input.parentId && !experiments.some((e) => e.id === input.parentId)) throw new ValidationError(`experiment-tracker: parent experiment "${input.parentId}" not found`);
5260
- const provenance = input.provenance ?? await this.provenanceReader();
5261
- const experiment = {
5262
- id: input.id,
5263
- label: input.label,
5264
- provenance,
5265
- parentId: input.parentId,
5266
- changeSummary: input.changeSummary,
5267
- reps: [],
5268
- stats: computeExperimentStats([], this.thresholds),
5269
- verdict: "ITERATE",
5270
- createdAt: new Date(this.now()).toISOString()
5271
- };
5272
- experiments.push(experiment);
5273
- await this.store.save(experiments);
5274
- return structuredClone(experiment);
5275
- }
5276
- /** Append a rep (its `rep` index defaults to the current rep count) and
5277
- * recompute stats + verdict. Returns the updated experiment. */
5278
- async addRep(experimentId, rep) {
5279
- const experiments = await this.store.load();
5280
- const exp = experiments.find((e) => e.id === experimentId);
5281
- if (!exp) throw new ValidationError(`experiment-tracker: experiment "${experimentId}" not found`);
5282
- const fullRep = {
5283
- rep: rep.rep ?? exp.reps.length,
5284
- score: rep.score,
5285
- passed: rep.passed,
5286
- metrics: rep.metrics,
5287
- timestamp: rep.timestamp ?? new Date(this.now()).toISOString()
5288
- };
5289
- exp.reps.push(fullRep);
5290
- exp.stats = computeExperimentStats(exp.reps, this.thresholds);
5291
- const parent = exp.parentId ? experiments.find((e) => e.id === exp.parentId) : void 0;
5292
- exp.verdict = improvementVerdict(exp.stats, parent?.stats ?? null, this.thresholds).verdict;
5293
- await this.store.save(experiments);
5294
- return structuredClone(exp);
5295
- }
5296
- async get(experimentId) {
5297
- const found = (await this.store.load()).find((e) => e.id === experimentId);
5298
- return found ? structuredClone(found) : void 0;
5299
- }
5300
- async list() {
5301
- return this.store.load();
5302
- }
5303
- /** Full verdict (not just the enum) for an experiment vs its parent. */
5304
- async verdictFor(experimentId) {
5305
- const experiments = await this.store.load();
5306
- const exp = experiments.find((e) => e.id === experimentId);
5307
- if (!exp) throw new ValidationError(`experiment-tracker: experiment "${experimentId}" not found`);
5308
- const parent = exp.parentId ? experiments.find((e) => e.id === exp.parentId) : void 0;
5309
- return improvementVerdict(exp.stats, parent?.stats ?? null, this.thresholds);
5310
- }
5311
- };
5312
- //#endregion
5313
4855
  //#region src/leaderboard.ts
5314
4856
  function mean$1(xs) {
5315
4857
  return xs.length ? xs.reduce((a, b) => a + b, 0) / xs.length : null;
@@ -6228,6 +5770,162 @@ function statusAdvanced(key, progression) {
6228
5770
  };
6229
5771
  }
6230
5772
  //#endregion
5773
+ //#region src/verification-strategy.ts
5774
+ /**
5775
+ * The family registry. `Record` over the union keeps it exhaustive: adding
5776
+ * a member to `VerificationStrategySource` without a profile here fails to
5777
+ * compile. The failure mode travels with the taxonomy so a reader of a
5778
+ * certification can surface it without this package's docs at hand.
5779
+ */
5780
+ const VERIFICATION_STRATEGIES = {
5781
+ compile: {
5782
+ determinism: "deterministic",
5783
+ failureMode: "code that compiles is not code that is correct"
5784
+ },
5785
+ test: {
5786
+ determinism: "deterministic",
5787
+ failureMode: "assumes an answer key; certifies nothing outside suite coverage, and a stubbed integration reports green"
5788
+ },
5789
+ schema: {
5790
+ determinism: "deterministic",
5791
+ failureMode: "shape is not meaning; a well-formed wrong answer passes"
5792
+ },
5793
+ sandbox: {
5794
+ determinism: "deterministic",
5795
+ failureMode: "an exit code compresses the run to one bit; a faked success exits 0"
5796
+ },
5797
+ judge: {
5798
+ determinism: "probabilistic",
5799
+ failureMode: "drifts across model versions and is Goodhart-gameable by the graded policy"
5800
+ },
5801
+ composite: {
5802
+ determinism: "inherited",
5803
+ failureMode: "scalar collapse: the blend hides which member carried the score"
5804
+ },
5805
+ "proof-kernel": {
5806
+ determinism: "deterministic",
5807
+ failureMode: "the formalization gap: the kernel certifies the formal statement, never that it matches the informal claim"
5808
+ },
5809
+ invariant: {
5810
+ determinism: "deterministic",
5811
+ failureMode: "weak invariants pass everything; a set uncalibrated by seeded bugs is a rubber stamp"
5812
+ },
5813
+ replication: {
5814
+ determinism: "deterministic",
5815
+ failureMode: "re-runs the method, so it catches drift and nondeterminism, never an error the method itself carries"
5816
+ },
5817
+ agreement: {
5818
+ determinism: "probabilistic",
5819
+ failureMode: "the shared blind spot: derivers with common corpora or priors agree for the same wrong reason"
5820
+ }
5821
+ };
5822
+ /** Every family member, derived from the registry so it cannot drift. */
5823
+ const VERIFICATION_STRATEGY_SOURCES = Object.keys(VERIFICATION_STRATEGIES);
5824
+ //#endregion
5825
+ //#region src/equivalence-check.ts
5826
+ /** A refused equivalence check. `code` names the exact refusal for programmatic handling. */
5827
+ var EquivalenceProtocolError = class extends Error {
5828
+ code;
5829
+ constructor(code, message) {
5830
+ super(message);
5831
+ this.name = "EquivalenceProtocolError";
5832
+ this.code = code;
5833
+ }
5834
+ };
5835
+ /**
5836
+ * Validate and freeze a two-arm blind equivalence check spec.
5837
+ *
5838
+ * The literal types already refuse a wide design at compile time; the
5839
+ * runtime checks hold the same line for untyped callers. There is no
5840
+ * escape hatch: `arms: 3` or `blind: false` throws, never downgrades.
5841
+ */
5842
+ function defineEquivalenceCheck(spec) {
5843
+ if (spec.arms !== 2) throw new EquivalenceProtocolError("arm-count", `equivalence check requires exactly 2 arms, received ${String(spec.arms)}`);
5844
+ if (spec.blind !== true) throw new EquivalenceProtocolError("not-blind", "equivalence check requires blind: true — a non-blind run is not a weaker check, it is no check");
5845
+ if (typeof spec.artifact !== "string" || spec.artifact.trim() === "") throw new EquivalenceProtocolError("empty-field", "spec.artifact must identify the claim under verification");
5846
+ if (!VERIFICATION_STRATEGY_SOURCES.includes(spec.source)) throw new EquivalenceProtocolError("unknown-source", `spec.source '${String(spec.source)}' is not a verification-strategy member`);
5847
+ return Object.freeze({ spec: Object.freeze({ ...spec }) });
5848
+ }
5849
+ /**
5850
+ * Assemble the equivalence record, refusing every asymmetry.
5851
+ *
5852
+ * Refusals (each throws `EquivalenceProtocolError`):
5853
+ * - an arm whose `blindness.toOtherArms` is false — it saw the other
5854
+ * statement, so nothing was independently derived;
5855
+ * - an arm whose `blindness.toOutcome` is false — it could steer its
5856
+ * statement toward (or away from) the known result;
5857
+ * - duplicate arm ids, empty statements, empty derivations;
5858
+ * - an obligation whose fields contradict its status (see
5859
+ * `EquivalenceObligation`).
5860
+ */
5861
+ function buildEquivalenceRecord(definition, arms, obligation) {
5862
+ assertArms(arms);
5863
+ assertObligation(obligation);
5864
+ return Object.freeze({
5865
+ spec: definition.spec,
5866
+ arms: Object.freeze([Object.freeze({ ...arms[0] }), Object.freeze({ ...arms[1] })]),
5867
+ obligation: Object.freeze({ ...obligation })
5868
+ });
5869
+ }
5870
+ /**
5871
+ * Discharge the obligation through an injected checker and return the
5872
+ * record.
5873
+ *
5874
+ * Order matters: every arm refusal fires BEFORE the checker runs — an
5875
+ * invalid check must not spend. A checker whose `strategy` differs from
5876
+ * `spec.source` is refused for the same reason: a judge cannot silently
5877
+ * discharge a proof-kernel obligation.
5878
+ *
5879
+ * A checker failure (`succeeded: false`) is not thrown: it becomes an
5880
+ * `'unresolved'` obligation carrying the full error text, which is the
5881
+ * honest record of an undischarged check.
5882
+ */
5883
+ async function runEquivalenceCheck(definition, arms, checker) {
5884
+ assertArms(arms);
5885
+ if (checker.strategy !== definition.spec.source) throw new EquivalenceProtocolError("checker-strategy-mismatch", `spec.source is '${definition.spec.source}' but the bound checker declares '${checker.strategy}'`);
5886
+ const outcome = await checker.check({
5887
+ artifact: definition.spec.artifact,
5888
+ statements: [arms[0].statement, arms[1].statement]
5889
+ });
5890
+ if (!outcome.succeeded) return buildEquivalenceRecord(definition, arms, {
5891
+ status: "unresolved",
5892
+ unresolvedReason: outcome.error,
5893
+ checker: checker.identity
5894
+ });
5895
+ const { status, separatingWitness, evidenceDigest } = outcome.value;
5896
+ return buildEquivalenceRecord(definition, arms, {
5897
+ status,
5898
+ ...separatingWitness === void 0 ? {} : { separatingWitness },
5899
+ checker: checker.identity,
5900
+ evidenceDigest
5901
+ });
5902
+ }
5903
+ function assertArms(arms) {
5904
+ if (arms.length !== 2) throw new EquivalenceProtocolError("arm-count", `equivalence record requires exactly 2 arms, received ${arms.length}`);
5905
+ if (arms[0].armId === arms[1].armId) throw new EquivalenceProtocolError("duplicate-arm-id", `both arms declare armId '${arms[0].armId}' — two labels for one derivation is one arm`);
5906
+ for (const arm of arms) {
5907
+ if (arm.statement.trim() === "") throw new EquivalenceProtocolError("empty-field", `arm '${arm.armId}' committed an empty statement`);
5908
+ if (arm.derivedFrom.trim() === "") throw new EquivalenceProtocolError("empty-field", `arm '${arm.armId}' declares no derivation provenance`);
5909
+ if (arm.blindness.toOtherArms !== true) throw new EquivalenceProtocolError("arm-saw-other", `arm '${arm.armId}' saw another arm's statement — the derivation is not independent and the check is invalid`);
5910
+ if (arm.blindness.toOutcome !== true) throw new EquivalenceProtocolError("arm-saw-outcome", `arm '${arm.armId}' saw the outcome before committing — the check is invalid`);
5911
+ }
5912
+ }
5913
+ function assertObligation(obligation) {
5914
+ const { status, separatingWitness, unresolvedReason, evidenceDigest } = obligation;
5915
+ if (status === "refuted-with-separating-witness") {
5916
+ if (typeof separatingWitness !== "string" || separatingWitness.trim() === "") throw new EquivalenceProtocolError("witness-missing", "a refuted equivalence must carry the separating witness — a refutation without one is an assertion");
5917
+ if (typeof evidenceDigest !== "string" || evidenceDigest.trim() === "") throw new EquivalenceProtocolError("evidence-missing", "a refuted equivalence must carry its evidence digest");
5918
+ return;
5919
+ }
5920
+ if (status === "proved") {
5921
+ if (separatingWitness !== void 0) throw new EquivalenceProtocolError("witness-on-proved", "a proved equivalence cannot carry a separating witness — the two claims contradict");
5922
+ if (typeof evidenceDigest !== "string" || evidenceDigest.trim() === "") throw new EquivalenceProtocolError("evidence-missing", "a proved equivalence must carry its evidence digest");
5923
+ return;
5924
+ }
5925
+ if (separatingWitness !== void 0) throw new EquivalenceProtocolError("witness-on-unresolved", "an unresolved obligation cannot carry a separating witness — a witness in hand is a refutation");
5926
+ if (typeof unresolvedReason !== "string" || unresolvedReason.trim() === "") throw new EquivalenceProtocolError("reason-missing", "an unresolved obligation must say why — 'unresolved' with no reason erases the diagnostic");
5927
+ }
5928
+ //#endregion
6231
5929
  //#region src/ui-finding.ts
6232
5930
  /** Frozen tuple of lenses for validation + iteration. */
6233
5931
  const UI_LENSES = [
@@ -7428,126 +7126,6 @@ async function promptBisect(options) {
7428
7126
  };
7429
7127
  }
7430
7128
  //#endregion
7431
- //#region src/counterfactual.ts
7432
- /**
7433
- * Counterfactual replay — "what would have happened if we'd changed
7434
- * exactly one thing at turn N?"
7435
- *
7436
- * The framework does NOT drive the agent — it sets up the replay
7437
- * context (prior spans, prior state, mutation spec) and records the
7438
- * resulting divergence. Consumers supply an `executeFrom(ctx)` callback
7439
- * that runs their agent starting from turn N with the mutation applied.
7440
- *
7441
- * Counterfactual runs are recorded as a new Run with `layer='meta'` and
7442
- * `parentRunId = originalRunId`, so downstream diff + correlation
7443
- * pipelines see them natively.
7444
- */
7445
- async function runCounterfactual(store, originalRunId, mutation, runner) {
7446
- const originalRun = await store.getRun(originalRunId);
7447
- if (!originalRun) throw new NotFoundError(`counterfactual: run ${originalRunId} not found`);
7448
- const trajectory = await buildTrajectory(store, originalRunId);
7449
- if (mutation.at < 0 || mutation.at >= trajectory.steps.length) throw new ValidationError(`counterfactual: mutation.at=${mutation.at} out of range [0, ${trajectory.steps.length})`);
7450
- const targetStep = trajectory.steps[mutation.at];
7451
- const mutatedStep = applyMutation(targetStep, mutation);
7452
- const cfEmitter = new TraceEmitter(store);
7453
- await cfEmitter.startRun({
7454
- scenarioId: originalRun.scenarioId,
7455
- variantId: originalRun.variantId ? `${originalRun.variantId}+cf:${mutation.kind}@${mutation.at}` : `cf:${mutation.kind}@${mutation.at}`,
7456
- projectId: originalRun.projectId,
7457
- parentRunId: originalRunId,
7458
- layer: "meta",
7459
- tags: {
7460
- counterfactual: "true",
7461
- mutationKind: mutation.kind,
7462
- mutationAt: String(mutation.at)
7463
- }
7464
- });
7465
- await runner.executeFrom({
7466
- originalRunId,
7467
- originalTrajectory: trajectory,
7468
- prefix: trajectory.steps.slice(0, mutation.at),
7469
- mutation,
7470
- mutatedStep
7471
- }, cfEmitter);
7472
- const counterfactual = await store.getRun(cfEmitter.runId);
7473
- const delta = {
7474
- originalOutcomeScore: originalRun.outcome?.score ?? null,
7475
- counterfactualOutcomeScore: counterfactual?.outcome?.score ?? null,
7476
- deltaScore: originalRun.outcome?.score !== void 0 && counterfactual?.outcome?.score !== void 0 ? counterfactual.outcome.score - originalRun.outcome.score : null
7477
- };
7478
- return {
7479
- counterfactualRunId: cfEmitter.runId,
7480
- originalRunId,
7481
- mutation,
7482
- delta
7483
- };
7484
- }
7485
- function applyMutation(step, mutation) {
7486
- if (mutation.kind === "swap-model" && step.span.kind === "llm") {
7487
- const llm = step.span;
7488
- return {
7489
- ...step,
7490
- span: {
7491
- ...llm,
7492
- model: mutation.newModel
7493
- }
7494
- };
7495
- }
7496
- if (mutation.kind === "swap-tool-result" && step.span.kind === "tool") {
7497
- const tool = step.span;
7498
- return {
7499
- ...step,
7500
- span: {
7501
- ...tool,
7502
- result: mutation.newResult
7503
- }
7504
- };
7505
- }
7506
- if (mutation.kind === "inject-system-message" && step.span.kind === "llm") {
7507
- const llm = step.span;
7508
- return {
7509
- ...step,
7510
- span: {
7511
- ...llm,
7512
- messages: [{
7513
- role: "system",
7514
- content: mutation.content
7515
- }, ...llm.messages]
7516
- }
7517
- };
7518
- }
7519
- if (mutation.kind === "custom") return mutation.apply(step);
7520
- return step;
7521
- }
7522
- /**
7523
- * Aggregate a batch of counterfactuals into a simple attribution table:
7524
- * which mutation kinds move outcomes most? (Useful when you run a grid
7525
- * over the same trajectory — swap-model at every llm span, swap-tool
7526
- * at every tool span — and want a ranked summary.)
7527
- */
7528
- function attributeCounterfactuals(results) {
7529
- const grouped = /* @__PURE__ */ new Map();
7530
- for (const r of results) {
7531
- const arr = grouped.get(r.mutation.kind) ?? [];
7532
- arr.push(r);
7533
- grouped.set(r.mutation.kind, arr);
7534
- }
7535
- const out = [];
7536
- for (const [kind, items] of grouped) {
7537
- const deltas = items.map((i) => i.delta.deltaScore).filter((d) => typeof d === "number");
7538
- if (deltas.length === 0) continue;
7539
- const meanAbs = deltas.reduce((a, b) => a + Math.abs(b), 0) / deltas.length;
7540
- const meanSigned = deltas.reduce((a, b) => a + b, 0) / deltas.length;
7541
- out.push({
7542
- mutationKind: kind,
7543
- n: deltas.length,
7544
- meanAbsDelta: meanAbs,
7545
- meanSignedDelta: meanSigned
7546
- });
7547
- }
7548
- return out.sort((a, b) => b.meanAbsDelta - a.meanAbsDelta);
7549
- }
7550
- //#endregion
7551
7129
  //#region src/cross-trace-diff.ts
7552
7130
  async function crossTraceDiff(store, runA, runB, options = {}) {
7553
7131
  const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)]);
@@ -11402,14 +10980,20 @@ function attachCostToReport(report, ledger) {
11402
10980
  /**
11403
10981
  * Tier presets — plain data, swap or spread freely.
11404
10982
  *
11405
- * `economy` names ids that a live router probe answered from the provider
11406
- * their name implies, so the judge trio spans three provider families
11407
- * (deepseek / zhipu / google) both as configured AND as served. A preset is
11408
- * only as good as its last probe: an id can go dead or start resolving
11409
- * elsewhere at any time, so gate a real run on `assertModelsServed({ probe:
11410
- * true })` and assert `assertServedModel` per call rather than trusting this
11411
- * list. `assertCrossFamily` over these ids proves configuration; only
11412
- * `assertCrossFamilyServed` over the echoed ids proves the run.
10983
+ * A preset names REQUESTED ids, and a requested id is NOT a guarantee of
10984
+ * family, of provider, or of liveness. A routing gateway can accept any id
10985
+ * below and answer from a different model on HTTP 200, with only the response
10986
+ * body's `model` field betraying the swap. The served assertion is the
10987
+ * guarantee: gate a run on `assertModelsServed({ probe: true })` and assert
10988
+ * `assertServedModel` per call. `assertCrossFamily` over these ids proves the
10989
+ * configuration is diverse; only `assertCrossFamilyServed` over the ids that
10990
+ * ANSWERED proves the run was.
10991
+ *
10992
+ * `economy` names ids a live router probe answered from the provider their
10993
+ * name implies, so the judge trio spanned three provider families (deepseek /
10994
+ * zhipu / google) as configured and, at that probe, as served. A preset is
10995
+ * only as good as its last probe: ids go dead and start resolving elsewhere
10996
+ * without notice, so re-probe rather than trusting this list.
11413
10997
  *
11414
10998
  * `frontier` is deliberately EMPTY: entitled frontier ids vary per router
11415
10999
  * account, and a hardcoded claude/gpt-5 id 401s on keys that lack it. Supply
@@ -12400,6 +11984,6 @@ function assertProductBenchmarkRun(runDir) {
12400
11984
  return report;
12401
11985
  }
12402
11986
  //#endregion
12403
- export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CONTROL_INTEGRITY_ANALYST, CallbackResearcher, CaptureIntegrityError, ConfigError, ControlIntegrityAnalyst, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExactAnalystRunExecutionError, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_BRIEFS, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LimitExceededError, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelSubstitutionError, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_INTEGRITY_SCHEMA, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, ServedCrossFamilyError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYSIS_LIMITS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TOOL_NAMESPACE, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, ToolTraceMissingError, TraceAnalysisLimitError, TraceAnalysisStoreContractError, TraceAnalysisValidationError, TraceContractBuilder, TraceEmitter, TraceFileMalformedError, TraceFileMissingError, TraceFileTooLargeError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, WORKER_DRIVER_DOCTRINE, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunIntegrity, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertExactRegistryRunOpts, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalysisToolDescriptors, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkServedModel, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDefaultReviewer, createDspyRlmTraceEngine, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalyst, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, defineTraceAnalyst, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, emitControlIntegrityFindings, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isRuntimeSupervisorRunDir, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeModelId, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpTextToTraceAnalysisStore, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, readRuntimeSupervisorRun, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resetDeprecationWarnings, resolveModelPricing, resolveSeat, resolveTraceAnalystLimits, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runTraceAnalyst, runsForScenario, runtimeSupervisorRunReader, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, servedModelAcceptable, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, warnDeprecatedOnce, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
11987
+ export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CONTROL_INTEGRITY_ANALYST, CallbackResearcher, CaptureIntegrityError, ConfigError, ControlIntegrityAnalyst, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, EvalTraceStore, ExactAnalystRunExecutionError, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_BRIEFS, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LimitExceededError, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelSubstitutionError, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PROBE_MAX_TOKENS, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_INTEGRITY_SCHEMA, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, ServedCrossFamilyError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYSIS_LIMITS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TOOL_NAMESPACE, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, ToolTraceMissingError, TraceAnalysisLimitError, TraceAnalysisStoreContractError, TraceAnalysisValidationError, TraceContractBuilder, TraceEmitter, TraceFileMalformedError, TraceFileMissingError, TraceFileTooLargeError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, WORKER_DRIVER_DOCTRINE, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunIntegrity, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertExactRegistryRunOpts, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildEquivalenceRecord, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalysisToolDescriptors, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkServedModel, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDefaultReviewer, createDspyRlmTraceEngine, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalyst, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, defineEquivalenceCheck, defineTraceAnalyst, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, emitControlIntegrityFindings, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isRuntimeSupervisorRunDir, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeModelId, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpTextToTraceAnalysisStore, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, readRuntimeSupervisorRun, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resetDeprecationWarnings, resolveModelPricing, resolveSeat, resolveTraceAnalystLimits, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEquivalenceCheck, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runTraceAnalyst, runsForScenario, runtimeSupervisorRunReader, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, servedModelAcceptable, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, warnDeprecatedOnce, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
12404
11988
 
12405
11989
  //# sourceMappingURL=index.js.map