@tangle-network/agent-eval 0.144.5 → 0.144.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (257) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
  6. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
  7. package/dist/analyst/index.d.ts +471 -89
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +24 -5
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  12. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  13. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  14. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  15. package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
  16. package/dist/benchmark-command-BCafwNrf.js.map +1 -0
  17. package/dist/benchmarks/index.d.ts +1 -1
  18. package/dist/benchmarks/index.js +1 -1
  19. package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
  20. package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
  21. package/dist/campaign/index.d.ts +8 -6
  22. package/dist/campaign/index.js +5 -3
  23. package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
  24. package/dist/campaign-Tdy3h62h.js.map +1 -0
  25. package/dist/cli.js +2 -2
  26. package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
  27. package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  28. package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
  29. package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  30. package/dist/contract/index.d.ts +11 -10
  31. package/dist/contract/index.d.ts.map +1 -1
  32. package/dist/contract/index.js +7 -6
  33. package/dist/contract/index.js.map +1 -1
  34. package/dist/control.d.ts +2 -2
  35. package/dist/control.js +1 -1
  36. package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
  37. package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
  38. package/dist/counterfactual-CWPTrMH7.js +126 -0
  39. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  40. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  41. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  42. package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
  43. package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
  44. package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  45. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  46. package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
  47. package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  48. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  49. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  50. package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
  51. package/dist/engine-nB64f48I.d.ts.map +1 -0
  52. package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
  53. package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
  54. package/dist/errors-D-LKuDhb.js.map +1 -1
  55. package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
  56. package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  57. package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  58. package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  59. package/dist/exec-BLtYZdWo.js +49 -0
  60. package/dist/exec-BLtYZdWo.js.map +1 -0
  61. package/dist/experiment/index.d.ts +802 -0
  62. package/dist/experiment/index.d.ts.map +1 -0
  63. package/dist/experiment/index.js +1108 -0
  64. package/dist/experiment/index.js.map +1 -0
  65. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  66. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  67. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  68. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  69. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
  70. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  71. package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
  72. package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  73. package/dist/fuzz.d.ts +2 -2
  74. package/dist/hosted/index.d.ts +3 -3
  75. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  76. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  77. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  78. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  79. package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
  80. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  81. package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
  82. package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
  83. package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
  84. package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  85. package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
  86. package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  87. package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
  88. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  89. package/dist/index.d.ts +252 -451
  90. package/dist/index.d.ts.map +1 -1
  91. package/dist/index.js +271 -751
  92. package/dist/index.js.map +1 -1
  93. package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
  94. package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  95. package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
  96. package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  97. package/dist/integrity-MLzHOfV9.js +141 -0
  98. package/dist/integrity-MLzHOfV9.js.map +1 -0
  99. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  100. package/dist/ledger-core/index.d.ts +1 -1
  101. package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
  102. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  103. package/dist/matrix/index.d.ts +2 -2
  104. package/dist/meta-eval/index.d.ts +2 -2
  105. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  106. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  107. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  108. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  109. package/dist/multishot/index.d.ts +3 -3
  110. package/dist/openapi.json +1 -1
  111. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  112. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  113. package/dist/pipelines/index.d.ts +2 -1
  114. package/dist/pipelines/index.d.ts.map +1 -1
  115. package/dist/pre-registration-DakwTRXk.js +96 -0
  116. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  117. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  118. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  119. package/dist/prime-protocol-BfSalTfR.js +453 -0
  120. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  121. package/dist/profile-cell.d.ts +1 -1
  122. package/dist/profile-cell.js +242 -1
  123. package/dist/profile-cell.js.map +1 -0
  124. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  125. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  126. package/dist/promotion-policy-CrLrmys8.js +682 -0
  127. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  128. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  129. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  130. package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
  131. package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  132. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  133. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  134. package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
  135. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  136. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  137. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  138. package/dist/reporting.d.ts +4 -4
  139. package/dist/reporting.js +2 -2
  140. package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
  141. package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  142. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  143. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  144. package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
  145. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  146. package/dist/rl.d.ts +19 -9
  147. package/dist/rl.d.ts.map +1 -1
  148. package/dist/rl.js +16 -7
  149. package/dist/rl.js.map +1 -1
  150. package/dist/rollout/index.d.ts +2 -2
  151. package/dist/rollout/index.js +2 -2
  152. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  153. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  154. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
  155. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
  156. package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
  157. package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  158. package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
  159. package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
  160. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  161. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  162. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  163. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  164. package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  165. package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  166. package/dist/sequential-D-BLJBKU.js +299 -0
  167. package/dist/sequential-D-BLJBKU.js.map +1 -0
  168. package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
  169. package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
  170. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  171. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  172. package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
  173. package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  174. package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
  175. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  176. package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
  177. package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
  178. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  179. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  180. package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  181. package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  182. package/dist/steps-BArUxhna.d.ts +51 -0
  183. package/dist/steps-BArUxhna.d.ts.map +1 -0
  184. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  185. package/dist/store-DNe_Uv1Q.js.map +1 -0
  186. package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
  187. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  188. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  189. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  190. package/dist/supervisor-run/index.d.ts +2 -2
  191. package/dist/supervisor-run/index.js +1 -1
  192. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  193. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  194. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  195. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  196. package/dist/trace-repair/index.d.ts +2102 -0
  197. package/dist/trace-repair/index.d.ts.map +1 -0
  198. package/dist/trace-repair/index.js +3878 -0
  199. package/dist/trace-repair/index.js.map +1 -0
  200. package/dist/traces.d.ts +6 -6
  201. package/dist/traces.js +3 -2
  202. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  203. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  204. package/dist/trajectory-replay/index.d.ts +781 -0
  205. package/dist/trajectory-replay/index.d.ts.map +1 -0
  206. package/dist/trajectory-replay/index.js +2103 -0
  207. package/dist/trajectory-replay/index.js.map +1 -0
  208. package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
  209. package/dist/types-D216SgwM.d.ts.map +1 -0
  210. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  211. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  212. package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
  213. package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  214. package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
  215. package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  216. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  217. package/dist/verdict-DExhxfgR.d.ts +201 -0
  218. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  219. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  220. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  221. package/dist/wire/index.d.ts +3 -3
  222. package/dist/wire/index.js +1 -1
  223. package/docs/building-doctrine.md +15 -0
  224. package/docs/charter.md +112 -0
  225. package/docs/experiment.md +104 -0
  226. package/docs/prime-analyst.md +1 -0
  227. package/docs/trace-analysis.md +26 -0
  228. package/docs/trace-repair-admission.md +194 -0
  229. package/docs/trace-repair-analyst-arms.md +121 -0
  230. package/docs/trace-repair-continuation.md +107 -0
  231. package/docs/trace-repair-grader.md +163 -0
  232. package/docs/trajectory-replay.md +110 -0
  233. package/docs/verification-strategies.md +103 -0
  234. package/package.json +21 -4
  235. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  236. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  237. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  238. package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
  239. package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
  240. package/dist/campaign-ClpnD7Ug.js.map +0 -1
  241. package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
  242. package/dist/index-CpxZSlB7.d.ts.map +0 -1
  243. package/dist/index-CsuAo2-J.d.ts.map +0 -1
  244. package/dist/integrity-fdt8XPAv.js.map +0 -1
  245. package/dist/llm-client-D3EoChAU.js.map +0 -1
  246. package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
  247. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  248. package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
  249. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  250. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  251. package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
  252. package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
  253. package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
  254. package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
  255. package/dist/types-BjMFz88h.d.ts.map +0 -1
  256. package/dist/verdict-Dps8_okt.d.ts +0 -37
  257. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -1,12 +1,14 @@
1
1
  import { c as ValidationError, i as JudgeError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
2
2
  import { i as CostLedger, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
3
- import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, m as stripFencedJson, u as costReceiptFromLlmError } from "./llm-client-D3EoChAU.js";
3
+ import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, m as stripFencedJson, u as costReceiptFromLlmError } from "./llm-client-DzvMUsS_.js";
4
4
  import { a as clamp01, o as combineAbortSignals, t as assertProposalFindings } from "./proposal-findings-2GIUo1et.js";
5
- import { A as removeCredentialEnvironment, D as isCandidateText, E as assertNoCredentialValues, M as resolveExternalOptimizerProcessLimits, N as safePathComponent, O as isExternalTextCandidate, T as assertJsonValue, d as closeExternalOptimizerResources, f as runWithCleanup, j as resolveExternalOptimizerCallbackLimits, k as isRecord, l as startExternalOptimizerCallback, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock, u as runExternalOptimizerProcess, v as canonicalJson, w as assertExternalOptimizerModelBudget, y as contentHash } from "./single-run-lock-D5iN0Xzb.js";
5
+ import { C as isExternalTextCandidate, D as resolveExternalOptimizerProcessLimits, E as resolveExternalOptimizerCallbackLimits, O as safePathComponent, S as isCandidateText, T as removeCredentialEnvironment, b as assertJsonValue, d as closeExternalOptimizerResources, f as runWithCleanup, l as startExternalOptimizerCallback, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock, u as runExternalOptimizerProcess, w as isRecord, x as assertNoCredentialValues, y as assertExternalOptimizerModelBudget } from "./single-run-lock-BMQEv1wG.js";
6
+ import { n as canonicalJson, r as contentHash } from "./verdict-cache-BCcOh0kF.js";
6
7
  import { p as mapConcurrent } from "./ledger-core-DXZIqu17.js";
7
- import { E as pairedBootstrap, H as weightedComposite, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, T as pairedBinaryScale, d as confidenceInterval, j as pairedRiskDifferenceExact } from "./statistics-ByxzSiOM.js";
8
+ import { E as pairedBootstrap, H as weightedComposite, d as confidenceInterval } from "./statistics-ByxzSiOM.js";
9
+ import { c as heldoutSignificance, l as pairHoldout, s as dimensionRegressions } from "./promotion-policy-CrLrmys8.js";
8
10
  import { t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
9
- import { a as campaignCellCostProvenance, o as campaignCellExecutionEvidence, t as detectRewardHacking, u as projectCampaignCellQuality } from "./reward-hacking-CyuzxKly.js";
11
+ import { a as campaignCellCostProvenance, o as campaignCellExecutionEvidence, t as detectRewardHacking, u as projectCampaignCellQuality } from "./reward-hacking-B2tPL9a3.js";
10
12
  import { z } from "zod";
11
13
  import { writeFileSync } from "node:fs";
12
14
  import { basename, isAbsolute, join } from "node:path";
@@ -153,243 +155,6 @@ function assertBackendReport(report, opts) {
153
155
  return report;
154
156
  }
155
157
  //#endregion
156
- //#region src/paired-delta-test.ts
157
- /** Smallest all-positive sample that can clear a one-sided exact sign test. */
158
- function minimumPairsForPairedDeltaTest(confidence = .95) {
159
- if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
160
- const oneSidedAlpha = (1 - confidence) / 2;
161
- return Math.ceil(Math.log2(1 / oneSidedAlpha));
162
- }
163
- /**
164
- * Tests whether a paired candidate-minus-baseline delta clears a threshold.
165
- *
166
- * At 20 or more pairs, the percentile bootstrap lower bound carries the
167
- * decision. Below that point the interval is descriptive only, so the function
168
- * switches to a pre-registered one-sided exact sign test. The exact path is
169
- * deliberately conservative: it requires both a point estimate above the
170
- * threshold and enough consistently positive paired differences.
171
- *
172
- * ## A zero-width interval is never significant
173
- *
174
- * When every paired delta is identical the resample distribution is a point
175
- * mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n
176
- * identical deltas of g. Neither says the effect is certain — both say the
177
- * sample carries no information about how far the estimate could be wrong, and
178
- * `low > threshold` then answers on the point estimate alone. It fails in both
179
- * directions: `[0, 0]` clears every NEGATIVE threshold, which is how a
180
- * tie-dominated pass/fail comparison laundered a regression into a
181
- * noninferiority pass, and `[g, g]` clears every threshold below g with no
182
- * spread behind it. Under a bounded asymmetric null whose true mean paired
183
- * delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —
184
- * every sample that misses the drop is exactly that shape, and deciding on
185
- * `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.
186
- *
187
- * So `indeterminate` is reported and `significant` is false whenever the
188
- * interval has zero width, on BOTH paths: at small n the exact sign test is a
189
- * test of the MEDIAN and a zero-spread sample is precisely where it stops
190
- * saying anything about the mean the caller is thresholding.
191
- *
192
- * `threshold` may be negative — that is a noninferiority margin, and it is the
193
- * regime the zero-width hole is worst in. For a two-point (pass/fail) outcome
194
- * the percentile bootstrap is not a valid interval at a nonzero margin at all;
195
- * use {@link decidePairedPromotion}, which routes those to Tango's score
196
- * interval, rather than thresholding this function's bootstrap directly.
197
- */
198
- function pairedDeltaTest(before, after, options = {}) {
199
- const threshold = options.threshold ?? 0;
200
- if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
201
- const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
202
- const requestedMinimum = options.minPairs ?? exactMinimum;
203
- if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
204
- const minimumPairs = Math.max(requestedMinimum, exactMinimum);
205
- const bootstrap = pairedBootstrap(before, after, options);
206
- const sufficient = bootstrap.n >= minimumPairs;
207
- const indeterminate = bootstrap.n > 0 && (!Number.isFinite(bootstrap.low) || !Number.isFinite(bootstrap.high) || bootstrap.low === bootstrap.high);
208
- if (bootstrap.gateEligible) return {
209
- bootstrap,
210
- method: "bootstrap-ci",
211
- pValue: null,
212
- minimumPairs,
213
- sufficient,
214
- indeterminate,
215
- significant: sufficient && !indeterminate && bootstrap.low > threshold
216
- };
217
- const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
218
- const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
219
- return {
220
- bootstrap,
221
- method: "exact-sign",
222
- pValue: exact.pValue,
223
- minimumPairs,
224
- sufficient,
225
- indeterminate,
226
- significant: sufficient && !indeterminate && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
227
- };
228
- }
229
- //#endregion
230
- //#region src/paired-promotion-decision.ts
231
- /**
232
- * @module
233
- * ONE rule for "does this paired interval clear a promotion threshold".
234
- *
235
- * The rule below was derived on `HeldOutGate` (#479) after the same estimator
236
- * bug shipped twice. It then turned out that a SECOND gate — the composable
237
- * `heldOutGate`, plus everything else routed through `heldoutSignificance` —
238
- * still carried the original defect, because the rule had been written into one
239
- * gate's method body rather than into a shared function. Two copies of a
240
- * statistical rule is how a defect survives in one of them, so there is now
241
- * exactly one copy and both gates call it.
242
- *
243
- * Three things the rule does that a bare `pairedBootstrap(...).low > threshold`
244
- * does not:
245
- *
246
- * 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a
247
- * pass/fail eval the paired delta vector is dominated by ties, so the
248
- * bootstrap of the mean is a resample of a lattice with three atoms and its
249
- * percentile interval is not valid at a nonzero margin. The score interval
250
- * (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under
251
- * each hypothesised margin instead of fixing it at the observed value, which
252
- * is the only construction that stays a confidence interval as the margin
253
- * moves off zero — the regime every noninferiority threshold lives in.
254
- * Measured on the composable gate before this change, at a true risk
255
- * difference sitting exactly on the production caller's -0.05 margin and a
256
- * nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.
257
- * 2. **McNemar's exact test holds a VETO at every non-negative threshold.**
258
- * Redundant with the interval by construction and kept anyway, so that
259
- * swapping the estimator for one without that duality cannot silently
260
- * reintroduce "promotes what the exact test refuses". Witness: n = 6, b = 5,
261
- * c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs
262
- * (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE
263
- * threshold is a noninferiority question, which McNemar's test of "no
264
- * difference" is not the right test for, so the veto does not apply there.
265
- * 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it
266
- * cannot tell a gain from a regression and clears every negative threshold.
267
- * Away from zero it fails the opposite way: n identical positive deltas give
268
- * [g, g], which clears threshold 0 on no spread at all. Both are an absence
269
- * of evidence. Measured on the composable gate before this change, under a
270
- * bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %
271
- * false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.
272
- *
273
- * Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that
274
- * picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered
275
- * exact sign test below it), this picks the ESTIMATOR from the outcome's shape.
276
- * Both are needed — an exact sign test applied to a tie-pinned median is still
277
- * blind, and a mean bootstrap CI at n = 6 is still not a valid test.
278
- */
279
- /**
280
- * Which estimator {@link decidePairedPromotion} would use on this data, and the
281
- * shape facts behind it — for callers that must report the shape on a path
282
- * where no interval is computed at all (an early rejection, or zero pairs).
283
- * Cheap: no bootstrap, no interval.
284
- */
285
- function pairedDecisionShape(before, after, statistic = "mean") {
286
- const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after);
287
- if (statistic === "median") return {
288
- statistic: "median_bootstrap",
289
- binaryScale: null,
290
- tieFraction
291
- };
292
- const binaryScale = pairedBinaryScale(before, after);
293
- if (binaryScale !== null) return {
294
- statistic: "paired_risk_difference",
295
- binaryScale,
296
- tieFraction
297
- };
298
- return {
299
- statistic: "mean_bootstrap",
300
- binaryScale: null,
301
- tieFraction
302
- };
303
- }
304
- /**
305
- * Decide whether a paired candidate-minus-baseline delta clears a promotion
306
- * threshold. `before` is the baseline arm, `after` the candidate arm, paired by
307
- * position. Throws on unequal lengths.
308
- */
309
- function decidePairedPromotion(before, after, options = {}) {
310
- if (before.length !== after.length) throw new Error(`decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`);
311
- const threshold = options.threshold ?? 0;
312
- if (!Number.isFinite(threshold)) throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`);
313
- const confidence = options.confidence ?? .95;
314
- const exactMinimum = minimumPairsForPairedDeltaTest(confidence);
315
- const requestedMinimum = options.minPairs ?? exactMinimum;
316
- if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`);
317
- const minimumPairs = Math.max(requestedMinimum, exactMinimum);
318
- const n = before.length;
319
- const sufficient = n >= minimumPairs;
320
- const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic);
321
- let core;
322
- if (binaryScale !== null) {
323
- const unitControl = before.map((v) => v / binaryScale);
324
- const unitTreatment = after.map((v) => v / binaryScale);
325
- const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence);
326
- const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence);
327
- const low = score.lower * binaryScale;
328
- core = {
329
- statistic: "paired_risk_difference",
330
- method: "score-interval",
331
- delta: score.riskDifference * binaryScale,
332
- low,
333
- high: score.upper * binaryScale,
334
- bootstrap: null,
335
- mcnemar: {
336
- b: exact.b,
337
- c: exact.c,
338
- nDiscordant: exact.nDiscordant,
339
- pValue: exact.pValue
340
- },
341
- pValue: null,
342
- clearsThreshold: low > threshold,
343
- label: "success-rate",
344
- methodDetail: ""
345
- };
346
- } else {
347
- const bootstrapStatistic = options.statistic === "median" ? "median" : "mean";
348
- const test = pairedDeltaTest(before, after, {
349
- confidence,
350
- resamples: options.resamples,
351
- statistic: bootstrapStatistic,
352
- seed: options.seed,
353
- threshold,
354
- minPairs: options.minPairs
355
- });
356
- const ci = test.bootstrap;
357
- core = {
358
- statistic: bootstrapStatistic === "mean" ? "mean_bootstrap" : "median_bootstrap",
359
- method: test.method,
360
- delta: bootstrapStatistic === "mean" ? ci.mean : ci.median,
361
- low: ci.low,
362
- high: ci.high,
363
- bootstrap: ci,
364
- mcnemar: null,
365
- pValue: test.pValue,
366
- clearsThreshold: test.significant,
367
- label: bootstrapStatistic,
368
- methodDetail: test.method === "exact-sign" ? ` Below ${test.minimumPairs} pairs the interval is descriptive only; the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.` : ""
369
- };
370
- }
371
- const indeterminate = !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.low === core.high;
372
- const indeterminateCause = !indeterminate ? "" : tieFraction === 1 ? "every paired delta is an exact tie" : core.mcnemar !== null && core.mcnemar.nDiscordant === 0 ? "every pair is concordant (0 discordant pairs)" : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`;
373
- const exactTestVetoes = core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence);
374
- return {
375
- n,
376
- threshold,
377
- confidence,
378
- binaryScale,
379
- tieFraction,
380
- minimumPairs,
381
- sufficient,
382
- indeterminate,
383
- indeterminateCause,
384
- exactTestVetoes,
385
- promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,
386
- ...core
387
- };
388
- }
389
- function fmt(x) {
390
- return x.toFixed(4);
391
- }
392
- //#endregion
393
158
  //#region src/json-recovery.ts
394
159
  /**
395
160
  * Truncation-tolerant JSON recovery — shared by every parser that reads JSON
@@ -4364,190 +4129,6 @@ function mean(xs) {
4364
4129
  if (xs.length === 0) return 0;
4365
4130
  return xs.reduce((s, x) => s + x, 0) / xs.length;
4366
4131
  }
4367
- /**
4368
- * Pair candidate vs baseline holdout observations by FULL cellId. `select`
4369
- * pulls the scalar from a cell's judge reports (composite, or a named
4370
- * dimension); a cell contributes the mean of `select` across its judges. Cells
4371
- * whose scenario is not in `scenarioIds`, or where `select` is undefined for
4372
- * every judge on either side, are skipped on BOTH sides so the arrays stay
4373
- * paired. Throws when the two maps disagree on which holdout cells exist — a
4374
- * load-bearing invariant: the baseline + winner holdout campaigns run the same
4375
- * scenarios with the same seed base, so their cellIds MUST align; a mismatch
4376
- * means a silent pairing bug, not a soft fallback.
4377
- */
4378
- function pairHoldout(candidate, baseline, scenarioIds, select) {
4379
- const cellValue = (byCell, cellId) => {
4380
- const scores = byCell.get(cellId);
4381
- if (!scores) return void 0;
4382
- const vals = [];
4383
- for (const s of Object.values(scores)) {
4384
- if (s.failed === true) throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`);
4385
- const v = select(s);
4386
- if (typeof v === "number" && !Number.isFinite(v)) throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`);
4387
- if (typeof v === "number") vals.push(v);
4388
- }
4389
- if (vals.length === 0) return void 0;
4390
- return vals.reduce((a, b) => a + b, 0) / vals.length;
4391
- };
4392
- const inScope = (cellId) => scenarioIds.has(cellId.split(":")[0] ?? "");
4393
- const candCells = [...candidate.keys()].filter(inScope).sort();
4394
- const baseCells = [...baseline.keys()].filter(inScope).sort();
4395
- if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) throw new Error(`pairHoldout: candidate/baseline holdout cells do not align — candidate=[${candCells.join(",")}] baseline=[${baseCells.join(",")}]. Both holdout campaigns must run the same scenarios with the same seed base.`);
4396
- const before = [];
4397
- const after = [];
4398
- const cellIds = [];
4399
- for (const cellId of candCells) {
4400
- const b = cellValue(baseline, cellId);
4401
- const a = cellValue(candidate, cellId);
4402
- if (b === void 0 && a === void 0) continue;
4403
- if (b === void 0 || a === void 0) throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`);
4404
- before.push(b);
4405
- after.push(a);
4406
- cellIds.push(cellId);
4407
- }
4408
- return {
4409
- before,
4410
- after,
4411
- cellIds
4412
- };
4413
- }
4414
- /**
4415
- * Significance of the held-out composite lift: ship only when the lower bound
4416
- * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
4417
- * 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
4418
- * scale.
4419
- *
4420
- * The decision is delegated whole to {@link decidePairedPromotion}, the one
4421
- * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
4422
- * also calls. That module's header carries the measurements; the short version
4423
- * is three guards a bare `bootstrap.low > threshold` does not have:
4424
- *
4425
- * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
4426
- * only paired-binary construction that stays valid at a nonzero margin;
4427
- * - McNemar's exact test VETOES at any non-negative threshold;
4428
- * - a ZERO-WIDTH interval is refused rather than promoted, in either
4429
- * direction — [0,0] clears every negative threshold and [g,g] clears every
4430
- * threshold below g, and both are an absence of evidence, not a result.
4431
- *
4432
- * Measured on this function before those guards landed, at a nominal 5 %:
4433
- * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
4434
- * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
4435
- * delta is exactly 0.
4436
- *
4437
- * At small n, where the percentile bootstrap is descriptive only, a
4438
- * pre-registered exact sign test still carries the bootstrap path.
4439
- */
4440
- function heldoutSignificance(paired, opts = {}) {
4441
- const deltaThreshold = opts.deltaThreshold ?? 0;
4442
- const confidence = opts.confidence ?? .95;
4443
- const resamples = opts.resamples ?? 2e3;
4444
- const seed = opts.seed ?? 1337;
4445
- const statistic = opts.statistic ?? "mean";
4446
- const decision = decidePairedPromotion(paired.before, paired.after, {
4447
- confidence,
4448
- resamples,
4449
- statistic,
4450
- seed,
4451
- threshold: deltaThreshold,
4452
- minPairs: opts.minProductiveRuns
4453
- });
4454
- const bootstrap = decision.bootstrap ?? pairedBootstrap(paired.before, paired.after, {
4455
- confidence,
4456
- resamples,
4457
- statistic,
4458
- seed
4459
- });
4460
- const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
4461
- confidence,
4462
- resamples,
4463
- statistic: "median",
4464
- seed
4465
- });
4466
- const n = paired.before.length;
4467
- let ties = 0;
4468
- for (let i = 0; i < n; i += 1) {
4469
- const after = paired.after[i] ?? 0;
4470
- const before = paired.before[i] ?? 0;
4471
- if (Math.abs(after - before) < 1e-9) ties += 1;
4472
- }
4473
- const tieFraction = n === 0 ? 0 : ties / n;
4474
- return {
4475
- paired,
4476
- bootstrap,
4477
- medianBootstrap,
4478
- decision,
4479
- decisionStatistic: decision.statistic,
4480
- mcnemar: decision.mcnemar,
4481
- tieFraction,
4482
- n,
4483
- minimumRequired: decision.minimumPairs,
4484
- decisionMethod: decision.method,
4485
- pValue: decision.pValue,
4486
- significant: decision.promote,
4487
- fewRuns: !decision.sufficient
4488
- };
4489
- }
4490
- /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
4491
- * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
4492
- * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
4493
- function detectScale(values) {
4494
- return values.some((v) => Math.abs(v) > 1.5) ? 100 : 1;
4495
- }
4496
- /** Per-critical-dimension regression guard. For each dimension, pair the
4497
- * candidate vs baseline values by full cellId and bootstrap the paired delta;
4498
- * a dimension is "regressed" when the CI lower bound < −tolerance (conservative
4499
- * — blocks if the credible worst case exceeds tolerance, which is the right
4500
- * posture for safety dimensions like `hallucination_free`). When `tolerance`
4501
- * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
4502
- *
4503
- * The interval comes from {@link decidePairedPromotion}, so a pass/fail
4504
- * dimension is judged on Tango's score interval rather than a percentile
4505
- * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
4506
- * is not a valid interval at one. That matters most here because this guard
4507
- * fails OPEN by construction: `tolerance` is positive, so an interval pinned at
4508
- * [0,0] never satisfies `low < −tolerance` and a real regression on a safety
4509
- * dimension would be reported as `regressed: false`. On the median it fails the
4510
- * same way for the same reason — when most pairs tie, which is automatic for a
4511
- * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
4512
- * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
4513
- * restore the pre-0.134 behaviour. */
4514
- function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
4515
- const out = [];
4516
- for (const dim of criticalDimensions) {
4517
- const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
4518
- if (paired.before.length === 0) continue;
4519
- const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
4520
- const bootstrapStatistic = opts.statistic ?? "mean";
4521
- const shared = {
4522
- confidence: opts.confidence ?? .95,
4523
- resamples: opts.resamples ?? 2e3,
4524
- statistic: bootstrapStatistic,
4525
- seed: opts.seed ?? 1337
4526
- };
4527
- const guard = decidePairedPromotion(paired.before, paired.after, shared);
4528
- const regression = decidePairedPromotion(paired.after, paired.before, {
4529
- ...shared,
4530
- threshold: tolerance
4531
- });
4532
- const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared);
4533
- out.push({
4534
- dimension: dim,
4535
- bootstrap,
4536
- bootstrapStatistic,
4537
- ci: {
4538
- low: guard.low,
4539
- high: guard.high
4540
- },
4541
- decisionStatistic: guard.statistic,
4542
- mcnemar: guard.mcnemar,
4543
- indeterminate: guard.indeterminate,
4544
- regressed: bootstrap.low < -tolerance || regression.promote,
4545
- tolerance,
4546
- n: paired.before.length
4547
- });
4548
- }
4549
- return out;
4550
- }
4551
4132
  //#endregion
4552
4133
  //#region src/campaign/gates/default-production-gate.ts
4553
4134
  /**
@@ -4887,231 +4468,6 @@ function heldOutGate(options) {
4887
4468
  };
4888
4469
  }
4889
4470
  //#endregion
4890
- //#region src/campaign/gates/power-preflight.ts
4891
- /** Two-sided z for the common confidence levels; interpolation is overkill here. */
4892
- function zFor(confidence) {
4893
- if (confidence >= .99) return 2.576;
4894
- if (confidence >= .95) return 1.96;
4895
- if (confidence >= .9) return 1.645;
4896
- return 1.282;
4897
- }
4898
- /** Estimate the minimum detectable lift a paired-holdout improvement run can
4899
- * ship at a given budget, from the baseline holdout composites — call it BEFORE
4900
- * spending a search to learn whether the effect you are hunting is even
4901
- * observable at this holdout size and worker variance. */
4902
- function powerPreflight(opts) {
4903
- const composites = opts.baselineComposites.filter((v) => Number.isFinite(v));
4904
- if (composites.length < 3) throw new Error(`powerPreflight: need >= 3 finite baseline composites to estimate variance, got ${composites.length}`);
4905
- const deltaThreshold = opts.deltaThreshold ?? .05;
4906
- const confidence = opts.confidence ?? .95;
4907
- const n = opts.pairedN ?? composites.length;
4908
- if (n < 2) throw new Error(`powerPreflight: pairedN must be >= 2, got ${n}`);
4909
- const mean = composites.reduce((a, b) => a + b, 0) / composites.length;
4910
- const variance = composites.reduce((a, b) => a + (b - mean) * (b - mean), 0) / (composites.length - 1);
4911
- const sd = Math.sqrt(variance);
4912
- const z = zFor(confidence);
4913
- const mde = deltaThreshold + z * Math.SQRT2 * sd / Math.sqrt(n);
4914
- const scaleAssumed = composites.every((v) => v >= -.001 && v <= 1.5);
4915
- const headroom = Math.max(0, 1 - mean);
4916
- const underpowered = scaleAssumed && mde > headroom;
4917
- const sharedChannelCaveat = opts.sharedScorerChannel ? "Holdout and gate share one scoring channel: raising n/reps reduces only idiosyncratic noise — systematic judge bias remains and this MDE is a lower bound. Full debiasing needs an independent second scoring channel (different judge/benchmark family)." : void 0;
4918
- const recommendation = underpowered ? `UNDERPOWERED: minimum detectable lift ${mde.toFixed(3)} exceeds the ${headroom.toFixed(3)} headroom above the baseline (${mean.toFixed(3)}) — no achievable effect can ship at this budget. Raise paired n (scenarios x reps) to ~${Math.ceil((z * Math.SQRT2 * sd / Math.max(headroom - deltaThreshold, .01)) ** 2)} or reduce worker variance before searching.` : `Minimum detectable lift at n=${n}: ${mde.toFixed(3)} (baseline sd ${sd.toFixed(3)}). Effects smaller than this cannot clear the gate; budget the search for effects you believe exceed it.`;
4919
- return {
4920
- n,
4921
- sd,
4922
- mde,
4923
- baselineMean: mean,
4924
- headroom,
4925
- underpowered,
4926
- scaleAssumed,
4927
- deltaThreshold,
4928
- confidence,
4929
- ...sharedChannelCaveat ? { sharedChannelCaveat } : {},
4930
- recommendation: sharedChannelCaveat ? `${recommendation} ${sharedChannelCaveat}` : recommendation
4931
- };
4932
- }
4933
- //#endregion
4934
- //#region src/campaign/gates/promotion-policy.ts
4935
- /**
4936
- * Promotion policy over the evidence VECTOR — the substrate's answer to "never
4937
- * collapse the multi-objective promotion decision into one scalar." A
4938
- * `defaultProductionGate` is one opinionated composition; this module factors
4939
- * the decision into two reusable pieces so MANY policies can compete over the
4940
- * SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):
4941
- *
4942
- * buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus
4943
- * PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy
4944
- * paretoPolicy(ev) // the default strategy
4945
- * paretoSignificanceGate(options): Gate // bus + policy as a Gate
4946
- *
4947
- * The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a
4948
- * potential gain source AND a safety floor (unlike `defaultProductionGate`,
4949
- * where only `composite` can win and `criticalDimensions` are pure floors). A
4950
- * candidate ships iff it weakly DOMINATES the baseline at the confidence level —
4951
- * no objective credibly worse (CI floor breach) AND at least one objective
4952
- * credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work
4953
- * (NOT folded into hold: "gather more reps" and "reject" are different actions).
4954
- *
4955
- * Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate
4956
- * per-side cost, no per-cell observation vector to bootstrap. Treat them as hard
4957
- * constraints (compose with a budget gate via `composeGate`), not faked CIs.
4958
- */
4959
- /**
4960
- * The Evidence Bus. For each objective, pair candidate vs baseline by full
4961
- * cellId and bootstrap a CI on the good-direction paired delta. Reuses the
4962
- * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
4963
- * a single source of truth governs pairing granularity + scale handling.
4964
- */
4965
- function buildEvidenceVector(ctx, objectives, opts = {}) {
4966
- if (objectives.length === 0) throw new Error("buildEvidenceVector: at least 1 objective required");
4967
- const confidence = opts.confidence ?? .95;
4968
- const resamples = opts.resamples ?? 2e3;
4969
- const seed = opts.seed ?? 1337;
4970
- const baseline = ctx.baselineJudgeScores ?? ctx.judgeScores;
4971
- const scenarioIds = new Set(ctx.scenarios.map((s) => s.id));
4972
- const axes = [];
4973
- for (const obj of objectives) {
4974
- let select;
4975
- if (obj.source.kind === "composite") select = (s) => s.composite;
4976
- else {
4977
- const dim = obj.source.dimension;
4978
- select = (s) => s.dimensions[dim];
4979
- }
4980
- const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select);
4981
- const before = obj.direction === "maximize" ? paired.before : paired.after;
4982
- const after = obj.direction === "maximize" ? paired.after : paired.before;
4983
- const n = paired.before.length;
4984
- const floorTolerance = obj.floorTolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
4985
- const gainThreshold = obj.gainThreshold ?? 0;
4986
- const bootstrapStatistic = opts.statistic ?? "mean";
4987
- const improvement = decidePairedPromotion(before, after, {
4988
- confidence,
4989
- resamples,
4990
- statistic: bootstrapStatistic,
4991
- seed,
4992
- threshold: gainThreshold,
4993
- minPairs: opts.minProductiveRuns
4994
- });
4995
- const regression = decidePairedPromotion(after, before, {
4996
- confidence,
4997
- resamples,
4998
- statistic: bootstrapStatistic,
4999
- seed,
5000
- threshold: floorTolerance,
5001
- minPairs: opts.minProductiveRuns
5002
- });
5003
- const bootstrap = improvement.bootstrap ?? pairedBootstrap(before, after, {
5004
- confidence,
5005
- resamples,
5006
- statistic: bootstrapStatistic,
5007
- seed
5008
- });
5009
- const floorBreached = bootstrap.low < -floorTolerance || regression.promote;
5010
- const verdict = !improvement.sufficient ? "few_runs" : floorBreached ? "regressed" : improvement.promote ? "improved" : "flat";
5011
- axes.push({
5012
- name: obj.name,
5013
- source: obj.source,
5014
- direction: obj.direction,
5015
- bootstrap,
5016
- bootstrapStatistic,
5017
- ci: {
5018
- low: improvement.low,
5019
- high: improvement.high
5020
- },
5021
- decisionStatistic: improvement.statistic,
5022
- mcnemar: improvement.mcnemar,
5023
- indeterminate: improvement.indeterminate,
5024
- n,
5025
- minimumRequired: improvement.minimumPairs,
5026
- decisionMethod: improvement.method,
5027
- gainThreshold,
5028
- floorTolerance,
5029
- verdict
5030
- });
5031
- }
5032
- const ns = axes.map((a) => a.n).filter((n) => n > 0);
5033
- return {
5034
- axes,
5035
- minN: ns.length > 0 ? Math.min(...ns) : 0,
5036
- cost: {
5037
- candidate: ctx.cost.candidate,
5038
- baseline: ctx.cost.baseline
5039
- }
5040
- };
5041
- }
5042
- /**
5043
- * The default strategy: symmetric multi-objective Pareto significance. Ship iff
5044
- * the candidate weakly dominates the baseline at the confidence level — no axis
5045
- * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
5046
- * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
5047
- * need_more_work. Statistically equivalent → hold (never ship noise).
5048
- */
5049
- const paretoPolicy = (ev) => {
5050
- const contributingGates = ev.axes.map((ax) => ({
5051
- name: `objective:${ax.name}`,
5052
- status: ax.verdict === "regressed" ? "fail" : ax.verdict === "few_runs" ? "not_evaluated" : "pass",
5053
- detail: {
5054
- direction: ax.direction,
5055
- source: ax.source,
5056
- verdict: ax.verdict,
5057
- n: ax.n,
5058
- deltaMedian: ax.bootstrap.median,
5059
- ciLow: ax.ci.low,
5060
- ciHigh: ax.ci.high,
5061
- decisionStatistic: ax.decisionStatistic,
5062
- decisionMethod: ax.decisionMethod,
5063
- mcnemar: ax.mcnemar,
5064
- indeterminate: ax.indeterminate,
5065
- bootstrapCiLow: ax.bootstrap.low,
5066
- bootstrapCiHigh: ax.bootstrap.high,
5067
- confidence: ax.bootstrap.confidence,
5068
- gainThreshold: ax.gainThreshold,
5069
- floorTolerance: ax.floorTolerance
5070
- }
5071
- }));
5072
- const regressed = ev.axes.filter((a) => a.verdict === "regressed");
5073
- const fewRuns = ev.axes.filter((a) => a.verdict === "few_runs");
5074
- const improved = ev.axes.filter((a) => a.verdict === "improved");
5075
- let decision;
5076
- const reasons = [];
5077
- if (regressed.length > 0) {
5078
- decision = "hold";
5079
- for (const a of regressed) reasons.push(`objective '${a.name}' regressed: good-direction CI.low ${a.ci.low.toFixed(3)} < -${a.floorTolerance} (n=${a.n})`);
5080
- } else if (fewRuns.length > 0) {
5081
- decision = "need_more_work";
5082
- for (const a of fewRuns) reasons.push(`objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance`);
5083
- } else if (improved.length > 0) {
5084
- decision = "ship";
5085
- reasons.push(`Pareto improvement at the confidence level: ${improved.map((a) => `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`).join(", ")}; no objective regressed`);
5086
- } else {
5087
- decision = "hold";
5088
- reasons.push("no Pareto improvement: candidate statistically equivalent to baseline on every objective");
5089
- }
5090
- const composite = ev.axes.find((a) => a.source.kind === "composite") ?? ev.axes[0];
5091
- return {
5092
- decision,
5093
- reasons,
5094
- contributingGates,
5095
- delta: composite?.bootstrap.median
5096
- };
5097
- };
5098
- /**
5099
- * Wrap the bus + a policy as a `Gate`. Plugs into the existing
5100
- * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
5101
- * loop behavior is unchanged because consumers opt in by passing this gate.
5102
- */
5103
- function paretoSignificanceGate(options) {
5104
- if (options.objectives.length === 0) throw new Error("paretoSignificanceGate: at least 1 objective required");
5105
- const policy = options.policy ?? paretoPolicy;
5106
- return {
5107
- name: options.name ?? "paretoSignificanceGate",
5108
- async decide(ctx) {
5109
- const ev = buildEvidenceVector(ctx, options.objectives, options);
5110
- return policy(ev);
5111
- }
5112
- };
5113
- }
5114
- //#endregion
5115
4471
  //#region src/campaign/external-optimizer-accounting.ts
5116
4472
  const TOKEN_USAGE_FIELDS = [
5117
4473
  "inputTokens",
@@ -7162,6 +6518,6 @@ function skillOptOptimizationMethod(config) {
7162
6518
  };
7163
6519
  }
7164
6520
  //#endregion
7165
- export { assertCampaignSplitIdentity as $, assertOptimizationResult as A, summarizeAgentReceiptIntegrity as At, surfaceContentHash as B, detectScale as C, decidePairedPromotion as Ct, runCanaries as D, BackendIntegrityError as Dt, pairHoldout as E, pairedDeltaTest as Et, assertCodeSurfaceIdentity as F, DEFAULT_MUTATION_PRIMITIVES as G, campaignBreakdown as H, assertComponentSurface as I, planCampaignRun as J, buildReflectionPrompt as K, codeSurfaceIdentityMaterial as L, compareOptimizationMethods as M, JudgeParseError as Mt, costFromLedgerSummary as N, composeGate as O, assertRealAgentReceipts as Ot, optimizationTokenUsageFromSummary as P, assertCampaignDesign as Q, componentSurfaceIdentityMaterial as R, defaultProductionGate as S, recoverTruncatedJson as St, heldoutSignificance as T, minimumPairsForPairedDeltaTest as Tt, campaignMeanComposite as U, surfaceHash as V, compareRankKeys as W, resolveRunDir as X, runCampaign as Y, tangleTracesRoot as Z, buildEvidenceVector as _, crowdingDistance as _t, emitLoopProvenance as a, REFERENCE_EQUIVALENCE_JUDGE_VERSION as at, powerPreflight as b, paretoFrontierWithCrowding as bt, provenanceRecordPath as c, DEFAULT_RED_TEAM_CORPUS as ct, runImprovementLoop as d, scoreRedTeamOutput as dt, campaignScenarioIdentity as et, runOptimization as f, toolNamesForRun as ft, gepaOptimizationMethod as g, llmJudge as gt, labelTrustRank as h, hashScenarios as ht, canonicalDigest as i, REFERENCE_EQUIVALENCE_INPUT_LIMITS as it, combineComparisonCosts as j, summarizeBackendIntegrity as jt, externalTextOptimizationMethod as k, assertRealBackend as kt, provenanceSpansPath as l, redTeamDataset as lt, isProposedCandidate as m, HoldoutLockedError as mt, buildLoopProvenanceRecord as n, campaignSplitDigestFromIdentities as nt, loopProvenanceArgsFromResult as o, createReferenceEquivalenceJudge as ot, runEval as p, Dataset as pt, parseReflectionResponse as q, campaignMeasurementDigest as r, openAutoPr as rt, loopProvenanceSpans as s, runReferenceEquivalenceJudge as st, skillOptOptimizationMethod as t, campaignSplitDigest as tt, verifyLoopProvenanceRecord as u, redTeamReport as ut, paretoPolicy as v, dominates as vt, dimensionRegressions as w, pairedDecisionShape as wt, heldOutGate as x, scalarScore as xt, paretoSignificanceGate as y, paretoFrontier as yt, renderSurfaceDiff as z };
6521
+ export { runReferenceEquivalenceJudge as $, componentSurfaceIdentityMaterial as A, planCampaignRun as B, combineComparisonCosts as C, assertCodeSurfaceIdentity as D, optimizationTokenUsageFromSummary as E, campaignMeanComposite as F, assertCampaignSplitIdentity as G, resolveRunDir as H, compareRankKeys as I, campaignSplitDigestFromIdentities as J, campaignScenarioIdentity as K, DEFAULT_MUTATION_PRIMITIVES as L, surfaceContentHash as M, surfaceHash as N, assertComponentSurface as O, campaignBreakdown as P, createReferenceEquivalenceJudge as Q, buildReflectionPrompt as R, assertOptimizationResult as S, costFromLedgerSummary as T, tangleTracesRoot as U, runCampaign as V, assertCampaignDesign as W, REFERENCE_EQUIVALENCE_INPUT_LIMITS as X, openAutoPr as Y, REFERENCE_EQUIVALENCE_JUDGE_VERSION as Z, heldOutGate as _, assertRealBackend as _t, emitLoopProvenance as a, Dataset as at, composeGate as b, JudgeParseError as bt, provenanceRecordPath as c, llmJudge as ct, runImprovementLoop as d, paretoFrontier as dt, DEFAULT_RED_TEAM_CORPUS as et, runOptimization as f, paretoFrontierWithCrowding as ft, gepaOptimizationMethod as g, assertRealAgentReceipts as gt, labelTrustRank as h, BackendIntegrityError as ht, canonicalDigest as i, toolNamesForRun as it, renderSurfaceDiff as j, codeSurfaceIdentityMaterial as k, provenanceSpansPath as l, crowdingDistance as lt, isProposedCandidate as m, recoverTruncatedJson as mt, buildLoopProvenanceRecord as n, redTeamReport as nt, loopProvenanceArgsFromResult as o, HoldoutLockedError as ot, runEval as p, scalarScore as pt, campaignSplitDigest as q, campaignMeasurementDigest as r, scoreRedTeamOutput as rt, loopProvenanceSpans as s, hashScenarios as st, skillOptOptimizationMethod as t, redTeamDataset as tt, verifyLoopProvenanceRecord as u, dominates as ut, defaultProductionGate as v, summarizeAgentReceiptIntegrity as vt, compareOptimizationMethods as w, externalTextOptimizationMethod as x, runCanaries as y, summarizeBackendIntegrity as yt, parseReflectionResponse as z };
7166
6522
 
7167
- //# sourceMappingURL=skillopt-optimization-method-C4FX42dy.js.map
6523
+ //# sourceMappingURL=skillopt-optimization-method-CQwZ-ZX8.js.map