@tangle-network/agent-eval 0.144.5 → 0.144.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (257) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
  6. package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
  7. package/dist/analyst/index.d.ts +471 -89
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +24 -5
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  12. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  13. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  14. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  15. package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
  16. package/dist/benchmark-command-BCafwNrf.js.map +1 -0
  17. package/dist/benchmarks/index.d.ts +1 -1
  18. package/dist/benchmarks/index.js +1 -1
  19. package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
  20. package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
  21. package/dist/campaign/index.d.ts +8 -6
  22. package/dist/campaign/index.js +5 -3
  23. package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
  24. package/dist/campaign-Tdy3h62h.js.map +1 -0
  25. package/dist/cli.js +2 -2
  26. package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
  27. package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  28. package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
  29. package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  30. package/dist/contract/index.d.ts +11 -10
  31. package/dist/contract/index.d.ts.map +1 -1
  32. package/dist/contract/index.js +7 -6
  33. package/dist/contract/index.js.map +1 -1
  34. package/dist/control.d.ts +2 -2
  35. package/dist/control.js +1 -1
  36. package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
  37. package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
  38. package/dist/counterfactual-CWPTrMH7.js +126 -0
  39. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  40. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  41. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  42. package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
  43. package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
  44. package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  45. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  46. package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
  47. package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  48. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  49. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  50. package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
  51. package/dist/engine-nB64f48I.d.ts.map +1 -0
  52. package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
  53. package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
  54. package/dist/errors-D-LKuDhb.js.map +1 -1
  55. package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
  56. package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  57. package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  58. package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  59. package/dist/exec-BLtYZdWo.js +49 -0
  60. package/dist/exec-BLtYZdWo.js.map +1 -0
  61. package/dist/experiment/index.d.ts +802 -0
  62. package/dist/experiment/index.d.ts.map +1 -0
  63. package/dist/experiment/index.js +1108 -0
  64. package/dist/experiment/index.js.map +1 -0
  65. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  66. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  67. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  68. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  69. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
  70. package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  71. package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
  72. package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  73. package/dist/fuzz.d.ts +2 -2
  74. package/dist/hosted/index.d.ts +3 -3
  75. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  76. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  77. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  78. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  79. package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
  80. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  81. package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
  82. package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
  83. package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
  84. package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  85. package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
  86. package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  87. package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
  88. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  89. package/dist/index.d.ts +252 -451
  90. package/dist/index.d.ts.map +1 -1
  91. package/dist/index.js +271 -751
  92. package/dist/index.js.map +1 -1
  93. package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
  94. package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  95. package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
  96. package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  97. package/dist/integrity-MLzHOfV9.js +141 -0
  98. package/dist/integrity-MLzHOfV9.js.map +1 -0
  99. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  100. package/dist/ledger-core/index.d.ts +1 -1
  101. package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
  102. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  103. package/dist/matrix/index.d.ts +2 -2
  104. package/dist/meta-eval/index.d.ts +2 -2
  105. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  106. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  107. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  108. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  109. package/dist/multishot/index.d.ts +3 -3
  110. package/dist/openapi.json +1 -1
  111. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  112. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  113. package/dist/pipelines/index.d.ts +2 -1
  114. package/dist/pipelines/index.d.ts.map +1 -1
  115. package/dist/pre-registration-DakwTRXk.js +96 -0
  116. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  117. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  118. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  119. package/dist/prime-protocol-BfSalTfR.js +453 -0
  120. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  121. package/dist/profile-cell.d.ts +1 -1
  122. package/dist/profile-cell.js +242 -1
  123. package/dist/profile-cell.js.map +1 -0
  124. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  125. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  126. package/dist/promotion-policy-CrLrmys8.js +682 -0
  127. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  128. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  129. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  130. package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
  131. package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  132. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  133. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  134. package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
  135. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  136. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  137. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  138. package/dist/reporting.d.ts +4 -4
  139. package/dist/reporting.js +2 -2
  140. package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
  141. package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  142. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  143. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  144. package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
  145. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  146. package/dist/rl.d.ts +19 -9
  147. package/dist/rl.d.ts.map +1 -1
  148. package/dist/rl.js +16 -7
  149. package/dist/rl.js.map +1 -1
  150. package/dist/rollout/index.d.ts +2 -2
  151. package/dist/rollout/index.js +2 -2
  152. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  153. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  154. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
  155. package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
  156. package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
  157. package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  158. package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
  159. package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
  160. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  161. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  162. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  163. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  164. package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  165. package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  166. package/dist/sequential-D-BLJBKU.js +299 -0
  167. package/dist/sequential-D-BLJBKU.js.map +1 -0
  168. package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
  169. package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
  170. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  171. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  172. package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
  173. package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  174. package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
  175. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  176. package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
  177. package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
  178. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  179. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  180. package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  181. package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  182. package/dist/steps-BArUxhna.d.ts +51 -0
  183. package/dist/steps-BArUxhna.d.ts.map +1 -0
  184. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  185. package/dist/store-DNe_Uv1Q.js.map +1 -0
  186. package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
  187. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  188. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  189. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  190. package/dist/supervisor-run/index.d.ts +2 -2
  191. package/dist/supervisor-run/index.js +1 -1
  192. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  193. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  194. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  195. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  196. package/dist/trace-repair/index.d.ts +2102 -0
  197. package/dist/trace-repair/index.d.ts.map +1 -0
  198. package/dist/trace-repair/index.js +3878 -0
  199. package/dist/trace-repair/index.js.map +1 -0
  200. package/dist/traces.d.ts +6 -6
  201. package/dist/traces.js +3 -2
  202. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  203. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  204. package/dist/trajectory-replay/index.d.ts +781 -0
  205. package/dist/trajectory-replay/index.d.ts.map +1 -0
  206. package/dist/trajectory-replay/index.js +2103 -0
  207. package/dist/trajectory-replay/index.js.map +1 -0
  208. package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
  209. package/dist/types-D216SgwM.d.ts.map +1 -0
  210. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  211. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  212. package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
  213. package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  214. package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
  215. package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  216. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  217. package/dist/verdict-DExhxfgR.d.ts +201 -0
  218. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  219. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  220. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  221. package/dist/wire/index.d.ts +3 -3
  222. package/dist/wire/index.js +1 -1
  223. package/docs/building-doctrine.md +15 -0
  224. package/docs/charter.md +112 -0
  225. package/docs/experiment.md +104 -0
  226. package/docs/prime-analyst.md +1 -0
  227. package/docs/trace-analysis.md +26 -0
  228. package/docs/trace-repair-admission.md +194 -0
  229. package/docs/trace-repair-analyst-arms.md +121 -0
  230. package/docs/trace-repair-continuation.md +107 -0
  231. package/docs/trace-repair-grader.md +163 -0
  232. package/docs/trajectory-replay.md +110 -0
  233. package/docs/verification-strategies.md +103 -0
  234. package/package.json +21 -4
  235. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  236. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  237. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  238. package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
  239. package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
  240. package/dist/campaign-ClpnD7Ug.js.map +0 -1
  241. package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
  242. package/dist/index-CpxZSlB7.d.ts.map +0 -1
  243. package/dist/index-CsuAo2-J.d.ts.map +0 -1
  244. package/dist/integrity-fdt8XPAv.js.map +0 -1
  245. package/dist/llm-client-D3EoChAU.js.map +0 -1
  246. package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
  247. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  248. package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
  249. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  250. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  251. package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
  252. package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
  253. package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
  254. package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
  255. package/dist/types-BjMFz88h.d.ts.map +0 -1
  256. package/dist/verdict-Dps8_okt.d.ts +0 -37
  257. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -1,15 +1,16 @@
1
- import { l as ValidationError, t as AgentEvalError } from "./errors-DkfjIDvD.js";
2
- import { T as RunPaidCallInput, m as CostReceipt, p as CostProvenance } from "./cost-ledger-FuQvHxPm.js";
3
- import { a as RunRecord, s as RunSplitTag } from "./run-record-ooo9FWns.js";
4
- import { f as AnalystUsageReceipt, i as AnalystFinding, l as AnalystRunResult, x as TraceAnalysisStore } from "./types-CZt1PBIk.js";
5
- import { n as CompletionVerdict, o as ProducedState, r as CorrectnessChecker, t as CompletionRequirement } from "./completion-verifier-VvpHRu78.js";
6
- import { _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, t as AnalystBenchmarkCase } from "./benchmark-Fmo42QVE.js";
7
- import { A as LabeledScenarioWrite, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, O as LabeledScenarioSource, R as Scenario, S as JudgeConfig, T as LabelTrust, a as CampaignResult, b as GenerationRecord, d as DispatchContext, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, p as Gate, u as ComponentSurface, w as JudgeScore } from "./types-Dcoaqcsc.js";
8
- import { E as RiskDifferenceResult, _ as McNemarResult, d as EProcessState, v as PairedBootstrapOptions, y as PairedBootstrapResult } from "./statistics-B4u_CiFd.js";
9
- import { A as PairedMcNemarEvidence, D as PairedDecisionMethod, j as PairedPromotionDecision, k as PairedDecisionStatistic } from "./summary-report-BOM6dfP7.js";
10
- import { in as PlanCampaignRunOptions, nn as CampaignRunPlan, un as CampaignStorage } from "./skillopt-optimization-method-CGz9ywhM.js";
11
- import { D as LedgerHash, m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-DEb46kc6.js";
12
- import { AgentProfile, AgentProfile as AgentProfile$1, HarnessType, HarnessType as HarnessType$1 } from "@tangle-network/agent-interface";
1
+ import { l as ValidationError, t as AgentEvalError } from "./errors-CKPfb2aH.js";
2
+ import { T as RunPaidCallInput, m as CostReceipt, p as CostProvenance } from "./cost-ledger-Bv_e8XHY.js";
3
+ import { a as RunRecord, s as RunSplitTag } from "./run-record-DdSa93_W.js";
4
+ import { f as AnalystUsageReceipt, i as AnalystFinding, l as AnalystRunResult, w as TraceAnalysisStore } from "./types-DF_Udrp-.js";
5
+ import { n as CompletionVerdict, o as ProducedState, r as CorrectnessChecker, t as CompletionRequirement } from "./completion-verifier-foUCLif_.js";
6
+ import { D as AnalystIssueExpectation, d as AnalystBenchmarkCase, h as AnalystBenchmarkLabelState, t as AgentProfile$1 } from "./agent-profile-CgDTo40f.js";
7
+ import { A as LabeledScenarioWrite, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, O as LabeledScenarioSource, R as Scenario, S as JudgeConfig, T as LabelTrust, a as CampaignResult, d as DispatchContext, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, p as Gate, u as ComponentSurface } from "./types-DYuNHo9R.js";
8
+ import { y as PairedBootstrapResult } from "./statistics-D6Uebe_4.js";
9
+ import { Ht as CampaignStorage, It as PlanCampaignRunOptions, Pt as CampaignRunPlan } from "./skillopt-optimization-method-BO7NAl3b.js";
10
+ import { N as PairedArmsComparison } from "./statistical-heldout-Dn9ruizm.js";
11
+ import "./promotion-policy-ChWhTDBH.js";
12
+ import { D as LedgerHash, m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-CvSN3IG1.js";
13
+ import { AgentProfile } from "@tangle-network/agent-interface";
13
14
  //#region src/campaign/analyst-surface.d.ts
14
15
  interface TraceAnalystScenario extends Scenario {
15
16
  kind: 'trace-analyst';
@@ -34,152 +35,6 @@ interface BuildTraceAnalystSurfaceDispatchOptions {
34
35
  declare function buildTraceAnalystSurfaceDispatch(options: BuildTraceAnalystSurfaceDispatchOptions): (surface: MutableSurface, scenario: TraceAnalystScenario, context: DispatchContext) => Promise<TraceAnalystArtifact>;
35
36
  declare function traceAnalystQualityJudge(): JudgeConfig<TraceAnalystArtifact, TraceAnalystScenario>;
36
37
  //#endregion
37
- //#region src/paired-arms.d.ts
38
- /** One arm observation of one work item. Structural on purpose: callers
39
- * project their own record type (e.g. a `RunRecord`) into this shape. */
40
- interface PairedArmRow {
41
- /** Matching key — rows sharing a `pairKey` across both arms form pairs
42
- * (typically the task/scenario/seed identity). */
43
- pairKey: string;
44
- /** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
45
- * every row of a `pairKey` that has more than one rep in either arm; reps
46
- * then pair only on exact (`pairKey`, `repKey`) match, never on outcome
47
- * content. Optional when each arm has at most one rep of the item. */
48
- repKey?: string;
49
- /** Arm label this row was produced under. */
50
- arm: string;
51
- /** Binary outcome; omit when the comparison has no pass/fail notion. */
52
- pass?: boolean;
53
- /** Named numeric measurements (score, cost, latency, …). */
54
- metrics?: Record<string, number>;
55
- }
56
- interface PairArmsOptions {
57
- /** Arm treated as the control side of every pair. */
58
- baselineArm: string;
59
- /** Arm treated as the treatment side of every pair. */
60
- treatmentArm: string;
61
- }
62
- /** One matched (baseline, treatment) observation of the same work item. */
63
- interface MatchedPair {
64
- pairKey: string;
65
- /** 0-based position of this pair within its `pairKey`, ordered by sorted
66
- * `repKey` (always 0 for a single-rep item). The rep identity itself is on
67
- * the rows (`baseline.repKey` / `treatment.repKey`). */
68
- repIndex: number;
69
- baseline: PairedArmRow;
70
- treatment: PairedArmRow;
71
- }
72
- interface PairArmsResult {
73
- /** Matched pairs, ordered by (`pairKey`, `repIndex`). */
74
- pairs: MatchedPair[];
75
- /** Baseline rows left without a treatment counterpart — reported, never
76
- * silently dropped. */
77
- unpairedBaseline: PairedArmRow[];
78
- /** Treatment rows left without a baseline counterpart. */
79
- unpairedTreatment: PairedArmRow[];
80
- }
81
- /**
82
- * Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
83
- *
84
- * A `pairKey` with at most one row per arm pairs directly, no `repKey`
85
- * needed. A `pairKey` with multiple reps in either arm requires `repKey` on
86
- * every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
87
- * match — pairing is keyed purely on row identity, never on outcome content
88
- * (outcome-keyed matching deflates discordant counts and biases McNemar), and
89
- * is therefore independent of input order. Reps whose `repKey` has no
90
- * counterpart in the other arm, and items present in only one arm, land in
91
- * the unpaired lists — reported, never truncated.
92
- *
93
- * Fail-loud: throws when either named arm has zero rows (an unknown arm
94
- * name would otherwise read as "everything unpaired"), when the two arm
95
- * names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
96
- * when a (`pairKey`, arm) group repeats a `repKey` (the match would be
97
- * ambiguous).
98
- */
99
- declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
100
- /** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
101
- interface PairedCorrectness {
102
- /** Discordant pairs where the treatment passed and the baseline failed. */
103
- b10: number;
104
- /** Discordant pairs where the baseline passed and the treatment failed. */
105
- b01: number;
106
- /** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
107
- mcnemar: McNemarResult;
108
- /** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
109
- riskDifference: RiskDifferenceResult;
110
- }
111
- /** Paired delta summary for one named metric (delta = treatment − baseline). */
112
- interface PairedMetricDelta {
113
- name: string;
114
- /** Pairs where BOTH sides carry a finite value for this metric. */
115
- n: number;
116
- /** Pairs where at least one side does not carry the metric. */
117
- nMissing: number;
118
- /** Median paired delta, or null when `n === 0`. */
119
- medianDelta: number | null;
120
- /** Mean paired delta, or null when `n === 0`. */
121
- meanDelta: number | null;
122
- /** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
123
- * `n === 0` — a zero-width [0, 0] interval on no data would read as a
124
- * measured tight null. */
125
- bootstrapCi: PairedBootstrapResult | null;
126
- /** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
127
- wilcoxon: {
128
- w: number;
129
- p: number;
130
- } | null;
131
- }
132
- interface ComparePairedArmsOptions extends PairArmsOptions {
133
- /** Metrics to compare. Default: every metric name observed on any matched
134
- * pair, sorted. A name that appears on no pair is still reported (with
135
- * `n = 0`) so a misspelled metric is visible instead of vanishing. */
136
- metricNames?: string[];
137
- /** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
138
- bootstrap?: PairedBootstrapOptions;
139
- }
140
- interface PairedArmsComparison {
141
- nPairs: number;
142
- nUnpairedBaseline: number;
143
- nUnpairedTreatment: number;
144
- /** null when no matched pair carries `pass` on both sides — a pass/fail
145
- * verdict over rows that never measured pass/fail would be fabricated. */
146
- correctness: PairedCorrectness | null;
147
- metricDeltas: PairedMetricDelta[];
148
- }
149
- /**
150
- * Full matched-pair arm comparison: pair via {@link pairArms}, then compose
151
- * the paired estimators from `statistics` over the matched pairs.
152
- *
153
- * Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
154
- * is that subset's size); each metric uses only the pairs where both sides
155
- * carry a finite value for it, with the remainder counted in `nMissing`.
156
- * Deltas are treatment − baseline throughout.
157
- *
158
- * Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
159
- * non-finite metric value — silently treating corrupt telemetry as "metric
160
- * absent" would misreport it as missing coverage.
161
- */
162
- declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
163
- interface MatchedRunRecordPair {
164
- pairKey: string;
165
- repKey: string;
166
- baseline: RunRecord;
167
- treatment: RunRecord;
168
- }
169
- interface PairRunRecordsResult {
170
- pairs: MatchedRunRecordPair[];
171
- unpairedBaseline: RunRecord[];
172
- unpairedTreatment: RunRecord[];
173
- }
174
- /**
175
- * Pair two RunRecord arms by the identity of the evaluated work:
176
- * `(experimentId, scenarioId, seed)`.
177
- *
178
- * Falling back to array order, candidate id, or experiment id can compare
179
- * different tasks and fabricate lift. Duplicate identities throw.
180
- */
181
- declare function pairRunRecords(baselineRuns: readonly RunRecord[], treatmentRuns: readonly RunRecord[]): PairRunRecordsResult;
182
- //#endregion
183
38
  //#region src/campaign/cross-surface-types.d.ts
184
39
  /** Whether one candidate attempt produced a usable executable outcome. */
185
40
  type CrossSurfaceAttemptCompleteness = 'complete' | 'missing' | 'invalid';
@@ -525,409 +380,6 @@ interface NeutralizationGateOptions<TScenario extends Scenario = Scenario> {
525
380
  */
526
381
  declare function neutralizationGate<TArtifact, TScenario extends Scenario>(options: NeutralizationGateOptions<TScenario>): Gate<TArtifact, TScenario>;
527
382
  //#endregion
528
- //#region src/pre-registration.d.ts
529
- /**
530
- * Pre-registered hypotheses — declare what you're testing BEFORE the
531
- * run, check it AFTER. Prevents p-hacking, optional stopping, and the
532
- * "we ran until it looked good" failure mode.
533
- *
534
- * Manifest is a plain JSON-friendly object. Sign it with a content hash
535
- * + timestamp; the registered record becomes immutable. Post-run,
536
- * evaluate the manifest against observed results — the library refuses
537
- * to let you re-interpret a different metric as the declared one.
538
- */
539
- interface HypothesisManifest {
540
- id: string;
541
- /** Human prose — goes into the audit trail. */
542
- hypothesis: string;
543
- /** Metric the hypothesis claims to move. */
544
- metric: string;
545
- /** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
546
- direction: 'increase' | 'decrease';
547
- /** Minimum effect size to count (same units as the metric). */
548
- minEffect: number;
549
- /** Alpha threshold. */
550
- alpha: number;
551
- /** Target statistical power at which sample size was pre-computed. */
552
- power: number;
553
- /** Declared N per arm before running. */
554
- preRegisteredN: number;
555
- /** ISO8601 timestamp the manifest was registered. */
556
- registeredAt: string;
557
- /** Optional identifiers to tie into the trace corpus. */
558
- baselineLabel?: string;
559
- candidateLabel?: string;
560
- }
561
- /**
562
- * Identifier for the hashing scheme used to produce `contentHash`.
563
- *
564
- * `'sha256-content'` — sha256 hex over the canonicalized manifest with
565
- * the `contentHash` and `algo` fields stripped. Held as a string union
566
- * so future schemes can be added without breaking parsers; SignedManifest
567
- * values without `algo` deserialize cleanly because the field is optional.
568
- */
569
- type SignedManifestAlgo = 'sha256-content';
570
- interface SignedManifest extends HypothesisManifest {
571
- /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
572
- contentHash: string;
573
- /**
574
- * Algorithm string describing how `contentHash` was produced.
575
- *
576
- * Optional on the type so serialized manifests without it still parse,
577
- * but ALWAYS populated by {@link signManifest}. Consumers that want to
578
- * enforce a known algorithm should reject manifests where this field
579
- * is missing or unrecognized.
580
- */
581
- algo?: SignedManifestAlgo;
582
- }
583
- interface HypothesisResult {
584
- manifest: SignedManifest;
585
- observedN: number;
586
- observedEffect: number;
587
- observedPValue: number;
588
- /** True iff the observed effect hits the pre-declared direction with
589
- * magnitude ≥ minEffect AND p < alpha. */
590
- confirmed: boolean;
591
- /** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
592
- rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
593
- notes?: string;
594
- }
595
- /**
596
- * Deterministic JSON canonicalization — sort object keys recursively.
597
- *
598
- * Two semantically-equal objects produce byte-identical canonicalized output;
599
- * this is what makes a content-hash stable across encoders, key insertion
600
- * orders, and runtime versions. Exported for any consumer that needs the same
601
- * canonicalization guarantee outside the manifest-signing path (e.g., signing
602
- * an artifact bundle, hashing a dataset version, etc.).
603
- */
604
- declare function canonicalize(v: unknown): unknown;
605
- /**
606
- * SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.
607
- *
608
- * The same primitive `signManifest` and `verifyManifest` are built on, exposed
609
- * directly so consumers signing arbitrary structured content (artifact bundles,
610
- * production packets, dataset manifests, etc.) don't have to re-derive
611
- * canonicalize+sha256 from scratch.
612
- *
613
- * Stable across:
614
- * - object key insertion order (canonicalization sorts keys recursively)
615
- * - encoder choice (UTF-8 via TextEncoder, fixed)
616
- * - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)
617
- *
618
- * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
619
- * which takes a string input and returns a truncated 12-char prompt id.
620
- * Use `hashJson` when you mean "canonicalize then hash."
621
- *
622
- * @example
623
- * const hash = await hashJson({ id: '1', kind: 'spec' })
624
- * // 'a3f1...' (64 hex chars)
625
- */
626
- declare function hashJson<T>(obj: T): Promise<string>;
627
- /**
628
- * Sign a manifest with a SHA-256 content hash.
629
- *
630
- * The hash covers the canonicalized manifest with the `contentHash`
631
- * and `algo` fields stripped; this lets verifiers re-sign the rest and
632
- * compare. Returned manifest always carries `algo: 'sha256-content'`
633
- * so downstream consumers can identify the scheme; manifests without
634
- * `algo` still verify because it is stripped before hashing on both sides.
635
- */
636
- declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
637
- /**
638
- * Verify that a signed manifest has not been tampered with.
639
- *
640
- * Strips `contentHash` and `algo` before re-signing so manifests without
641
- * `algo` verify identically to ones that carry it.
642
- */
643
- declare function verifyManifest(m: SignedManifest): Promise<boolean>;
644
- /**
645
- * Evaluate a pre-registered hypothesis against observed results.
646
- * Mechanical — no re-interpretation permitted.
647
- */
648
- declare function evaluateHypothesis(manifest: SignedManifest, observed: {
649
- n: number;
650
- effect: number;
651
- pValue: number;
652
- }): Promise<HypothesisResult>;
653
- //#endregion
654
- //#region src/campaign/gates/sequential.d.ts
655
- type SequentialDecision = 'promote' | 'continue' | 'undecided-at-maxN';
656
- interface SequentialObservation {
657
- decision: SequentialDecision;
658
- /** Current e-value (the betting wealth) against H0. */
659
- eValue: number;
660
- /** Paired deltas consumed so far. */
661
- n: number;
662
- /** Names the decision basis. For 'undecided-at-maxN' it states explicitly
663
- * that exhausting the budget is NOT evidence of no effect. */
664
- reason: string;
665
- }
666
- interface SequentialPairedGateOptions {
667
- /** Type-I budget. With `preRegistration` bound this MUST match
668
- * `manifest.alpha` (conflict throws). Default 0.05. */
669
- alpha?: number;
670
- /** Minimum paired deltas before a promote may fire. The stopping rule is
671
- * "first n ≥ minN with e-value ≥ 1/alpha" — still a valid stopping time.
672
- * Default 5. */
673
- minN?: number;
674
- /** Pre-registered observation budget. Required unless `preRegistration`
675
- * supplies it via `preRegisteredN` (conflict throws). */
676
- maxN?: number;
677
- /** Bet truncation forwarded to `eProcess`. Default 0.5. */
678
- maxBet?: number;
679
- /** Bound on |delta| in the judge's native scale; deltas are mapped to
680
- * x = (d/scale + 1)/2 ∈ [0,1]. A delta outside ±scale throws (use
681
- * `detectScale` to pick 1 vs 100 BEFORE streaming). Default 1. */
682
- scale?: number;
683
- /** Seed for the data-independent shuffle of paired deltas in `decide(ctx)`
684
- * (exchangeability guard). Default 1337. */
685
- shuffleSeed?: number;
686
- /** Bind the pre-registered hypothesis. Verified (content hash) at
687
- * construction; alpha/maxN/direction/minEffect come FROM the manifest. */
688
- preRegistration?: SignedManifest;
689
- /** Override the gate name in reports. */
690
- name?: string;
691
- }
692
- interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario> extends Gate<TArtifact, TScenario> {
693
- /** Streaming entry point: feed one paired per-scenario delta
694
- * (candidate − baseline, native scale). Each gate instance carries ONE
695
- * observe-stream; `decide(ctx)` runs on its own fresh stream and never
696
- * consumes or advances this one. 'promote' is sticky; observing past the
697
- * pre-registered maxN throws (extending a finished stream after seeing
698
- * the result reopens optional stopping — start a NEW pre-registered
699
- * test). */
700
- observe(delta: number): SequentialObservation;
701
- /** Read-only snapshot of the observe-stream. */
702
- state(): EProcessState & {
703
- decision: SequentialDecision;
704
- };
705
- }
706
- /**
707
- * Anytime-valid sequential paired gate. Conforms to the existing `Gate`
708
- * contract (`decide(ctx)` consumes candidate vs baseline judge scores via
709
- * `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
710
- * never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
711
- * that score cells incrementally and want to stop mid-stream.
712
- *
713
- * Decision mapping onto the substrate's five-valued `GateDecision`:
714
- * - 'promote' → 'ship'
715
- * - 'continue' → 'need_more_work' (stream ended before maxN with
716
- * the e-value undecided — more reps could decide)
717
- * - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
718
- * evidence of no effect (never a silent default)
719
- */
720
- declare function sequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: SequentialPairedGateOptions): SequentialPairedGate<TArtifact, TScenario>;
721
- interface SequentialDecideOptions {
722
- /** Type-I budget for the early-stop evidence. Default 0.05. */
723
- alpha?: number;
724
- /** Minimum paired deltas before a stop may fire. Default 5. */
725
- minN?: number;
726
- /** Bet truncation forwarded to `eProcess`. Default 0.5. */
727
- maxBet?: number;
728
- /** Bound on |per-scenario composite delta|. Default 1. */
729
- scale?: number;
730
- }
731
- interface SequentialDecideFn {
732
- (args: {
733
- history: GenerationRecord[];
734
- }): {
735
- stop: boolean;
736
- reason?: string;
737
- };
738
- /** Read-only snapshot of the accumulated e-process (observability + tests). */
739
- state(): EProcessState;
740
- }
741
- /**
742
- * `SurfaceProposer.decide` adapter — stops the optimization loop the moment
743
- * the e-process decides the loop has produced a real improvement, instead of
744
- * always running `maxGenerations`.
745
- *
746
- * Stream: for each generation g ≥ 1, the per-scenario composite deltas of
747
- * generation g's top candidate vs the generation-0 top candidate (the
748
- * incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
749
- * surface improves any scenario's expected composite over the incumbent —
750
- * under it every delta has conditional mean ≤ 0 and the e-process is valid.
751
- * Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
752
- * gate (which re-scores on HELD-OUT data — this adapter only spends the
753
- * exploration budget, it never promotes).
754
- *
755
- * Honesty caveats: (1) the incumbent's scores are measured once and shared
756
- * across all generations' deltas, so type-I control is exact only insofar as
757
- * those scores approximate the incumbent's true per-scenario means (more reps
758
- * → tighter); (2) an UNDECIDED process never stops the loop — absence of a
759
- * crossing is NOT evidence of no effect, so the loop simply runs its normal
760
- * course. Calling the adapter repeatedly with a growing history consumes each
761
- * generation exactly once (re-feeding an already-seen record would double-count
762
- * evidence).
763
- */
764
- declare function sequentialDecide(options?: SequentialDecideOptions): SequentialDecideFn;
765
- //#endregion
766
- //#region src/campaign/gates/statistical-heldout.d.ts
767
- interface PairedHoldout {
768
- /** Baseline scalar per paired cell (same order as `after`/`cellIds`). */
769
- before: number[];
770
- /** Candidate scalar per paired cell. */
771
- after: number[];
772
- /** The full cellIds (`scenario:rep`) that paired, in order. */
773
- cellIds: string[];
774
- }
775
- /**
776
- * Pair candidate vs baseline holdout observations by FULL cellId. `select`
777
- * pulls the scalar from a cell's judge reports (composite, or a named
778
- * dimension); a cell contributes the mean of `select` across its judges. Cells
779
- * whose scenario is not in `scenarioIds`, or where `select` is undefined for
780
- * every judge on either side, are skipped on BOTH sides so the arrays stay
781
- * paired. Throws when the two maps disagree on which holdout cells exist — a
782
- * load-bearing invariant: the baseline + winner holdout campaigns run the same
783
- * scenarios with the same seed base, so their cellIds MUST align; a mismatch
784
- * means a silent pairing bug, not a soft fallback.
785
- */
786
- declare function pairHoldout(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, select: (s: JudgeScore) => number | undefined): PairedHoldout;
787
- interface HeldoutSignificance {
788
- paired: PairedHoldout;
789
- /**
790
- * The paired bootstrap on the requested statistic (MEAN by default — see the
791
- * tie note on `heldoutSignificance`).
792
- *
793
- * DIAGNOSTIC, not necessarily the interval the verdict keyed on. On a
794
- * two-point (pass/fail) outcome the decision routes to Tango's score interval
795
- * instead, because a percentile bootstrap of the mean over a three-atom
796
- * lattice is not a valid interval at a nonzero margin. Read
797
- * `decision.low`/`decision.high` for the interval that actually decided, and
798
- * `decisionStatistic` for which one it is.
799
- */
800
- bootstrap: PairedBootstrapResult;
801
- /** The MEDIAN paired-delta bootstrap, reported as a diagnostic. When many
802
- * scenarios are tied (both sides solve them), the median is pinned near 0
803
- * regardless of the mean lift — comparing the two exposes tie-domination. */
804
- medianBootstrap: PairedBootstrapResult;
805
- /**
806
- * The full promotion decision: which estimator the outcome's shape admits,
807
- * the interval it produced, McNemar's exact veto on the two-point path, and
808
- * whether the interval was zero-width (no evidence in either direction). The
809
- * single source of `significant`.
810
- */
811
- decision: PairedPromotionDecision;
812
- /** Which paired estimator the verdict was decided on. */
813
- decisionStatistic: PairedDecisionStatistic;
814
- /** McNemar's exact evidence on the two-point path; null otherwise. */
815
- mcnemar: PairedMcNemarEvidence | null;
816
- /** Fraction of paired observations that are exact ties (|delta| < 1e-9). A
817
- * high tie fraction is WHY a median-based gate would have missed a real lift;
818
- * it is the observability the tie fix adds. */
819
- tieFraction: number;
820
- /** n paired observations. */
821
- n: number;
822
- /** Effective minimum after applying the bootstrap's hard statistical floor. */
823
- minimumRequired: number;
824
- /** Statistical method that carried the decision. */
825
- decisionMethod: PairedDecisionMethod;
826
- /** Exact one-sided p-value on the small-sample path; otherwise null. */
827
- pValue: number | null;
828
- /** True iff n >= minimumRequired, the DECIDING interval has nonzero width,
829
- * its lower bound clears the threshold, and McNemar's exact test does not
830
- * veto at a non-negative threshold. */
831
- significant: boolean;
832
- /** Set when n < minimumRequired — too little evidence to claim significance. */
833
- fewRuns: boolean;
834
- }
835
- interface HeldoutSignificanceOptions {
836
- deltaThreshold?: number;
837
- minProductiveRuns?: number;
838
- confidence?: number;
839
- resamples?: number;
840
- /** Fixed by default for a deterministic, reproducible gate verdict. */
841
- seed?: number;
842
- statistic?: 'mean' | 'median';
843
- }
844
- /**
845
- * Significance of the held-out composite lift: ship only when the lower bound
846
- * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
847
- * 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
848
- * scale.
849
- *
850
- * The decision is delegated whole to {@link decidePairedPromotion}, the one
851
- * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
852
- * also calls. That module's header carries the measurements; the short version
853
- * is three guards a bare `bootstrap.low > threshold` does not have:
854
- *
855
- * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
856
- * only paired-binary construction that stays valid at a nonzero margin;
857
- * - McNemar's exact test VETOES at any non-negative threshold;
858
- * - a ZERO-WIDTH interval is refused rather than promoted, in either
859
- * direction — [0,0] clears every negative threshold and [g,g] clears every
860
- * threshold below g, and both are an absence of evidence, not a result.
861
- *
862
- * Measured on this function before those guards landed, at a nominal 5 %:
863
- * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
864
- * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
865
- * delta is exactly 0.
866
- *
867
- * At small n, where the percentile bootstrap is descriptive only, a
868
- * pre-registered exact sign test still carries the bootstrap path.
869
- */
870
- declare function heldoutSignificance(paired: PairedHoldout, opts?: HeldoutSignificanceOptions): HeldoutSignificance;
871
- interface DimensionRegression {
872
- dimension: string;
873
- /** Paired bootstrap on (candidate − baseline). DIAGNOSTIC on a pass/fail
874
- * dimension, where `ci` carries the interval that decided instead. */
875
- bootstrap: PairedBootstrapResult;
876
- /** Which paired statistic `bootstrap.low` is the lower bound of. `'mean'`
877
- * unless the caller asked for the median. `bootstrap.median` still carries
878
- * the median point estimate either way. */
879
- bootstrapStatistic: 'median' | 'mean';
880
- /** The interval `regressed` was decided on, in the dimension's native units. */
881
- ci: {
882
- low: number;
883
- high: number;
884
- };
885
- /** Which estimator produced `ci`. */
886
- decisionStatistic: PairedDecisionStatistic;
887
- /** McNemar's exact evidence on a pass/fail dimension; null otherwise. */
888
- mcnemar: PairedMcNemarEvidence | null;
889
- /** `ci` has zero width — no evidence in either direction. */
890
- indeterminate: boolean;
891
- /** True iff the candidate may have regressed this dimension by more than
892
- * tolerance: the lower bound of the DECIDING interval on (candidate −
893
- * baseline) is below −tolerance, OR the exact small-sample test proves a drop
894
- * past tolerance. */
895
- regressed: boolean;
896
- tolerance: number;
897
- n: number;
898
- }
899
- /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
900
- * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
901
- * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
902
- declare function detectScale(values: number[]): 1 | 100;
903
- /** Per-critical-dimension regression guard. For each dimension, pair the
904
- * candidate vs baseline values by full cellId and bootstrap the paired delta;
905
- * a dimension is "regressed" when the CI lower bound < −tolerance (conservative
906
- * — blocks if the credible worst case exceeds tolerance, which is the right
907
- * posture for safety dimensions like `hallucination_free`). When `tolerance`
908
- * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
909
- *
910
- * The interval comes from {@link decidePairedPromotion}, so a pass/fail
911
- * dimension is judged on Tango's score interval rather than a percentile
912
- * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
913
- * is not a valid interval at one. That matters most here because this guard
914
- * fails OPEN by construction: `tolerance` is positive, so an interval pinned at
915
- * [0,0] never satisfies `low < −tolerance` and a real regression on a safety
916
- * dimension would be reported as `regressed: false`. On the median it fails the
917
- * same way for the same reason — when most pairs tie, which is automatic for a
918
- * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
919
- * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
920
- * restore the pre-0.134 behaviour. */
921
- declare function dimensionRegressions(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, criticalDimensions: string[], opts?: {
922
- tolerance?: number;
923
- confidence?: number;
924
- resamples?: number;
925
- seed?: number;
926
- /** Paired statistic the CI is computed on. Default `'mean'` — see
927
- * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
928
- statistic?: 'mean' | 'median';
929
- }): DimensionRegression[];
930
- //#endregion
931
383
  //#region src/campaign/grounded-reflection.d.ts
932
384
  /**
933
385
  * Evidence grounding for reflective optimizers (GEPA-style revise loops).
@@ -1118,92 +570,6 @@ type RuntimeEventLike = ToolCallEventLike | ArtifactEventLike | ProposalEventLik
1118
570
  */
1119
571
  declare function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState;
1120
572
  //#endregion
1121
- //#region src/agent-profile.d.ts
1122
- /**
1123
- * The agentic coding harnesses an eval sweeps by default — the ones we care about
1124
- * ranking. This is the SINGLE source of that list; consumers import it instead of
1125
- * re-declaring their own (a re-declared list is how the fleet drifts). Pass an
1126
- * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
1127
- * harness) to widen beyond these.
1128
- */
1129
- declare const CODING_HARNESSES: readonly HarnessType[];
1130
- interface ProfileAxisSpec {
1131
- /** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the
1132
- * harness and model vary. `model.default` is the fallback model. */
1133
- base: AgentProfile;
1134
- /** Harnesses to cross. Default: {@link CODING_HARNESSES}. */
1135
- harnesses?: readonly HarnessType[];
1136
- /** Models to cross. Default: `[base.model.default]` — one model, i.e. today's
1137
- * single-model behaviour, so omitting this never changes an existing run. */
1138
- models?: readonly string[];
1139
- /** Force every (harness, model) pair verbatim, even ones the harness can't run —
1140
- * for deliberately testing failure modes. Default (false): SNAP instead — a
1141
- * vendor-locked harness runs only the swept models in its family, or its native
1142
- * default when it supports none, so no harness is dropped and none gets a
1143
- * guaranteed-failing foreign-model cell. */
1144
- keepIncompatible?: boolean;
1145
- }
1146
- /** Model sentinel for a vendor-locked harness that supports none of the swept models:
1147
- * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
1148
- * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
1149
- * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
1150
- * table that would rot as router catalogs change. */
1151
- declare const HARNESS_NATIVE_MODEL = "default";
1152
- /**
1153
- * Expand a base profile across the harness × model matrix into the `AgentProfile[]`
1154
- * that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
1155
- * which models do we evaluate" lives, so no product hand-rolls its own harness list
1156
- * or column→profile mapping (the pattern that let those copies drift and silently
1157
- * break the harness pivot).
1158
- *
1159
- * Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,
1160
- * and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both
1161
- * metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row
1162
- * and results join back by harness/model via {@link harnessAxisOf} with no
1163
- * hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or
1164
- * its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every
1165
- * requested harness runs; `keepIncompatible` forces every pair verbatim.
1166
- *
1167
- * Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
1168
- * everything we care about" switch, identical in shape whether one harness or all.
1169
- */
1170
- declare function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[];
1171
- /**
1172
- * Read the (harness, model) a matrix cell ran under, off a profile or a result row's
1173
- * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
1174
- * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
1175
- * this instead of recomputing an id (recomputing the wrong key is what broke the pivot
1176
- * in the hand-rolled copies).
1177
- */
1178
- declare function harnessAxisOf(profile: Pick<AgentProfile, 'metadata'>): {
1179
- harness: HarnessType;
1180
- model: string;
1181
- } | undefined;
1182
- /**
1183
- * Collision-resistant, path-safe, human-readable profile id for eval artifacts.
1184
- * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
1185
- * keys, and directory names where two profiles must not collapse onto one row.
1186
- * The suffix is the first 64 bits of the behaviour hash, enough for ordinary
1187
- * eval matrices while keeping filenames readable.
1188
- */
1189
- declare function agentProfileId(profile: AgentProfile): string;
1190
- /**
1191
- * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete
1192
- * model id because run records reject bare/missing model aliases.
1193
- */
1194
- declare function agentProfileModelId(profile: AgentProfile): string;
1195
- /**
1196
- * Deterministic behaviour identity for the canonical
1197
- * `@tangle-network/agent-interface` AgentProfile.
1198
- *
1199
- * `name` and `description` are labels and do not affect the hash. Profile
1200
- * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
1201
- * and extensions do affect the hash. Resource array order is hash-bearing
1202
- * because mount order can change agent behaviour. Undefined fields are treated
1203
- * as absent; explicit `null` fields remain hash-bearing.
1204
- */
1205
- declare function agentProfileHash(profile: AgentProfile): string;
1206
- //#endregion
1207
573
  //#region src/integrity/backend-integrity.d.ts
1208
574
  interface BackendIntegrityReport {
1209
575
  /** Total records inspected. */
@@ -2145,5 +1511,5 @@ declare function verifyCodeSurface(surface: CodeSurface, worktreeDir?: string):
2145
1511
  * identity against the checkout at `worktreeRef`. */
2146
1512
  declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
2147
1513
  //#endregion
2148
- export { SearchPlannedEvent as $, EvalFixtureScenario as $n, pairRunRecords as $r, HarnessType$1 as $t, SearchAccountingAudit as A, SequentialDecideFn as An, CrossSurfaceInteractionTask as Ar, UserStory as At, SearchCostAccounting as B, SignedManifest as Bn, CrossSurfaceTaskRow as Br, RunProfileMatrixOptions as Bt, surfaceHash as C, HeldoutSignificance as Cn, CrossSurfaceEligibility as Cr, tangleTracesRoot as Ct, FileSearchLedger as D, dimensionRegressions as Dn, CrossSurfaceInteractionEffect as Dr, ScoreboardRenderOptions as Dt, acquireSingleRunLock as E, detectScale as En, CrossSurfaceInteractionAwareSelection as Er, PlaybackStep as Et, SearchCandidateRegisteredEvent as F, SequentialPairedGateOptions as Fn, CrossSurfacePairwiseEntry as Fr, scoreboardSummary as Ft, SearchLedgerEvent as G, signManifest as Gn, PairArmsResult as Gr, BackendIntegrityReport as Gt, SearchLedger as H, canonicalize as Hn, MatchedPair as Hr, ScenarioRollup as Ht, SearchCandidateSlot as I, sequentialDecide as In, CrossSurfaceRankedSingle as Ir, userStoryScoreboard as It, SearchLedgerTrustedHeadMode as J, neutralizationGate as Jn, PairedArmsComparison as Jr, summarizeAgentReceiptIntegrity as Jt, SearchLedgerHash as K, verifyManifest as Kn, PairRunRecordsResult as Kr, assertRealAgentReceipts as Kt, SearchCandidateSlotClosedEvent as L, sequentialPairedGate as Ln, CrossSurfaceRelativeCost as Lr, ProfileDispatchFn as Lt, SearchAttemptAccounting as M, SequentialDecision as Mn, CrossSurfacePairCompatibility as Mr, makePlaybackDispatch as Mt, SearchCandidateDecidedEvent as N, SequentialObservation as Nn, CrossSurfacePairEvidence as Nr, renderScoreboardMarkdown as Nt, OpenSearchLedgerOptions as O, heldoutSignificance as On, CrossSurfaceInteractionPath as Or, ScoreboardRow as Ot, SearchCandidateLineage as P, SequentialPairedGate as Pn, CrossSurfacePairIncompatibilityReason as Pr, scoreUserStory as Pt, SearchPlan as Q, EvalFixtureRunPlan as Qn, pairArms as Qr, HARNESS_NATIVE_MODEL as Qt, SearchCandidateSurface as R, HypothesisManifest as Rn, CrossSurfaceSelectionPolicy as Rr, ProfileMatrixError as Rt, surfaceContentHash as S, DimensionRegression as Sn, CrossSurfaceDistribution as Sr, resolveRunDir as St, SingleRunLockOptions as T, PairedHoldout as Tn, CrossSurfaceIneligibilityReason as Tr, PlaybackDriver as Tt, SearchLedgerAppendResult as U, evaluateHypothesis as Un, MatchedRunRecordPair as Ur, runProfileMatrix as Ut, SearchFailureReason as V, SignedManifestAlgo as Vn, ComparePairedArmsOptions as Vr, RunProfileMatrixResult as Vt, SearchLedgerEntry as W, hashJson as Wn, PairArmsOptions as Wr, BackendIntegrityError as Wt, SearchOperationKind as X, EvalFixtureFile as Xn, PairedMetricDelta as Xr, AgentProfile$1 as Xt, SearchModelIdentity as Y, EvalFixture as Yn, PairedCorrectness as Yr, summarizeBackendIntegrity as Yt, SearchOperationRecordedEvent as Z, EvalFixtureLoadOptions as Zn, comparePairedArms as Zr, CODING_HARNESSES as Zt, assertCodeSurfaceIdentity as _, RolloutCall as _n, CrossSurfaceCandidateOutcome as _r, compareRankKeys as _t, WorktreeAdapterError as a, harnessAxisOf as an, loadEvalFixtureScenarios as ar, SearchSurfaceKind as at, componentSurfaceIdentityMaterial as b, classifyUngroundedLiterals as bn, CrossSurfaceComponentEvidence as br, scoreDiscrimination as bt, verifyCodeSurface as c, RuntimeEventLike as cn, AnalyzeCrossSurfaceInteractionsInput as cr, SearchTokenAccounting as ct, PhoenixEvaluationResultLike as d, neutralizeText as dn, CrossSurfaceAttemptCompleteness as dr, SearchLedgerConflictError as dt, BuildTraceAnalystSurfaceDispatchOptions as ei, ProfileAxisSpec as en, EvalFixtureValidationMode as er, SearchPlannedOperation as et, PhoenixEvaluatorLike as f, FsLabeledScenarioStore as fn, CrossSurfaceBestSingleSelection as fr, SearchLedgerError as ft, isTransientTransportFailure as g, RolloutArgumentDiffOptions as gn, CrossSurfaceCandidateEvidence as gr, campaignMeanComposite as gt, TransientFailureOptions as h, RolloutArgumentDiff as hn, CrossSurfaceCandidateComparison as hr, campaignBreakdown as ht, WorktreeAdapter as i, traceAnalystQualityJudge as ii, expandProfileAxes as in, loadEvalFixture as ir, SearchSurfaceEvidence as it, SearchArtifactRef as j, SequentialDecideOptions as jn, CrossSurfaceNaiveStackSelection as jr, UserStoryVerdict as jt, SEARCH_LEDGER_SCHEMA as k, pairHoldout as kn, CrossSurfaceInteractionReport as kr, ScoreboardSummary as kt, AutoevalsScoreLike as l, ToolCallEventLike as ln, CrossSurfaceAdditionDecision as lr, openSearchLedger as lt, phoenixEvaluatorJudge as m, LabeledScenarioStoreError as mn, CrossSurfaceCandidate as mr, CampaignBreakdown as mt, GitWorktreeAdapterOptions as n, TraceAnalystScenario as ni, agentProfileId as nn, PlanEvalFixtureRunOptions as nr, SearchSourceRef as nt, gitWorktreeAdapter as o, ArtifactEventLike as on, planEvalFixtureRun as or, SearchTaskAttemptedEvent as ot, autoevalsScorerJudge as p, FsLabeledScenarioStoreOptions as pn, CrossSurfaceBootstrapPolicy as pr, SearchLedgerIntegrityError as pt, SearchLedgerReplay as q, NeutralizationGateOptions as qn, PairedArmRow as qr, assertRealBackend as qt, Worktree as r, buildTraceAnalystSurfaceDispatch as ri, agentProfileModelId as rn, discoverEvalFixtures as rr, SearchSurfaceEffect as rt, resolveWorktreePath as s, ProposalEventLike as sn, analyzeCrossSurfaceInteractions as sr, SearchTaskOutcome as st, CodeSurfaceVerification as t, TraceAnalystArtifact as ti, agentProfileHash as tn, LoadEvalFixtureScenariosOptions as tr, SearchPlannedTask as tt, AutoevalsScorerLike as u, extractProducedState as un, CrossSurfaceAdditionRejectionReason as ur, validateSearchLedgerEvent as ut, assertComponentSurface as v, ScoredRollout as vn, CrossSurfaceCandidateSummary as vr, DiscriminationScore as vt, SingleRunLock as w, HeldoutSignificanceOptions as wn, CrossSurfaceEvidenceBreakdown as wr, PlaybackContext as wt, renderSurfaceDiff as x, rolloutArgumentDiff as xn, CrossSurfaceCompositionStep as xr, selectDiscriminative as xt, codeSurfaceIdentityMaterial as y, UngroundedLiteralReport as yn, CrossSurfaceComponent as yr, ScenarioSignal as yt, SearchCompletedEvent as z, HypothesisResult as zn, CrossSurfaceSelections as zr, ProfileSummary as zt };
2149
- //# sourceMappingURL=index-CpxZSlB7.d.ts.map
1514
+ export { SearchPlannedEvent as $, CrossSurfacePairIncompatibilityReason as $n, ToolCallEventLike as $t, SearchAccountingAudit as A, CrossSurfaceAttemptCompleteness as An, UserStory as At, SearchCostAccounting as B, CrossSurfaceCompositionStep as Bn, RunProfileMatrixOptions as Bt, surfaceHash as C, loadEvalFixture as Cn, tangleTracesRoot as Ct, FileSearchLedger as D, AnalyzeCrossSurfaceInteractionsInput as Dn, ScoreboardRenderOptions as Dt, acquireSingleRunLock as E, analyzeCrossSurfaceInteractions as En, PlaybackStep as Et, SearchCandidateRegisteredEvent as F, CrossSurfaceCandidateEvidence as Fn, scoreboardSummary as Ft, SearchLedgerEvent as G, CrossSurfaceInteractionAwareSelection as Gn, BackendIntegrityReport as Gt, SearchLedger as H, CrossSurfaceEligibility as Hn, ScenarioRollup as Ht, SearchCandidateSlot as I, CrossSurfaceCandidateOutcome as In, userStoryScoreboard as It, SearchLedgerTrustedHeadMode as J, CrossSurfaceInteractionReport as Jn, summarizeAgentReceiptIntegrity as Jt, SearchLedgerHash as K, CrossSurfaceInteractionEffect as Kn, assertRealAgentReceipts as Kt, SearchCandidateSlotClosedEvent as L, CrossSurfaceCandidateSummary as Ln, ProfileDispatchFn as Lt, SearchAttemptAccounting as M, CrossSurfaceBootstrapPolicy as Mn, makePlaybackDispatch as Mt, SearchCandidateDecidedEvent as N, CrossSurfaceCandidate as Nn, renderScoreboardMarkdown as Nt, OpenSearchLedgerOptions as O, CrossSurfaceAdditionDecision as On, ScoreboardRow as Ot, SearchCandidateLineage as P, CrossSurfaceCandidateComparison as Pn, scoreUserStory as Pt, SearchPlan as Q, CrossSurfacePairEvidence as Qn, RuntimeEventLike as Qt, SearchCandidateSurface as R, CrossSurfaceComponent as Rn, ProfileMatrixError as Rt, surfaceContentHash as S, discoverEvalFixtures as Sn, resolveRunDir as St, SingleRunLockOptions as T, planEvalFixtureRun as Tn, PlaybackDriver as Tt, SearchLedgerAppendResult as U, CrossSurfaceEvidenceBreakdown as Un, runProfileMatrix as Ut, SearchFailureReason as V, CrossSurfaceDistribution as Vn, RunProfileMatrixResult as Vt, SearchLedgerEntry as W, CrossSurfaceIneligibilityReason as Wn, BackendIntegrityError as Wt, SearchOperationKind as X, CrossSurfaceNaiveStackSelection as Xn, ArtifactEventLike as Xt, SearchModelIdentity as Y, CrossSurfaceInteractionTask as Yn, summarizeBackendIntegrity as Yt, SearchOperationRecordedEvent as Z, CrossSurfacePairCompatibility as Zn, ProposalEventLike as Zt, assertCodeSurfaceIdentity as _, EvalFixtureRunPlan as _n, compareRankKeys as _t, WorktreeAdapterError as a, RolloutArgumentDiff as an, CrossSurfaceTaskRow as ar, SearchSurfaceKind as at, componentSurfaceIdentityMaterial as b, LoadEvalFixtureScenariosOptions as bn, scoreDiscrimination as bt, verifyCodeSurface as c, ScoredRollout as cn, TraceAnalystScenario as cr, SearchTokenAccounting as ct, PhoenixEvaluationResultLike as d, rolloutArgumentDiff as dn, SearchLedgerConflictError as dt, extractProducedState as en, CrossSurfacePairwiseEntry as er, SearchPlannedOperation as et, PhoenixEvaluatorLike as f, NeutralizationGateOptions as fn, SearchLedgerError as ft, isTransientTransportFailure as g, EvalFixtureLoadOptions as gn, campaignMeanComposite as gt, TransientFailureOptions as h, EvalFixtureFile as hn, campaignBreakdown as ht, WorktreeAdapter as i, LabeledScenarioStoreError as in, CrossSurfaceSelections as ir, SearchSurfaceEvidence as it, SearchArtifactRef as j, CrossSurfaceBestSingleSelection as jn, UserStoryVerdict as jt, SEARCH_LEDGER_SCHEMA as k, CrossSurfaceAdditionRejectionReason as kn, ScoreboardSummary as kt, AutoevalsScoreLike as l, UngroundedLiteralReport as ln, buildTraceAnalystSurfaceDispatch as lr, openSearchLedger as lt, phoenixEvaluatorJudge as m, EvalFixture as mn, CampaignBreakdown as mt, GitWorktreeAdapterOptions as n, FsLabeledScenarioStore as nn, CrossSurfaceRelativeCost as nr, SearchSourceRef as nt, gitWorktreeAdapter as o, RolloutArgumentDiffOptions as on, BuildTraceAnalystSurfaceDispatchOptions as or, SearchTaskAttemptedEvent as ot, autoevalsScorerJudge as p, neutralizationGate as pn, SearchLedgerIntegrityError as pt, SearchLedgerReplay as q, CrossSurfaceInteractionPath as qn, assertRealBackend as qt, Worktree as r, FsLabeledScenarioStoreOptions as rn, CrossSurfaceSelectionPolicy as rr, SearchSurfaceEffect as rt, resolveWorktreePath as s, RolloutCall as sn, TraceAnalystArtifact as sr, SearchTaskOutcome as st, CodeSurfaceVerification as t, neutralizeText as tn, CrossSurfaceRankedSingle as tr, SearchPlannedTask as tt, AutoevalsScorerLike as u, classifyUngroundedLiterals as un, traceAnalystQualityJudge as ur, validateSearchLedgerEvent as ut, assertComponentSurface as v, EvalFixtureScenario as vn, DiscriminationScore as vt, SingleRunLock as w, loadEvalFixtureScenarios as wn, PlaybackContext as wt, renderSurfaceDiff as x, PlanEvalFixtureRunOptions as xn, selectDiscriminative as xt, codeSurfaceIdentityMaterial as y, EvalFixtureValidationMode as yn, ScenarioSignal as yt, SearchCompletedEvent as z, CrossSurfaceComponentEvidence as zn, ProfileSummary as zt };
1515
+ //# sourceMappingURL=index-Sh2I0DRc.d.ts.map