@tangle-network/agent-eval 0.144.6 → 0.144.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (240) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/analyst/index.d.ts +470 -88
  6. package/dist/analyst/index.d.ts.map +1 -1
  7. package/dist/analyst/index.js +24 -5
  8. package/dist/analyst/index.js.map +1 -1
  9. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  10. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  11. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  12. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  13. package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BKENp2s5.js} +662 -605
  14. package/dist/benchmark-command-BKENp2s5.js.map +1 -0
  15. package/dist/benchmarks/index.d.ts +1 -1
  16. package/dist/benchmarks/index.js +1 -1
  17. package/dist/{benchmarks-BEOkuvIg.js → benchmarks-BlPmjd88.js} +4 -4
  18. package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-BlPmjd88.js.map} +1 -1
  19. package/dist/campaign/index.d.ts +7 -5
  20. package/dist/campaign/index.js +5 -3
  21. package/dist/{campaign-CXsdyym7.js → campaign--HVSuvV0.js} +17 -301
  22. package/dist/campaign--HVSuvV0.js.map +1 -0
  23. package/dist/cli.js +2 -2
  24. package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
  25. package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  26. package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
  27. package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  28. package/dist/contract/index.d.ts +9 -8
  29. package/dist/contract/index.d.ts.map +1 -1
  30. package/dist/contract/index.js +7 -6
  31. package/dist/contract/index.js.map +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/control.js +1 -1
  34. package/dist/counterfactual-CWPTrMH7.js +126 -0
  35. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  36. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  37. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  38. package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  39. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  40. package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
  41. package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  42. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  43. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  44. package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
  45. package/dist/engine-nB64f48I.d.ts.map +1 -0
  46. package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
  47. package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  48. package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  49. package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  50. package/dist/exec-BLtYZdWo.js +49 -0
  51. package/dist/exec-BLtYZdWo.js.map +1 -0
  52. package/dist/experiment/index.d.ts +802 -0
  53. package/dist/experiment/index.d.ts.map +1 -0
  54. package/dist/experiment/index.js +1108 -0
  55. package/dist/experiment/index.js.map +1 -0
  56. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  57. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  58. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  59. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  60. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
  61. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  62. package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
  63. package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  64. package/dist/fuzz.d.ts +1 -1
  65. package/dist/hosted/index.d.ts +3 -3
  66. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  67. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  68. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  69. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  70. package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
  71. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  72. package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
  73. package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  74. package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
  75. package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  76. package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
  77. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  78. package/dist/index.d.ts +214 -404
  79. package/dist/index.d.ts.map +1 -1
  80. package/dist/index.js +233 -649
  81. package/dist/index.js.map +1 -1
  82. package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
  83. package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  84. package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
  85. package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  86. package/dist/integrity-MLzHOfV9.js +141 -0
  87. package/dist/integrity-MLzHOfV9.js.map +1 -0
  88. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  89. package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
  90. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  91. package/dist/matrix/index.d.ts +2 -2
  92. package/dist/meta-eval/index.d.ts +1 -1
  93. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  94. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  95. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  96. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  97. package/dist/multishot/index.d.ts +2 -2
  98. package/dist/openapi.json +1 -1
  99. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  100. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  101. package/dist/pipelines/index.d.ts +2 -1
  102. package/dist/pipelines/index.d.ts.map +1 -1
  103. package/dist/pre-registration-DakwTRXk.js +96 -0
  104. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  105. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  106. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  107. package/dist/prime-protocol-BfSalTfR.js +453 -0
  108. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  109. package/dist/profile-cell.js +242 -1
  110. package/dist/profile-cell.js.map +1 -0
  111. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  112. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  113. package/dist/promotion-policy-CrLrmys8.js +682 -0
  114. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  115. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  116. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  117. package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
  118. package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  119. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  120. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  121. package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
  122. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  123. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  124. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  125. package/dist/reporting.d.ts +3 -3
  126. package/dist/reporting.js +2 -2
  127. package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
  128. package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  130. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  131. package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
  132. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  133. package/dist/rl.d.ts +17 -7
  134. package/dist/rl.d.ts.map +1 -1
  135. package/dist/rl.js +16 -7
  136. package/dist/rl.js.map +1 -1
  137. package/dist/rollout/index.d.ts +2 -2
  138. package/dist/rollout/index.js +2 -2
  139. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  140. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  141. package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
  142. package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  143. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  144. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  145. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  146. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  147. package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  148. package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  149. package/dist/sequential-D-BLJBKU.js +299 -0
  150. package/dist/sequential-D-BLJBKU.js.map +1 -0
  151. package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
  152. package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
  153. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  154. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  155. package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
  156. package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  157. package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
  158. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  159. package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CkaI2ly4.js} +19 -654
  160. package/dist/skillopt-optimization-method-CkaI2ly4.js.map +1 -0
  161. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  162. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  163. package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  164. package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  165. package/dist/steps-BArUxhna.d.ts +51 -0
  166. package/dist/steps-BArUxhna.d.ts.map +1 -0
  167. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  168. package/dist/store-DNe_Uv1Q.js.map +1 -0
  169. package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
  170. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  171. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  172. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  173. package/dist/supervisor-run/index.d.ts +2 -2
  174. package/dist/supervisor-run/index.js +1 -1
  175. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  176. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  177. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  178. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  179. package/dist/trace-repair/index.d.ts +2102 -0
  180. package/dist/trace-repair/index.d.ts.map +1 -0
  181. package/dist/trace-repair/index.js +3878 -0
  182. package/dist/trace-repair/index.js.map +1 -0
  183. package/dist/traces.d.ts +6 -6
  184. package/dist/traces.js +3 -2
  185. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  186. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  187. package/dist/trajectory-replay/index.d.ts +781 -0
  188. package/dist/trajectory-replay/index.d.ts.map +1 -0
  189. package/dist/trajectory-replay/index.js +2103 -0
  190. package/dist/trajectory-replay/index.js.map +1 -0
  191. package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
  192. package/dist/types-D216SgwM.d.ts.map +1 -0
  193. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  194. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  195. package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
  196. package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  197. package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
  198. package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  199. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  200. package/dist/verdict-DExhxfgR.d.ts +201 -0
  201. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  202. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  203. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  204. package/dist/wire/index.d.ts +2 -2
  205. package/dist/wire/index.js +1 -1
  206. package/docs/campaign-proposers.md +5 -0
  207. package/docs/charter.md +112 -0
  208. package/docs/experiment.md +104 -0
  209. package/docs/prime-analyst.md +1 -0
  210. package/docs/trace-analysis.md +26 -0
  211. package/docs/trace-repair-admission.md +194 -0
  212. package/docs/trace-repair-analyst-arms.md +121 -0
  213. package/docs/trace-repair-continuation.md +107 -0
  214. package/docs/trace-repair-grader.md +163 -0
  215. package/docs/trajectory-replay.md +110 -0
  216. package/docs/verification-strategies.md +103 -0
  217. package/package.json +19 -2
  218. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  219. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  220. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  221. package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
  222. package/dist/benchmark-command-CQd78YHt.js.map +0 -1
  223. package/dist/campaign-CXsdyym7.js.map +0 -1
  224. package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
  225. package/dist/index-4XwggC10.d.ts.map +0 -1
  226. package/dist/index-BIL5vxxt.d.ts.map +0 -1
  227. package/dist/integrity-fdt8XPAv.js.map +0 -1
  228. package/dist/llm-client-Dv5BiKLE.js.map +0 -1
  229. package/dist/replay-Krvb114g.d.ts.map +0 -1
  230. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  231. package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
  232. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  233. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  234. package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
  235. package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
  236. package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
  237. package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
  238. package/dist/types-XMVEdrE_.d.ts.map +0 -1
  239. package/dist/verdict-Dps8_okt.d.ts +0 -37
  240. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -0,0 +1,555 @@
1
+ import { a as RunRecord } from "./run-record-DdSa93_W.js";
2
+ import { R as Scenario, b as GenerationRecord, p as Gate, w as JudgeScore } from "./types-DYuNHo9R.js";
3
+ import { E as RiskDifferenceResult, _ as McNemarResult, d as EProcessState, v as PairedBootstrapOptions, y as PairedBootstrapResult } from "./statistics-D6Uebe_4.js";
4
+ import { a as PairedPromotionDecision, i as PairedMcNemarEvidence, r as PairedDecisionStatistic, t as PairedDecisionMethod } from "./paired-promotion-decision-B6zJ3gYM.js";
5
+ //#region src/paired-arms.d.ts
6
+ /** One arm observation of one work item. Structural on purpose: callers
7
+ * project their own record type (e.g. a `RunRecord`) into this shape. */
8
+ interface PairedArmRow {
9
+ /** Matching key — rows sharing a `pairKey` across both arms form pairs
10
+ * (typically the task/scenario/seed identity). */
11
+ pairKey: string;
12
+ /** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
13
+ * every row of a `pairKey` that has more than one rep in either arm; reps
14
+ * then pair only on exact (`pairKey`, `repKey`) match, never on outcome
15
+ * content. Optional when each arm has at most one rep of the item. */
16
+ repKey?: string;
17
+ /** Arm label this row was produced under. */
18
+ arm: string;
19
+ /** Binary outcome; omit when the comparison has no pass/fail notion. */
20
+ pass?: boolean;
21
+ /** Named numeric measurements (score, cost, latency, …). */
22
+ metrics?: Record<string, number>;
23
+ }
24
+ interface PairArmsOptions {
25
+ /** Arm treated as the control side of every pair. */
26
+ baselineArm: string;
27
+ /** Arm treated as the treatment side of every pair. */
28
+ treatmentArm: string;
29
+ }
30
+ /** One matched (baseline, treatment) observation of the same work item. */
31
+ interface MatchedPair {
32
+ pairKey: string;
33
+ /** 0-based position of this pair within its `pairKey`, ordered by sorted
34
+ * `repKey` (always 0 for a single-rep item). The rep identity itself is on
35
+ * the rows (`baseline.repKey` / `treatment.repKey`). */
36
+ repIndex: number;
37
+ baseline: PairedArmRow;
38
+ treatment: PairedArmRow;
39
+ }
40
+ interface PairArmsResult {
41
+ /** Matched pairs, ordered by (`pairKey`, `repIndex`). */
42
+ pairs: MatchedPair[];
43
+ /** Baseline rows left without a treatment counterpart — reported, never
44
+ * silently dropped. */
45
+ unpairedBaseline: PairedArmRow[];
46
+ /** Treatment rows left without a baseline counterpart. */
47
+ unpairedTreatment: PairedArmRow[];
48
+ }
49
+ /**
50
+ * Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
51
+ *
52
+ * A `pairKey` with at most one row per arm pairs directly, no `repKey`
53
+ * needed. A `pairKey` with multiple reps in either arm requires `repKey` on
54
+ * every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
55
+ * match — pairing is keyed purely on row identity, never on outcome content
56
+ * (outcome-keyed matching deflates discordant counts and biases McNemar), and
57
+ * is therefore independent of input order. Reps whose `repKey` has no
58
+ * counterpart in the other arm, and items present in only one arm, land in
59
+ * the unpaired lists — reported, never truncated.
60
+ *
61
+ * Fail-loud: throws when either named arm has zero rows (an unknown arm
62
+ * name would otherwise read as "everything unpaired"), when the two arm
63
+ * names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
64
+ * when a (`pairKey`, arm) group repeats a `repKey` (the match would be
65
+ * ambiguous).
66
+ */
67
+ declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
68
+ /** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
69
+ interface PairedCorrectness {
70
+ /** Discordant pairs where the treatment passed and the baseline failed. */
71
+ b10: number;
72
+ /** Discordant pairs where the baseline passed and the treatment failed. */
73
+ b01: number;
74
+ /** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
75
+ mcnemar: McNemarResult;
76
+ /** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
77
+ riskDifference: RiskDifferenceResult;
78
+ }
79
+ /** Paired delta summary for one named metric (delta = treatment − baseline). */
80
+ interface PairedMetricDelta {
81
+ name: string;
82
+ /** Pairs where BOTH sides carry a finite value for this metric. */
83
+ n: number;
84
+ /** Pairs where at least one side does not carry the metric. */
85
+ nMissing: number;
86
+ /** Median paired delta, or null when `n === 0`. */
87
+ medianDelta: number | null;
88
+ /** Mean paired delta, or null when `n === 0`. */
89
+ meanDelta: number | null;
90
+ /** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
91
+ * `n === 0` — a zero-width [0, 0] interval on no data would read as a
92
+ * measured tight null. */
93
+ bootstrapCi: PairedBootstrapResult | null;
94
+ /** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
95
+ wilcoxon: {
96
+ w: number;
97
+ p: number;
98
+ } | null;
99
+ }
100
+ interface ComparePairedArmsOptions extends PairArmsOptions {
101
+ /** Metrics to compare. Default: every metric name observed on any matched
102
+ * pair, sorted. A name that appears on no pair is still reported (with
103
+ * `n = 0`) so a misspelled metric is visible instead of vanishing. */
104
+ metricNames?: string[];
105
+ /** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
106
+ bootstrap?: PairedBootstrapOptions;
107
+ }
108
+ interface PairedArmsComparison {
109
+ nPairs: number;
110
+ nUnpairedBaseline: number;
111
+ nUnpairedTreatment: number;
112
+ /** null when no matched pair carries `pass` on both sides — a pass/fail
113
+ * verdict over rows that never measured pass/fail would be fabricated. */
114
+ correctness: PairedCorrectness | null;
115
+ metricDeltas: PairedMetricDelta[];
116
+ }
117
+ /**
118
+ * Full matched-pair arm comparison: pair via {@link pairArms}, then compose
119
+ * the paired estimators from `statistics` over the matched pairs.
120
+ *
121
+ * Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
122
+ * is that subset's size); each metric uses only the pairs where both sides
123
+ * carry a finite value for it, with the remainder counted in `nMissing`.
124
+ * Deltas are treatment − baseline throughout.
125
+ *
126
+ * Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
127
+ * non-finite metric value — silently treating corrupt telemetry as "metric
128
+ * absent" would misreport it as missing coverage.
129
+ */
130
+ declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
131
+ interface MatchedRunRecordPair {
132
+ pairKey: string;
133
+ repKey: string;
134
+ baseline: RunRecord;
135
+ treatment: RunRecord;
136
+ }
137
+ interface PairRunRecordsResult {
138
+ pairs: MatchedRunRecordPair[];
139
+ unpairedBaseline: RunRecord[];
140
+ unpairedTreatment: RunRecord[];
141
+ }
142
+ /**
143
+ * Pair two RunRecord arms by the identity of the evaluated work:
144
+ * `(experimentId, scenarioId, seed)`.
145
+ *
146
+ * Falling back to array order, candidate id, or experiment id can compare
147
+ * different tasks and fabricate lift. Duplicate identities throw.
148
+ */
149
+ declare function pairRunRecords(baselineRuns: readonly RunRecord[], treatmentRuns: readonly RunRecord[]): PairRunRecordsResult;
150
+ //#endregion
151
+ //#region src/pre-registration.d.ts
152
+ /**
153
+ * Pre-registered hypotheses — declare what you're testing BEFORE the
154
+ * run, check it AFTER. Prevents p-hacking, optional stopping, and the
155
+ * "we ran until it looked good" failure mode.
156
+ *
157
+ * Manifest is a plain JSON-friendly object. Sign it with a content hash
158
+ * + timestamp; the registered record becomes immutable. Post-run,
159
+ * evaluate the manifest against observed results — the library refuses
160
+ * to let you re-interpret a different metric as the declared one.
161
+ */
162
+ interface HypothesisManifest {
163
+ id: string;
164
+ /** Human prose — goes into the audit trail. */
165
+ hypothesis: string;
166
+ /** Metric the hypothesis claims to move. */
167
+ metric: string;
168
+ /** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
169
+ direction: 'increase' | 'decrease';
170
+ /** Minimum effect size to count (same units as the metric). */
171
+ minEffect: number;
172
+ /** Alpha threshold. */
173
+ alpha: number;
174
+ /** Target statistical power at which sample size was pre-computed. */
175
+ power: number;
176
+ /** Declared N per arm before running. */
177
+ preRegisteredN: number;
178
+ /** ISO8601 timestamp the manifest was registered. */
179
+ registeredAt: string;
180
+ /** Optional identifiers to tie into the trace corpus. */
181
+ baselineLabel?: string;
182
+ candidateLabel?: string;
183
+ }
184
+ /**
185
+ * Identifier for the hashing scheme used to produce `contentHash`.
186
+ *
187
+ * `'sha256-content'` — sha256 hex over the canonicalized manifest with
188
+ * the `contentHash` and `algo` fields stripped. Held as a string union
189
+ * so future schemes can be added without breaking parsers; SignedManifest
190
+ * values without `algo` deserialize cleanly because the field is optional.
191
+ */
192
+ type SignedManifestAlgo = 'sha256-content';
193
+ interface SignedManifest extends HypothesisManifest {
194
+ /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
195
+ contentHash: string;
196
+ /**
197
+ * Algorithm string describing how `contentHash` was produced.
198
+ *
199
+ * Optional on the type so serialized manifests without it still parse,
200
+ * but ALWAYS populated by {@link signManifest}. Consumers that want to
201
+ * enforce a known algorithm should reject manifests where this field
202
+ * is missing or unrecognized.
203
+ */
204
+ algo?: SignedManifestAlgo;
205
+ }
206
+ interface HypothesisResult {
207
+ manifest: SignedManifest;
208
+ observedN: number;
209
+ observedEffect: number;
210
+ observedPValue: number;
211
+ /** True iff the observed effect hits the pre-declared direction with
212
+ * magnitude ≥ minEffect AND p < alpha. */
213
+ confirmed: boolean;
214
+ /** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
215
+ rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
216
+ notes?: string;
217
+ }
218
+ /**
219
+ * Deterministic JSON canonicalization — sort object keys recursively.
220
+ *
221
+ * Two semantically-equal objects produce byte-identical canonicalized output;
222
+ * this is what makes a content-hash stable across encoders, key insertion
223
+ * orders, and runtime versions. Exported for any consumer that needs the same
224
+ * canonicalization guarantee outside the manifest-signing path (e.g., signing
225
+ * an artifact bundle, hashing a dataset version, etc.).
226
+ */
227
+ declare function canonicalize(v: unknown): unknown;
228
+ /**
229
+ * SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.
230
+ *
231
+ * The same primitive `signManifest` and `verifyManifest` are built on, exposed
232
+ * directly so consumers signing arbitrary structured content (artifact bundles,
233
+ * production packets, dataset manifests, etc.) don't have to re-derive
234
+ * canonicalize+sha256 from scratch.
235
+ *
236
+ * Stable across:
237
+ * - object key insertion order (canonicalization sorts keys recursively)
238
+ * - encoder choice (UTF-8 via TextEncoder, fixed)
239
+ * - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)
240
+ *
241
+ * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
242
+ * which takes a string input and returns a truncated 12-char prompt id.
243
+ * Use `hashJson` when you mean "canonicalize then hash."
244
+ *
245
+ * @example
246
+ * const hash = await hashJson({ id: '1', kind: 'spec' })
247
+ * // 'a3f1...' (64 hex chars)
248
+ */
249
+ declare function hashJson<T>(obj: T): Promise<string>;
250
+ /**
251
+ * Sign a manifest with a SHA-256 content hash.
252
+ *
253
+ * The hash covers the canonicalized manifest with the `contentHash`
254
+ * and `algo` fields stripped; this lets verifiers re-sign the rest and
255
+ * compare. Returned manifest always carries `algo: 'sha256-content'`
256
+ * so downstream consumers can identify the scheme; manifests without
257
+ * `algo` still verify because it is stripped before hashing on both sides.
258
+ */
259
+ declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
260
+ /**
261
+ * Verify that a signed manifest has not been tampered with.
262
+ *
263
+ * Strips `contentHash` and `algo` before re-signing so manifests without
264
+ * `algo` verify identically to ones that carry it.
265
+ */
266
+ declare function verifyManifest(m: SignedManifest): Promise<boolean>;
267
+ /**
268
+ * Evaluate a pre-registered hypothesis against observed results.
269
+ * Mechanical — no re-interpretation permitted.
270
+ */
271
+ declare function evaluateHypothesis(manifest: SignedManifest, observed: {
272
+ n: number;
273
+ effect: number;
274
+ pValue: number;
275
+ }): Promise<HypothesisResult>;
276
+ //#endregion
277
+ //#region src/campaign/gates/sequential.d.ts
278
+ type SequentialDecision = 'promote' | 'continue' | 'undecided-at-maxN';
279
+ interface SequentialObservation {
280
+ decision: SequentialDecision;
281
+ /** Current e-value (the betting wealth) against H0. */
282
+ eValue: number;
283
+ /** Paired deltas consumed so far. */
284
+ n: number;
285
+ /** Names the decision basis. For 'undecided-at-maxN' it states explicitly
286
+ * that exhausting the budget is NOT evidence of no effect. */
287
+ reason: string;
288
+ }
289
+ interface SequentialPairedGateOptions {
290
+ /** Type-I budget. With `preRegistration` bound this MUST match
291
+ * `manifest.alpha` (conflict throws). Default 0.05. */
292
+ alpha?: number;
293
+ /** Minimum paired deltas before a promote may fire. The stopping rule is
294
+ * "first n ≥ minN with e-value ≥ 1/alpha" — still a valid stopping time.
295
+ * Default 5. */
296
+ minN?: number;
297
+ /** Pre-registered observation budget. Required unless `preRegistration`
298
+ * supplies it via `preRegisteredN` (conflict throws). */
299
+ maxN?: number;
300
+ /** Bet truncation forwarded to `eProcess`. Default 0.5. */
301
+ maxBet?: number;
302
+ /** Bound on |delta| in the judge's native scale; deltas are mapped to
303
+ * x = (d/scale + 1)/2 ∈ [0,1]. A delta outside ±scale throws (use
304
+ * `detectScale` to pick 1 vs 100 BEFORE streaming). Default 1. */
305
+ scale?: number;
306
+ /** Seed for the data-independent shuffle of paired deltas in `decide(ctx)`
307
+ * (exchangeability guard). Default 1337. */
308
+ shuffleSeed?: number;
309
+ /** Bind the pre-registered hypothesis. Verified (content hash) at
310
+ * construction; alpha/maxN/direction/minEffect come FROM the manifest. */
311
+ preRegistration?: SignedManifest;
312
+ /** Override the gate name in reports. */
313
+ name?: string;
314
+ }
315
+ interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario> extends Gate<TArtifact, TScenario> {
316
+ /** Streaming entry point: feed one paired per-scenario delta
317
+ * (candidate − baseline, native scale). Each gate instance carries ONE
318
+ * observe-stream; `decide(ctx)` runs on its own fresh stream and never
319
+ * consumes or advances this one. 'promote' is sticky; observing past the
320
+ * pre-registered maxN throws (extending a finished stream after seeing
321
+ * the result reopens optional stopping — start a NEW pre-registered
322
+ * test). */
323
+ observe(delta: number): SequentialObservation;
324
+ /** Read-only snapshot of the observe-stream. */
325
+ state(): EProcessState & {
326
+ decision: SequentialDecision;
327
+ };
328
+ }
329
+ /**
330
+ * Anytime-valid sequential paired gate. Conforms to the existing `Gate`
331
+ * contract (`decide(ctx)` consumes candidate vs baseline judge scores via
332
+ * `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
333
+ * never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
334
+ * that score cells incrementally and want to stop mid-stream.
335
+ *
336
+ * Decision mapping onto the substrate's five-valued `GateDecision`:
337
+ * - 'promote' → 'ship'
338
+ * - 'continue' → 'need_more_work' (stream ended before maxN with
339
+ * the e-value undecided — more reps could decide)
340
+ * - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
341
+ * evidence of no effect (never a silent default)
342
+ */
343
+ declare function sequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: SequentialPairedGateOptions): SequentialPairedGate<TArtifact, TScenario>;
344
+ interface SequentialDecideOptions {
345
+ /** Type-I budget for the early-stop evidence. Default 0.05. */
346
+ alpha?: number;
347
+ /** Minimum paired deltas before a stop may fire. Default 5. */
348
+ minN?: number;
349
+ /** Bet truncation forwarded to `eProcess`. Default 0.5. */
350
+ maxBet?: number;
351
+ /** Bound on |per-scenario composite delta|. Default 1. */
352
+ scale?: number;
353
+ }
354
+ interface SequentialDecideFn {
355
+ (args: {
356
+ history: GenerationRecord[];
357
+ }): {
358
+ stop: boolean;
359
+ reason?: string;
360
+ };
361
+ /** Read-only snapshot of the accumulated e-process (observability + tests). */
362
+ state(): EProcessState;
363
+ }
364
+ /**
365
+ * `SurfaceProposer.decide` adapter — stops the optimization loop the moment
366
+ * the e-process decides the loop has produced a real improvement, instead of
367
+ * always running `maxGenerations`.
368
+ *
369
+ * Stream: for each generation g ≥ 1, the per-scenario composite deltas of
370
+ * generation g's top candidate vs the generation-0 top candidate (the
371
+ * incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
372
+ * surface improves any scenario's expected composite over the incumbent —
373
+ * under it every delta has conditional mean ≤ 0 and the e-process is valid.
374
+ * Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
375
+ * gate (which re-scores on HELD-OUT data — this adapter only spends the
376
+ * exploration budget, it never promotes).
377
+ *
378
+ * Honesty caveats: (1) the incumbent's scores are measured once and shared
379
+ * across all generations' deltas, so type-I control is exact only insofar as
380
+ * those scores approximate the incumbent's true per-scenario means (more reps
381
+ * → tighter); (2) an UNDECIDED process never stops the loop — absence of a
382
+ * crossing is NOT evidence of no effect, so the loop simply runs its normal
383
+ * course. Calling the adapter repeatedly with a growing history consumes each
384
+ * generation exactly once (re-feeding an already-seen record would double-count
385
+ * evidence).
386
+ */
387
+ declare function sequentialDecide(options?: SequentialDecideOptions): SequentialDecideFn;
388
+ //#endregion
389
+ //#region src/campaign/gates/statistical-heldout.d.ts
390
+ interface PairedHoldout {
391
+ /** Baseline scalar per paired cell (same order as `after`/`cellIds`). */
392
+ before: number[];
393
+ /** Candidate scalar per paired cell. */
394
+ after: number[];
395
+ /** The full cellIds (`scenario:rep`) that paired, in order. */
396
+ cellIds: string[];
397
+ }
398
+ /**
399
+ * Pair candidate vs baseline holdout observations by FULL cellId. `select`
400
+ * pulls the scalar from a cell's judge reports (composite, or a named
401
+ * dimension); a cell contributes the mean of `select` across its judges. Cells
402
+ * whose scenario is not in `scenarioIds`, or where `select` is undefined for
403
+ * every judge on either side, are skipped on BOTH sides so the arrays stay
404
+ * paired. Throws when the two maps disagree on which holdout cells exist — a
405
+ * load-bearing invariant: the baseline + winner holdout campaigns run the same
406
+ * scenarios with the same seed base, so their cellIds MUST align; a mismatch
407
+ * means a silent pairing bug, not a soft fallback.
408
+ */
409
+ declare function pairHoldout(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, select: (s: JudgeScore) => number | undefined): PairedHoldout;
410
+ interface HeldoutSignificance {
411
+ paired: PairedHoldout;
412
+ /**
413
+ * The paired bootstrap on the requested statistic (MEAN by default — see the
414
+ * tie note on `heldoutSignificance`).
415
+ *
416
+ * DIAGNOSTIC, not necessarily the interval the verdict keyed on. On a
417
+ * two-point (pass/fail) outcome the decision routes to Tango's score interval
418
+ * instead, because a percentile bootstrap of the mean over a three-atom
419
+ * lattice is not a valid interval at a nonzero margin. Read
420
+ * `decision.low`/`decision.high` for the interval that actually decided, and
421
+ * `decisionStatistic` for which one it is.
422
+ */
423
+ bootstrap: PairedBootstrapResult;
424
+ /** The MEDIAN paired-delta bootstrap, reported as a diagnostic. When many
425
+ * scenarios are tied (both sides solve them), the median is pinned near 0
426
+ * regardless of the mean lift — comparing the two exposes tie-domination. */
427
+ medianBootstrap: PairedBootstrapResult;
428
+ /**
429
+ * The full promotion decision: which estimator the outcome's shape admits,
430
+ * the interval it produced, McNemar's exact veto on the two-point path, and
431
+ * whether the interval was zero-width (no evidence in either direction). The
432
+ * single source of `significant`.
433
+ */
434
+ decision: PairedPromotionDecision;
435
+ /** Which paired estimator the verdict was decided on. */
436
+ decisionStatistic: PairedDecisionStatistic;
437
+ /** McNemar's exact evidence on the two-point path; null otherwise. */
438
+ mcnemar: PairedMcNemarEvidence | null;
439
+ /** Fraction of paired observations that are exact ties (|delta| < 1e-9). A
440
+ * high tie fraction is WHY a median-based gate would have missed a real lift;
441
+ * it is the observability the tie fix adds. */
442
+ tieFraction: number;
443
+ /** n paired observations. */
444
+ n: number;
445
+ /** Effective minimum after applying the bootstrap's hard statistical floor. */
446
+ minimumRequired: number;
447
+ /** Statistical method that carried the decision. */
448
+ decisionMethod: PairedDecisionMethod;
449
+ /** Exact one-sided p-value on the small-sample path; otherwise null. */
450
+ pValue: number | null;
451
+ /** True iff n >= minimumRequired, the DECIDING interval has nonzero width,
452
+ * its lower bound clears the threshold, and McNemar's exact test does not
453
+ * veto at a non-negative threshold. */
454
+ significant: boolean;
455
+ /** Set when n < minimumRequired — too little evidence to claim significance. */
456
+ fewRuns: boolean;
457
+ }
458
+ interface HeldoutSignificanceOptions {
459
+ deltaThreshold?: number;
460
+ minProductiveRuns?: number;
461
+ confidence?: number;
462
+ resamples?: number;
463
+ /** Fixed by default for a deterministic, reproducible gate verdict. */
464
+ seed?: number;
465
+ statistic?: 'mean' | 'median';
466
+ }
467
+ /**
468
+ * Significance of the held-out composite lift: ship only when the lower bound
469
+ * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
470
+ * 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
471
+ * scale.
472
+ *
473
+ * The decision is delegated whole to {@link decidePairedPromotion}, the one
474
+ * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
475
+ * also calls. That module's header carries the measurements; the short version
476
+ * is three guards a bare `bootstrap.low > threshold` does not have:
477
+ *
478
+ * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
479
+ * only paired-binary construction that stays valid at a nonzero margin;
480
+ * - McNemar's exact test VETOES at any non-negative threshold;
481
+ * - a ZERO-WIDTH interval is refused rather than promoted, in either
482
+ * direction — [0,0] clears every negative threshold and [g,g] clears every
483
+ * threshold below g, and both are an absence of evidence, not a result.
484
+ *
485
+ * Measured on this function before those guards landed, at a nominal 5 %:
486
+ * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
487
+ * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
488
+ * delta is exactly 0.
489
+ *
490
+ * At small n, where the percentile bootstrap is descriptive only, a
491
+ * pre-registered exact sign test still carries the bootstrap path.
492
+ */
493
+ declare function heldoutSignificance(paired: PairedHoldout, opts?: HeldoutSignificanceOptions): HeldoutSignificance;
494
+ interface DimensionRegression {
495
+ dimension: string;
496
+ /** Paired bootstrap on (candidate − baseline). DIAGNOSTIC on a pass/fail
497
+ * dimension, where `ci` carries the interval that decided instead. */
498
+ bootstrap: PairedBootstrapResult;
499
+ /** Which paired statistic `bootstrap.low` is the lower bound of. `'mean'`
500
+ * unless the caller asked for the median. `bootstrap.median` still carries
501
+ * the median point estimate either way. */
502
+ bootstrapStatistic: 'median' | 'mean';
503
+ /** The interval `regressed` was decided on, in the dimension's native units. */
504
+ ci: {
505
+ low: number;
506
+ high: number;
507
+ };
508
+ /** Which estimator produced `ci`. */
509
+ decisionStatistic: PairedDecisionStatistic;
510
+ /** McNemar's exact evidence on a pass/fail dimension; null otherwise. */
511
+ mcnemar: PairedMcNemarEvidence | null;
512
+ /** `ci` has zero width — no evidence in either direction. */
513
+ indeterminate: boolean;
514
+ /** True iff the candidate may have regressed this dimension by more than
515
+ * tolerance: the lower bound of the DECIDING interval on (candidate −
516
+ * baseline) is below −tolerance, OR the exact small-sample test proves a drop
517
+ * past tolerance. */
518
+ regressed: boolean;
519
+ tolerance: number;
520
+ n: number;
521
+ }
522
+ /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
523
+ * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
524
+ * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
525
+ declare function detectScale(values: number[]): 1 | 100;
526
+ /** Per-critical-dimension regression guard. For each dimension, pair the
527
+ * candidate vs baseline values by full cellId and bootstrap the paired delta;
528
+ * a dimension is "regressed" when the CI lower bound < −tolerance (conservative
529
+ * — blocks if the credible worst case exceeds tolerance, which is the right
530
+ * posture for safety dimensions like `hallucination_free`). When `tolerance`
531
+ * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
532
+ *
533
+ * The interval comes from {@link decidePairedPromotion}, so a pass/fail
534
+ * dimension is judged on Tango's score interval rather than a percentile
535
+ * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
536
+ * is not a valid interval at one. That matters most here because this guard
537
+ * fails OPEN by construction: `tolerance` is positive, so an interval pinned at
538
+ * [0,0] never satisfies `low < −tolerance` and a real regression on a safety
539
+ * dimension would be reported as `regressed: false`. On the median it fails the
540
+ * same way for the same reason — when most pairs tie, which is automatic for a
541
+ * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
542
+ * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
543
+ * restore the pre-0.134 behaviour. */
544
+ declare function dimensionRegressions(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, criticalDimensions: string[], opts?: {
545
+ tolerance?: number;
546
+ confidence?: number;
547
+ resamples?: number;
548
+ seed?: number;
549
+ /** Paired statistic the CI is computed on. Default `'mean'` — see
550
+ * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
551
+ statistic?: 'mean' | 'median';
552
+ }): DimensionRegression[];
553
+ //#endregion
554
+ export { PairArmsResult as A, hashJson as C, MatchedPair as D, ComparePairedArmsOptions as E, PairedMetricDelta as F, comparePairedArms as I, pairArms as L, PairedArmRow as M, PairedArmsComparison as N, MatchedRunRecordPair as O, PairedCorrectness as P, pairRunRecords as R, evaluateHypothesis as S, verifyManifest as T, HypothesisManifest as _, detectScale as a, SignedManifestAlgo as b, pairHoldout as c, SequentialDecision as d, SequentialObservation as f, sequentialPairedGate as g, sequentialDecide as h, PairedHoldout as i, PairRunRecordsResult as j, PairArmsOptions as k, SequentialDecideFn as l, SequentialPairedGateOptions as m, HeldoutSignificance as n, dimensionRegressions as o, SequentialPairedGate as p, HeldoutSignificanceOptions as r, heldoutSignificance as s, DimensionRegression as t, SequentialDecideOptions as u, HypothesisResult as v, signManifest as w, canonicalize as x, SignedManifest as y };
555
+ //# sourceMappingURL=statistical-heldout-Dn9ruizm.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"statistical-heldout-Dn9ruizm.d.ts","names":[],"sources":["../src/paired-arms.ts","../src/pre-registration.ts","../src/campaign/gates/sequential.ts","../src/campaign/gates/statistical-heldout.ts"],"mappings":";;;;;;;UAwCiB;;;EAGf;;;;;EAKA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;EAEf;;EAEA;;;UAIe;EACf;;;;EAIA;EACA,UAAU;EACV,WAAW;;UAGI;;EAEf,OAAO;;;EAGP,kBAAkB;;EAElB,mBAAmB;;;;;;;;;;;;;;;;;;;;iBAqBL,SAAS,eAAe,gBAAgB,MAAM,kBAAkB;;UA8G/D;;EAEf;;EAEA;;EAEA,SAAS;;EAET,gBAAgB;;;UAID;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,aAAa;;EAEb;IAAY;IAAW;;;UAGR,iCAAiC;;;;EAIhD;;EAEA,YAAY;;UAGG;EACf;EACA;EACA;;;EAGA,aAAa;EACb,cAAc;;;;;;;;;;;;;;;iBAgBA,kBACd,eAAe,gBACf,MAAM,2BACL;UAmEc;EACf;EACA;EACA,UAAU;EACV,WAAW;;UAGI;EACf,OAAO;EACP,kBAAkB;EAClB,mBAAmB;;;;;;;;;iBAeL,eACd,uBAAuB,aACvB,wBAAwB,cACvB;;;;;;;;;;;;;UC1Wc;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;;;;;;;;;;KAWU;UAEK,uBAAuB;;EAEtC;;;;;;;;;EASA,OAAO;;UAGQ;EACf,UAAU;EACV;EACA;EACA;;;EAGA;;EAEA,kBAAkB;EAGlB;;;;;;;;;;;iBAYc,aAAa;;;;;;;;;;;;;;;;;;;;;;iBA8BP,SAAS,GAAG,KAAK,IAAI;;;;;;;;;;iBAkBrB,aAAa,GAAG,qBAAqB,QAAQ;;;;;;;iBAW7C,eAAe,GAAG,iBAAiB;;;;;iBAWnC,mBACpB,UAAU,gBACV;EAAY;EAAW;EAAgB;IACtC,QAAQ;;;KCpHC;UAEK;EACf,UAAU;;EAEV;;EAEA;;;EAGA;;UAGe;;;EAGf;;;;EAIA;;;EAGA;;EAEA;;;;EAIA;;;EAGA;;;EAGA,kBAAkB;;EAElB;;UAGe,qBAAqB,qBAAqB,kBAAkB,WAAW,kBAC9E,KAAK,WAAW;;;;;;;;EAQxB,QAAQ,gBAAgB;;EAExB,SAAS;IAAkB,UAAU;;;;;;;;;;;;;;;;;iBA0MvB,qBAAqB,qBAAqB,kBAAkB,WAAW,UACrF,SAAS,8BACR,qBAAqB,WAAW;UAsFlB;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;GACd;IAAQ,SAAS;;IAAyB;IAAe;;;EAE1D,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;iBA0BK,iBAAiB,UAAS,0BAA+B;;;UCtXxD;;EAEf;;EAEA;;EAEA;;;;;;;;;;;;;iBAcc,YACd,WAAW,YAAY,eAAe,cACtC,UAAU,YAAY,eAAe,cACrC,aAAa,aACb,SAAS,GAAG,oCACX;UAsDc;EACf,QAAQ;;;;;;;;;;;;EAYR,WAAW;;;;EAIX,iBAAiB;;;;;;;EAOjB,UAAU;;EAEV,mBAAmB;;EAEnB,SAAS;;;;EAIT;;EAEA;;EAEA;;EAEA,gBAAgB;;EAEhB;;;;EAIA;;EAEA;;UAGe;EACf;EACA;EACA;EACA;;EAEA;EACA;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA6Bc,oBACd,QAAQ,eACR,OAAM,6BACL;UAkEc;EACf;;;EAGA,WAAW;;;;EAIX;;EAEA;IAAM;IAAa;;;EAEnB,mBAAmB;;EAEnB,SAAS;;EAET;;;;;EAKA;EACA;EACA;;;;;iBAMc,YAAY;;;;;;;;;;;;;;;;;;;iBAsBZ,qBACd,WAAW,YAAY,eAAe,cACtC,UAAU,YAAY,eAAe,cACrC,aAAa,aACb,8BACA;EACE;EACA;EACA;EACA;;;EAGA;IAED"}
@@ -1,4 +1,4 @@
1
- import { g as JudgeScore } from "./types-XMVEdrE_.js";
1
+ import { g as JudgeScore } from "./types-D216SgwM.js";
2
2
  //#region src/judge-calibration.d.ts
3
3
  /**
4
4
  * Judge calibration — measure judge quality against human gold + bias.
@@ -965,4 +965,4 @@ declare function eProcess(opts?: EProcessOptions): EProcess;
965
965
  declare function mulberry32(seed: number): () => number;
966
966
  //#endregion
967
967
  export { pairedCohensDz as $, WeightedCompositeInput as A, positionalBias as At, eProcess as B, RankTestMethod as C, GoldenItem as Ct, ScoreRiskDifferenceResult as D, calibrateJudge as Dt, RiskDifferenceResult as E, VerbosityBiasResult as Et, cliffsDelta as F, mannWhitneyU as G, interRaterReliability as H, cohensD as I, mcnemarRequiredN as J, mcnemar as K, confidenceInterval as L, WilcoxonSignedRankResult as M, verbosityBias as Mt, benjaminiHochberg as N, SignTestAlternative as O, calibrateJudgeContinuous as Ot, bonferroni as P, pairedBootstrap as Q, corpusInterRaterAgreement as R, ProportionInterval as S, ContinuousCalibrationResult as St, RankTestOptions as T, SelfPreferenceResult as Tt, interpretCliffs as U, holm as V, isBinaryOutcomeVector as W, normalizeScores as X, mulberry32 as Y, pairedBinaryScale as Z, McNemarResult as _, wilson as _t, CorpusAgreementReport as a, pairedSignTest as at, PairedSignTestResult as b, ContinuousAgreement as bt, DEFAULT_PERMUTATIONS as c, passAtK as ct, EProcessState as d, requiredPairedSampleSize as dt, pairedDeltaTieFraction as et, EProcessStep as f, requiredSampleSize as ft, MannWhitneyResult as g, wilcoxonSignedRank as gt, MANN_WHITNEY_EXACT_MAX_WORK as h, weightedMean as ht, CorpusAgreementPerDimension as i, pairedRiskDifferenceScore as it, WeightedCompositeResult as j, selfPreference as jt, WILCOXON_EXACT_MAX_N as k, continuousAgreement as kt, EProcess as l, pearsonR as lt, MANN_WHITNEY_EXACT_MAX_STATES as m, weightedComposite as mt, CliffsMagnitude as n, pairedRiskDifference as nt, CorpusScoreRecord as o, pairedTTest as ot, ExactRiskDifferenceResult as p, spearmanR as pt, mcnemarPower as q, CorpusAgreementOptions as r, pairedRiskDifferenceExact as rt, DECISION_PAIRED_DELTA_STATISTIC as s, partialCredit as st, BOOTSTRAP_GATE_MIN_N as t, pairedMde as tt, EProcessOptions as u, ranks as ut, PairedBootstrapOptions as v, CalibrationResult as vt, RankTestMethodRequest as w, PositionalBiasResult as wt, PairedTTestResult as x, ContinuousAgreementOptions as xt, PairedBootstrapResult as y, CandidateScore as yt, corpusInterRaterAgreementFromJudgeScores as z };
968
- //# sourceMappingURL=statistics-C-dm-J6H.d.ts.map
968
+ //# sourceMappingURL=statistics-D6Uebe_4.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"statistics-C-dm-J6H.d.ts","names":[],"sources":["../src/judge-calibration.ts","../src/statistics.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;UAuBiB;EACf;EACA;;EAEA;;UAGe;EACf;EACA;;EAEA;;UAGe;EACf;EACA;;EAEA;;EAEA;;EAEA,YAAY;IAAQ;IAAgB;IAAe;IAAe;;;;;;iBAMpD,eACd,QAAQ,cACR,WAAW,mBACV;UA0Bc;;;;;EAKf;EACA;;;;;;iBAOc,eAAe,QAAQ,mBAAmB;UAgBzC;;EAEf;EACA;;iBAGc,cACd,SAAS;EAAQ;EAAmB;KACnC;UAYc;;EAEf;EACA;EACA;EACA;;;;;;;iBAQc,eACd,SAAS;EAAQ;EAAe;KAC/B;UA8Ec;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;IACE;IACA;;;EAGF;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;;;;;;;;iBAUc,oBACd,oBACA,OAAM,6BACL;UAqEc,oCAAoC;;EAEnD;;EAEA;EACA;EACA;IACE;IACA;;;;;;;iBAQY,yBACd,QAAQ,cACR,WAAW,kBACX,OAAM,6BACL;;;;;cCnVU,kBAAe,QAAY,iBAAe;;iBAGvC,aAAa;EAAU;EAAe;;;;;;;;;;iBAoBtC,mBACd,kBACA,qBACA;EAAQ;EAAe;;EACpB;EAAc;EAAe;;;;;;;;;;;;;;;;;;;iBAiDlB,sBAAsB,aAAa;;KAkFvC;;;;;KAMA;UAEK;;EAEf,SAAS;;EAET;;;EAGA;;;cAIW;;cAEA;;cAEA;;cAEA;UAEI;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER;;;;;;;;;;;;;;iBAec,aACd,aACA,aACA,OAAM,kBACL;;iBAwFa,cAAc,iBAAiB;UAK9B;;EAEf;EACA;;EAEA;;;;;;;;;;;;;;;;iBAiBc,YAAY,kBAAkB,kBAAkB;UAyB/C;;;EAGf;;EAEA;;EAEA,QAAQ;;EAER;;EAEA;;;;;;;;;;;;;;;iBAgBc,mBACd,kBACA,iBACA,OAAM,kBACL;;;;;;;;;;;;;iBAmFa,QAAQ,aAAa;;;;;;;;;iBAqBrB,eAAe,kBAAkB;KAoBrC;;;;;;;;;;;iBAYI,YAAY,kBAAkB;;;;;;iBAkB9B,gBAAgB,gBAAgB;;;;;iBAuBhC,MAAM;;;;;;iBAmBN,SAAS,aAAa;;;;;iBAuBtB,UAAU,aAAa;UAKtB;;EAEf,MAAM;;;;;EAKN,SAAS;;EAET;;UAGe;EACf;EACA;;;;;;;;;;;iBAYc,kBAAkB,OAAO,yBAAyB;UA4CjD;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe,oCAAoC;EACnD;;EAEA;;EAEA;;UAGe;;EAEf,cAAc;;EAEd;;EAEA;;EAEA;;EAEA;;UAGe,+BAA+B;;;;;;EAM9C;;;;;;EAMA;;;;;;;;;;;;;;;;;;iBAmBc,0BACd,SAAS,qBACT,OAAM,yBACL;;;;;;;;;iBA0Ha,yCACd,aAAa;EAAQ;EAAgB,QAAQ;IAC7C,OAAM,yBACL;;;;;;;iBA8Ba,mBAAmB;EACjC;EACA;EACA;EACA;;;;;;;;;;;iBAsBc,yBAAyB;EACvC;EACA;EACA;EACA;;;;;;;;iBAkBc,UAAU;EACxB;EACA;EACA;EACA;;;;;;;;;;;;;;;;iBAyBc,iBAAiB;EAC/B;EACA;EACA;EACA;EACA;;;;;;;iBAyBc,aAAa;EAC3B;EACA;EACA;EACA;EACA;;;;;;;;;iBAwBc,WACd,4BACA;EACG;EAAoB;;;;;;;;;;iBAgBT,KACd,4BACA;EACG;EAAoB;;;;;;;;;iBA4BT,kBACd,4BACA;EACG;EAAmB;;UAuCP;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;cAaW;UAEI;;EAEf;;EAEA;;EAEA;;;EAGA;;;;;;;;;;;;iBAac,gBACd,kBACA,iBACA,OAAM,yBACL;;KA0DS;;UAGK;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,aAAa;;EAEb;;;;;;;;;;;;;;;iBAgBc,eACd,gCACA,aAAa,sBACZ;;UAgDc;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;iBA2CzD,sBAAsB,QAAQ;;UAU7B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;iBAkBc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;UAkCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+Bc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;UAyEc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8Dc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;;;;;;;;;;iBA6Ea,kBACd,QAAQ,mBACR,OAAO;;;iBAkBO,uBACd,QAAQ,mBACR,OAAO;;;;;;;;;;;;;;;;;;;;;;;;;;;;;cA4CI;;;;;;;;;;;iBAYG,QAAQ,WAAW,WAAW;UAgE7B;;;EAGf;;;;EAIA;;;;EAIA;;UAGe;;EAEf;;EAEA;;EAEA;;UAGe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;iBAsatC,WAAW"}
1
+ {"version":3,"file":"statistics-D6Uebe_4.d.ts","names":[],"sources":["../src/judge-calibration.ts","../src/statistics.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;UAuBiB;EACf;EACA;;EAEA;;UAGe;EACf;EACA;;EAEA;;UAGe;EACf;EACA;;EAEA;;EAEA;;EAEA,YAAY;IAAQ;IAAgB;IAAe;IAAe;;;;;;iBAMpD,eACd,QAAQ,cACR,WAAW,mBACV;UA0Bc;;;;;EAKf;EACA;;;;;;iBAOc,eAAe,QAAQ,mBAAmB;UAgBzC;;EAEf;EACA;;iBAGc,cACd,SAAS;EAAQ;EAAmB;KACnC;UAYc;;EAEf;EACA;EACA;EACA;;;;;;;iBAQc,eACd,SAAS;EAAQ;EAAe;KAC/B;UA8Ec;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;IACE;IACA;;;EAGF;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;;;;;;;;iBAUc,oBACd,oBACA,OAAM,6BACL;UAqEc,oCAAoC;;EAEnD;;EAEA;EACA;EACA;IACE;IACA;;;;;;;iBAQY,yBACd,QAAQ,cACR,WAAW,kBACX,OAAM,6BACL;;;;;cCnVU,kBAAe,QAAY,iBAAe;;iBAGvC,aAAa;EAAU;EAAe;;;;;;;;;;iBAoBtC,mBACd,kBACA,qBACA;EAAQ;EAAe;;EACpB;EAAc;EAAe;;;;;;;;;;;;;;;;;;;iBAiDlB,sBAAsB,aAAa;;KAkFvC;;;;;KAMA;UAEK;;EAEf,SAAS;;EAET;;;EAGA;;;cAIW;;cAEA;;cAEA;;cAEA;UAEI;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER;;;;;;;;;;;;;;iBAec,aACd,aACA,aACA,OAAM,kBACL;;iBAwFa,cAAc,iBAAiB;UAK9B;;EAEf;EACA;;EAEA;;;;;;;;;;;;;;;;iBAiBc,YAAY,kBAAkB,kBAAkB;UAyB/C;;;EAGf;;EAEA;;EAEA,QAAQ;;EAER;;EAEA;;;;;;;;;;;;;;;iBAgBc,mBACd,kBACA,iBACA,OAAM,kBACL;;;;;;;;;;;;;iBAmFa,QAAQ,aAAa;;;;;;;;;iBAqBrB,eAAe,kBAAkB;KAoBrC;;;;;;;;;;;iBAYI,YAAY,kBAAkB;;;;;;iBAkB9B,gBAAgB,gBAAgB;;;;;iBAuBhC,MAAM;;;;;;iBAmBN,SAAS,aAAa;;;;;iBAuBtB,UAAU,aAAa;UAKtB;;EAEf,MAAM;;;;;EAKN,SAAS;;EAET;;UAGe;EACf;EACA;;;;;;;;;;;iBAYc,kBAAkB,OAAO,yBAAyB;UA4CjD;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe,oCAAoC;EACnD;;EAEA;;EAEA;;UAGe;;EAEf,cAAc;;EAEd;;EAEA;;EAEA;;EAEA;;UAGe,+BAA+B;;;;;;EAM9C;;;;;;EAMA;;;;;;;;;;;;;;;;;;iBAmBc,0BACd,SAAS,qBACT,OAAM,yBACL;;;;;;;;;iBA0Ha,yCACd,aAAa;EAAQ;EAAgB,QAAQ;IAC7C,OAAM,yBACL;;;;;;;iBA8Ba,mBAAmB;EACjC;EACA;EACA;EACA;;;;;;;;;;;iBAsBc,yBAAyB;EACvC;EACA;EACA;EACA;;;;;;;;iBAkBc,UAAU;EACxB;EACA;EACA;EACA;;;;;;;;;;;;;;;;iBAyBc,iBAAiB;EAC/B;EACA;EACA;EACA;EACA;;;;;;;iBAyBc,aAAa;EAC3B;EACA;EACA;EACA;EACA;;;;;;;;;iBAwBc,WACd,4BACA;EACG;EAAoB;;;;;;;;;;iBAgBT,KACd,4BACA;EACG;EAAoB;;;;;;;;;iBA4BT,kBACd,4BACA;EACG;EAAmB;;UAuCP;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;cAaW;UAEI;;EAEf;;EAEA;;EAEA;;;EAGA;;;;;;;;;;;;iBAac,gBACd,kBACA,iBACA,OAAM,yBACL;;KA0DS;;UAGK;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,aAAa;;EAEb;;;;;;;;;;;;;;;iBAgBc,eACd,gCACA,aAAa,sBACZ;;UAgDc;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;iBA2CzD,sBAAsB,QAAQ;;UAU7B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;iBAkBc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;UAkCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+Bc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;UAyEc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8Dc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;;;;;;;;;;iBA6Ea,kBACd,QAAQ,mBACR,OAAO;;;iBAkBO,uBACd,QAAQ,mBACR,OAAO;;;;;;;;;;;;;;;;;;;;;;;;;;;;;cA4CI;;;;;;;;;;;iBAYG,QAAQ,WAAW,WAAW;UAgE7B;;;EAGf;;;;EAIA;;;;EAIA;;UAGe;;EAEf;;EAEA;;EAEA;;UAGe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;iBAsatC,WAAW"}
@@ -0,0 +1,51 @@
1
+ //#region src/trajectory-replay/steps.d.ts
2
+ /**
3
+ * Recorded shell-trajectory steps and the observation grammar they carry.
4
+ *
5
+ * A recorded trajectory is the action/observation sequence an agent actually
6
+ * ran. Scaffolds that execute one shell command per step (mini-SWE and the
7
+ * CodeTracer-normalized corpora built from it) tag each observation with the
8
+ * command's returncode and its combined output:
9
+ *
10
+ * <returncode>2</returncode>
11
+ * <output>
12
+ * …command output…
13
+ * </output>
14
+ *
15
+ * The parsers here are the only place that grammar is decoded. Everything
16
+ * downstream — replay verdicts, corpus enumeration, fix prompts — reads the
17
+ * returncode, the output, and the failure signature through these functions.
18
+ */
19
+ /**
20
+ * One step of a recorded shell trajectory. Structural: any richer step record
21
+ * (file refs, thinking text, tool type) satisfies it.
22
+ */
23
+ interface RecordedTrajectoryStep {
24
+ /** 1-based position in the trajectory. */
25
+ readonly step_id: number;
26
+ readonly action: string;
27
+ /** Null when the step recorded no observation (terminal submit steps). */
28
+ readonly observation: string | null;
29
+ }
30
+ /** Recorded returncode of a step, or null when the observation carries none. */
31
+ declare function parseRecordedReturncode(observation: string | null): number | null;
32
+ /** Text between the observation's <output> tags, or the raw observation when
33
+ * the tags are absent. */
34
+ declare function parseObservationOutput(observation: string | null): string;
35
+ /**
36
+ * Stable failure-signature candidate: the first line of the recorded output
37
+ * that contains the word "error". Null when no such line exists — a verdict
38
+ * then falls back to returncode-only matching and says so.
39
+ * Pass an explicit signature to override (compiler quote glyphs vary with
40
+ * locale, so a hand-picked ASCII substring is often more robust).
41
+ */
42
+ declare function deriveFailureSignature(observation: string | null): string | null;
43
+ /** mini-SWE's end-of-run submit convention: the agent echoes this sentinel
44
+ * and dumps the diff. A label on this step marks a bad SUBMIT DECISION, not a
45
+ * failed command — there is no executable failure to reproduce, so it is
46
+ * never a counterfactual replay target. */
47
+ declare const SUBMIT_ACTION_SIGNATURE = "COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT";
48
+ declare function isSubmitAction(action: string): boolean;
49
+ //#endregion
50
+ export { parseObservationOutput as a, isSubmitAction as i, SUBMIT_ACTION_SIGNATURE as n, parseRecordedReturncode as o, deriveFailureSignature as r, RecordedTrajectoryStep as t };
51
+ //# sourceMappingURL=steps-BArUxhna.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"steps-BArUxhna.d.ts","names":[],"sources":["../src/trajectory-replay/steps.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;UAsBiB;;WAEN;WACA;;WAEA;;;iBAIK,wBAAwB;;;iBAQxB,uBAAuB;;;;;;;;iBAavB,uBAAuB;;;;;cAW1B;iBAEG,eAAe"}