@tangle-network/agent-eval 0.144.6 → 0.144.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (239) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/README.md +2 -0
  3. package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
  4. package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
  5. package/dist/analyst/index.d.ts +470 -88
  6. package/dist/analyst/index.d.ts.map +1 -1
  7. package/dist/analyst/index.js +24 -5
  8. package/dist/analyst/index.js.map +1 -1
  9. package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
  10. package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
  11. package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
  12. package/dist/baseline-CavEbRyH.d.ts.map +1 -0
  13. package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BCafwNrf.js} +662 -605
  14. package/dist/benchmark-command-BCafwNrf.js.map +1 -0
  15. package/dist/benchmarks/index.d.ts +1 -1
  16. package/dist/benchmarks/index.js +1 -1
  17. package/dist/{benchmarks-BEOkuvIg.js → benchmarks-CDSolHq7.js} +4 -4
  18. package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-CDSolHq7.js.map} +1 -1
  19. package/dist/campaign/index.d.ts +7 -5
  20. package/dist/campaign/index.js +5 -3
  21. package/dist/{campaign-CXsdyym7.js → campaign-Tdy3h62h.js} +17 -301
  22. package/dist/campaign-Tdy3h62h.js.map +1 -0
  23. package/dist/cli.js +2 -2
  24. package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
  25. package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
  26. package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
  27. package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
  28. package/dist/contract/index.d.ts +9 -8
  29. package/dist/contract/index.d.ts.map +1 -1
  30. package/dist/contract/index.js +7 -6
  31. package/dist/contract/index.js.map +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/control.js +1 -1
  34. package/dist/counterfactual-CWPTrMH7.js +126 -0
  35. package/dist/counterfactual-CWPTrMH7.js.map +1 -0
  36. package/dist/counterfactual-CxmxAONP.d.ts +72 -0
  37. package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
  38. package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
  39. package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
  40. package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
  41. package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
  42. package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
  43. package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
  44. package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
  45. package/dist/engine-nB64f48I.d.ts.map +1 -0
  46. package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
  47. package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
  48. package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
  49. package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
  50. package/dist/exec-BLtYZdWo.js +49 -0
  51. package/dist/exec-BLtYZdWo.js.map +1 -0
  52. package/dist/experiment/index.d.ts +802 -0
  53. package/dist/experiment/index.d.ts.map +1 -0
  54. package/dist/experiment/index.js +1108 -0
  55. package/dist/experiment/index.js.map +1 -0
  56. package/dist/experiment-tracker-CnRICnMl.js +500 -0
  57. package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
  58. package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
  59. package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
  60. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
  61. package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
  62. package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
  63. package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
  64. package/dist/fuzz.d.ts +1 -1
  65. package/dist/hosted/index.d.ts +3 -3
  66. package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
  67. package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
  68. package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
  69. package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
  70. package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
  71. package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
  72. package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
  73. package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
  74. package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
  75. package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
  76. package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
  77. package/dist/index-Sh2I0DRc.d.ts.map +1 -0
  78. package/dist/index.d.ts +214 -404
  79. package/dist/index.d.ts.map +1 -1
  80. package/dist/index.js +233 -649
  81. package/dist/index.js.map +1 -1
  82. package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
  83. package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
  84. package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
  85. package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
  86. package/dist/integrity-MLzHOfV9.js +141 -0
  87. package/dist/integrity-MLzHOfV9.js.map +1 -0
  88. package/dist/kind-factory-BHIgPmzS.js.map +1 -1
  89. package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
  90. package/dist/llm-client-DzvMUsS_.js.map +1 -0
  91. package/dist/matrix/index.d.ts +2 -2
  92. package/dist/meta-eval/index.d.ts +1 -1
  93. package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
  94. package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
  95. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
  96. package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
  97. package/dist/multishot/index.d.ts +2 -2
  98. package/dist/openapi.json +1 -1
  99. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
  100. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
  101. package/dist/pipelines/index.d.ts +2 -1
  102. package/dist/pipelines/index.d.ts.map +1 -1
  103. package/dist/pre-registration-DakwTRXk.js +96 -0
  104. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  105. package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
  106. package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
  107. package/dist/prime-protocol-BfSalTfR.js +453 -0
  108. package/dist/prime-protocol-BfSalTfR.js.map +1 -0
  109. package/dist/profile-cell.js +242 -1
  110. package/dist/profile-cell.js.map +1 -0
  111. package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
  112. package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
  113. package/dist/promotion-policy-CrLrmys8.js +682 -0
  114. package/dist/promotion-policy-CrLrmys8.js.map +1 -0
  115. package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
  116. package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
  117. package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
  118. package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
  119. package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
  120. package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
  121. package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
  122. package/dist/replay-DFf-teiC.d.ts.map +1 -0
  123. package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
  124. package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
  125. package/dist/reporting.d.ts +3 -3
  126. package/dist/reporting.js +2 -2
  127. package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
  128. package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
  130. package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
  131. package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
  132. package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
  133. package/dist/rl.d.ts +17 -7
  134. package/dist/rl.d.ts.map +1 -1
  135. package/dist/rl.js +16 -7
  136. package/dist/rl.js.map +1 -1
  137. package/dist/rollout/index.d.ts +2 -2
  138. package/dist/rollout/index.js +2 -2
  139. package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
  140. package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
  141. package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
  142. package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
  143. package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
  144. package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
  145. package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
  146. package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
  147. package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
  148. package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
  149. package/dist/sequential-D-BLJBKU.js +299 -0
  150. package/dist/sequential-D-BLJBKU.js.map +1 -0
  151. package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
  152. package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
  153. package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
  154. package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
  155. package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
  156. package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
  157. package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
  158. package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
  159. package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
  160. package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
  161. package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
  162. package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
  163. package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
  164. package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
  165. package/dist/steps-BArUxhna.d.ts +51 -0
  166. package/dist/steps-BArUxhna.d.ts.map +1 -0
  167. package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
  168. package/dist/store-DNe_Uv1Q.js.map +1 -0
  169. package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
  170. package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
  171. package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
  172. package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
  173. package/dist/supervisor-run/index.d.ts +2 -2
  174. package/dist/supervisor-run/index.js +1 -1
  175. package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
  176. package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
  177. package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
  178. package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
  179. package/dist/trace-repair/index.d.ts +2102 -0
  180. package/dist/trace-repair/index.d.ts.map +1 -0
  181. package/dist/trace-repair/index.js +3878 -0
  182. package/dist/trace-repair/index.js.map +1 -0
  183. package/dist/traces.d.ts +6 -6
  184. package/dist/traces.js +3 -2
  185. package/dist/trajectory-YC15QDYQ.d.ts +24 -0
  186. package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
  187. package/dist/trajectory-replay/index.d.ts +781 -0
  188. package/dist/trajectory-replay/index.d.ts.map +1 -0
  189. package/dist/trajectory-replay/index.js +2103 -0
  190. package/dist/trajectory-replay/index.js.map +1 -0
  191. package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
  192. package/dist/types-D216SgwM.d.ts.map +1 -0
  193. package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
  194. package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
  195. package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
  196. package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
  197. package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
  198. package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
  199. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
  200. package/dist/verdict-DExhxfgR.d.ts +201 -0
  201. package/dist/verdict-DExhxfgR.d.ts.map +1 -0
  202. package/dist/verdict-cache-BCcOh0kF.js +159 -0
  203. package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
  204. package/dist/wire/index.d.ts +2 -2
  205. package/dist/wire/index.js +1 -1
  206. package/docs/charter.md +112 -0
  207. package/docs/experiment.md +104 -0
  208. package/docs/prime-analyst.md +1 -0
  209. package/docs/trace-analysis.md +26 -0
  210. package/docs/trace-repair-admission.md +194 -0
  211. package/docs/trace-repair-analyst-arms.md +121 -0
  212. package/docs/trace-repair-continuation.md +107 -0
  213. package/docs/trace-repair-grader.md +163 -0
  214. package/docs/trajectory-replay.md +110 -0
  215. package/docs/verification-strategies.md +103 -0
  216. package/package.json +19 -2
  217. package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
  218. package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
  219. package/dist/baseline-D_fT6277.d.ts.map +0 -1
  220. package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
  221. package/dist/benchmark-command-CQd78YHt.js.map +0 -1
  222. package/dist/campaign-CXsdyym7.js.map +0 -1
  223. package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
  224. package/dist/index-4XwggC10.d.ts.map +0 -1
  225. package/dist/index-BIL5vxxt.d.ts.map +0 -1
  226. package/dist/integrity-fdt8XPAv.js.map +0 -1
  227. package/dist/llm-client-Dv5BiKLE.js.map +0 -1
  228. package/dist/replay-Krvb114g.d.ts.map +0 -1
  229. package/dist/reward-hacking-CyuzxKly.js.map +0 -1
  230. package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
  231. package/dist/schema-Cef2cFmb.d.ts.map +0 -1
  232. package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
  233. package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
  234. package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
  235. package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
  236. package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
  237. package/dist/types-XMVEdrE_.d.ts.map +0 -1
  238. package/dist/verdict-Dps8_okt.d.ts +0 -37
  239. package/dist/verdict-Dps8_okt.d.ts.map +0 -1
@@ -121,6 +121,23 @@ function assertCrossFamily(models, opts = {}) {
121
121
  * does not.
122
122
  */
123
123
  /**
124
+ * Output-token budget a liveness probe must grant a model.
125
+ *
126
+ * Identity is only readable off a response the provider actually produced. A
127
+ * reasoning model spends budget on hidden reasoning tokens before it emits a
128
+ * single visible one, so a cap of a few tokens makes a HEALTHY deepseek/glm
129
+ * model fail with `reasoning_budget_exhausted` — it names no model, its
130
+ * identity reads as `unreported`, and a preflight scores it DEAD. 64 clears
131
+ * that floor. Every probe in this package reads this constant, so two probes
132
+ * cannot reach two different answers about the same router.
133
+ *
134
+ * Cost: a probe spends at most `PROBE_MAX_TOKENS` output tokens per model,
135
+ * plus whatever reasoning tokens a reasoning model bills — roughly 400 output
136
+ * tokens for a six-model preflight. That is fractions of a cent, and far
137
+ * cheaper than a campaign that runs on a model nobody proved was alive.
138
+ */
139
+ const PROBE_MAX_TOKENS = 64;
140
+ /**
124
141
  * Reduce a model id to its comparable core: lowercase, no surrounding space,
125
142
  * no `provider/` prefix, no `@snapshot` / `:batch` / `:free` tier suffix, no
126
143
  * trailing build date, and `.`/`_` folded to `-` so one version is spelled one
@@ -898,10 +915,11 @@ function describePattern(p) {
898
915
  * network, parse). Designed for sweep preflights — fail loud at the
899
916
  * boundary before burning a 30-leaf run on a misconfigured router.
900
917
  *
901
- * Sends a tiny `ping` message with `maxTokens=64`. Reasoning models
902
- * (glm-5.1, deepseek-v4) can burn the entire budget on internal reasoning
903
- * for short prompts, so don't tighten this further. We don't validate
904
- * content.
918
+ * Sends a tiny `ping` message with `maxTokens = PROBE_MAX_TOKENS`. Reasoning
919
+ * models (glm-5.1, deepseek-v4) can burn the entire budget on internal
920
+ * reasoning for short prompts, so don't tighten this further the shared
921
+ * constant keeps this probe and `preflightModels` on one answer. We don't
922
+ * validate content.
905
923
  *
906
924
  * Reachability and identity are separate answers: `ok` means the route
907
925
  * answered, `servedModel` / `substituted` say WHICH model answered. A gateway
@@ -964,6 +982,6 @@ var LlmClient = class {
964
982
  }
965
983
  };
966
984
  //#endregion
967
- export { CrossFamilyError as C, servedModelAcceptable as S, judgeFamily as T, assertCrossFamilyServed as _, assertLlmRoute as a, checkServedModel as b, callLlmJson as c, isTransientLlmError as d, maximumChargeForLlmRequest as f, ServedCrossFamilyError as g, ModelSubstitutionError as h, LlmRouteAssertionError as i, costReceiptFromLlm as l, stripFencedJson as m, LlmClient as n, backoffMs as o, probeLlm as p, LlmResponseError as r, callLlm as s, LlmCallError as t, costReceiptFromLlmError as u, assertServedModel as v, assertCrossFamily as w, normalizeModelId as x, assertServedModels as y };
985
+ export { servedModelAcceptable as C, judgeFamily as E, normalizeModelId as S, assertCrossFamily as T, ServedCrossFamilyError as _, assertLlmRoute as a, assertServedModels as b, callLlmJson as c, isTransientLlmError as d, maximumChargeForLlmRequest as f, PROBE_MAX_TOKENS as g, ModelSubstitutionError as h, LlmRouteAssertionError as i, costReceiptFromLlm as l, stripFencedJson as m, LlmClient as n, backoffMs as o, probeLlm as p, LlmResponseError as r, callLlm as s, LlmCallError as t, costReceiptFromLlmError as u, assertCrossFamilyServed as v, CrossFamilyError as w, checkServedModel as x, assertServedModel as y };
968
986
 
969
- //# sourceMappingURL=llm-client-Dv5BiKLE.js.map
987
+ //# sourceMappingURL=llm-client-DzvMUsS_.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"llm-client-DzvMUsS_.js","names":[],"sources":["../src/judge-families.ts","../src/integrity/served-model.ts","../src/llm-client.ts"],"sourcesContent":["/**\n * Judge model-family classification + cross-family enforcement.\n *\n * A judge ensemble built entirely from one provider family shares that\n * family's blind spots and self-preference — its \"agreement\" is correlated\n * bias, not independent signal. `assertCrossFamily` makes the consumer prove\n * the ensemble spans ≥2 families; `judgeFamily` is the single regex map that\n * replaces the per-consumer copies (tax/legal/creative/gtm each ship one).\n */\n\n/** Provider family a model belongs to. `unknown` when no rule matches. */\nexport type JudgeFamily =\n | 'anthropic'\n | 'openai'\n | 'google'\n | 'meta'\n | 'mistral'\n | 'deepseek'\n | 'xai'\n | 'qwen'\n | 'cohere'\n | 'amazon'\n | 'moonshot'\n | 'zhipu'\n | 'unknown'\n\n/** Explicit `provider/...` prefix → family (models.dev / OpenRouter style). */\nconst PROVIDER_PREFIX: Record<string, JudgeFamily> = {\n anthropic: 'anthropic',\n openai: 'openai',\n 'azure-openai': 'openai',\n google: 'google',\n 'google-vertex': 'google',\n meta: 'meta',\n 'meta-llama': 'meta',\n mistral: 'mistral',\n mistralai: 'mistral',\n deepseek: 'deepseek',\n xai: 'xai',\n qwen: 'qwen',\n alibaba: 'qwen',\n cohere: 'cohere',\n amazon: 'amazon',\n bedrock: 'amazon',\n moonshot: 'moonshot',\n moonshotai: 'moonshot',\n kimi: 'moonshot',\n 'kimi-code': 'moonshot',\n zhipu: 'zhipu',\n zhipuai: 'zhipu',\n zai: 'zhipu',\n 'z-ai': 'zhipu',\n glm: 'zhipu',\n}\n\n/** Fallback model-name patterns when there's no recognised provider prefix. */\nconst NAME_PATTERNS: Array<[RegExp, JudgeFamily]> = [\n [/claude/i, 'anthropic'],\n [/\\b(gpt|davinci|babbage)\\b|^o[134]\\b|[-/]o[134]\\b|gpt-/i, 'openai'],\n [/gemini|palm|gemma|bison/i, 'google'],\n [/llama/i, 'meta'],\n [/mi(s|x)tral|codestral|magistral/i, 'mistral'],\n [/deepseek/i, 'deepseek'],\n [/grok/i, 'xai'],\n [/qwen/i, 'qwen'],\n [/command-?(r|a)?/i, 'cohere'],\n [/\\b(nova|titan)\\b/i, 'amazon'],\n [/\\bkimi\\b|moonshot/i, 'moonshot'],\n [/\\bglm\\b|zhipu|\\bz-?ai\\b/i, 'zhipu'],\n]\n\n/**\n * Classify a model id into its provider family. Strips a `@snapshot` suffix\n * and prefers an explicit `provider/...` prefix; otherwise matches the model\n * name. Returns `unknown` when nothing matches (callers decide whether that's\n * acceptable — `assertCrossFamily` counts it as its own family).\n */\nexport function judgeFamily(modelId: string): JudgeFamily {\n const id = modelId.trim().split('@')[0]!.toLowerCase()\n const slash = id.indexOf('/')\n if (slash > 0) {\n const prefix = id.slice(0, slash)\n const mapped = PROVIDER_PREFIX[prefix]\n if (mapped) return mapped\n }\n for (const [pattern, family] of NAME_PATTERNS) {\n if (pattern.test(id)) return family\n }\n return 'unknown'\n}\n\nexport interface AssertCrossFamilyOptions {\n /** Minimum number of distinct families the ensemble must span. Default 2. */\n minFamilies?: number\n /** When false (default), `unknown`-family models do NOT count toward the\n * family total — an ensemble of all-unclassifiable models is not provably\n * cross-family. Set true to count `unknown` as one shared family. */\n allowUnknown?: boolean\n}\n\nexport class CrossFamilyError extends Error {\n constructor(\n message: string,\n public readonly families: JudgeFamily[],\n public readonly models: string[],\n ) {\n super(message)\n this.name = 'CrossFamilyError'\n }\n}\n\n/**\n * Throw unless the judge models span at least `minFamilies` distinct provider\n * families. Pass the model ids backing your judge ensemble. Fail-loud by\n * design — a correlated single-family ensemble silently inflates agreement.\n *\n * Scope: this reads the ids you REQUEST. It proves the panel was configured\n * across families; it cannot prove the panel RAN across families, because a\n * routing gateway may answer several different ids from one provider. Where\n * the diversity claim is load-bearing (a published leaderboard, a\n * certification, a non-self-judging exclusion), assert on the ids the\n * provider echoed instead: `assertCrossFamilyServed` in\n * ./integrity/served-model.\n */\nexport function assertCrossFamily(\n models: string[],\n opts: AssertCrossFamilyOptions = {},\n): JudgeFamily[] {\n const minFamilies = opts.minFamilies ?? 2\n const families = new Set<JudgeFamily>()\n for (const m of models) {\n const f = judgeFamily(m)\n if (f === 'unknown' && !opts.allowUnknown) continue\n families.add(f)\n }\n const list = [...families].sort()\n if (list.length < minFamilies) {\n throw new CrossFamilyError(\n `judge ensemble spans ${list.length} provider famil${list.length === 1 ? 'y' : 'ies'} ` +\n `(${list.join(', ') || 'none'}) but ${minFamilies} required — a single-family ensemble ` +\n 'is correlated bias, not independent signal',\n list,\n models,\n )\n }\n return list\n}\n","/**\n * Served-model identity: prove the model that ANSWERED is the model that was\n * REQUESTED.\n *\n * A routing gateway can accept `model: \"gpt-4.1-mini\"` and answer from a\n * different model entirely. Every guard that reasons about the requested id —\n * cross-family judge diversity, per-model leaderboard rows, non-self-judging\n * exclusions, cost attribution — then passes while measuring something else.\n * The requested id is an intent; only the id echoed on the response is\n * evidence.\n *\n * `checkServedModel` classifies one requested/served pair; `assertServedModel`\n * and `assertServedModels` are the fail-loud gates; `assertCrossFamilyServed`\n * is the family-diversity rule computed over SERVED ids (the requested-id\n * version lives in ../judge-families and cannot see substitution).\n *\n * Aliasing is tolerated, substitution is not: `openai/gpt-4.1-mini` and\n * `gpt-4.1-mini` are the same request expressed two ways, so a response\n * echoing either satisfies the other. A response echoing `gemini-2.5-flash-lite`\n * does not.\n */\n\nimport { AgentEvalError } from '../errors'\nimport { type JudgeFamily, judgeFamily } from '../judge-families'\n\n/**\n * Output-token budget a liveness probe must grant a model.\n *\n * Identity is only readable off a response the provider actually produced. A\n * reasoning model spends budget on hidden reasoning tokens before it emits a\n * single visible one, so a cap of a few tokens makes a HEALTHY deepseek/glm\n * model fail with `reasoning_budget_exhausted` — it names no model, its\n * identity reads as `unreported`, and a preflight scores it DEAD. 64 clears\n * that floor. Every probe in this package reads this constant, so two probes\n * cannot reach two different answers about the same router.\n *\n * Cost: a probe spends at most `PROBE_MAX_TOKENS` output tokens per model,\n * plus whatever reasoning tokens a reasoning model bills — roughly 400 output\n * tokens for a six-model preflight. That is fractions of a cent, and far\n * cheaper than a campaign that runs on a model nobody proved was alive.\n */\nexport const PROBE_MAX_TOKENS = 64\n\n/** How a served id relates to the id that was requested. */\nexport type ServedModelVerdict =\n /** Byte-identical after normalisation — the requested model answered. */\n | 'exact'\n /** Same model, different spelling (provider prefix, snapshot, tier suffix). */\n | 'alias'\n /** A different model of the SAME provider family answered. */\n | 'substituted-within-family'\n /** A different provider's model answered. */\n | 'substituted-cross-family'\n /** The response carried no model id — identity is unproven either way. */\n | 'unreported'\n\nexport interface ServedModelCheck {\n /** The id the caller asked for. */\n requested: string\n /** The id echoed on the response; `null` when the response omitted it. */\n served: string | null\n requestedFamily: JudgeFamily\n /** `null` when `served` is null. */\n servedFamily: JudgeFamily | null\n verdict: ServedModelVerdict\n /** True for every verdict except `exact` and `alias`. */\n substituted: boolean\n}\n\n/**\n * Reduce a model id to its comparable core: lowercase, no surrounding space,\n * no `provider/` prefix, no `@snapshot` / `:batch` / `:free` tier suffix, no\n * trailing build date, and `.`/`_` folded to `-` so one version is spelled one\n * way.\n *\n * Dropping the build date is what makes snapshot resolution legible as the\n * non-event it is: a router answering `gpt-4o-mini` with\n * `gpt-4o-mini-2024-07-18` pinned a floating alias to a reproducible build —\n * the same model, which is the behaviour we want. Only routing decoration is\n * stripped; version digits are load-bearing, so `deepseek-v3.2` and\n * `deepseek-v4-flash` stay distinct, and comparison is EXACT equality rather\n * than a prefix test (a prefix rule would accept `gpt-5` → `gpt-5-mini`, a\n * silent downgrade wearing the right vendor name).\n */\nexport function normalizeModelId(modelId: string): string {\n let id = modelId.trim().toLowerCase()\n const at = id.indexOf('@')\n if (at > 0) id = id.slice(0, at)\n const colon = id.indexOf(':')\n if (colon > 0) id = id.slice(0, colon)\n const slash = id.lastIndexOf('/')\n if (slash >= 0) id = id.slice(slash + 1)\n return id\n .replace(/-\\d{4}-\\d{2}-\\d{2}$/, '')\n .replace(/-\\d{8}$/, '')\n .replace(/[._]/g, '-')\n .replace(/-+$/, '')\n .trim()\n}\n\n/**\n * Classify one requested/served pair. Pure — no I/O — so it is safe inside\n * response handlers, reducers, and CI gates.\n *\n * `served` is the id echoed by the provider (OpenAI-compatible bodies put it\n * at `model`). `null`/`undefined` means the body omitted it; that is\n * `unreported`, NOT a pass — a provider that does not name what answered has\n * not proven identity, and a transport that drops the field must not read as\n * agreement.\n */\nexport function checkServedModel(\n requested: string,\n served: string | null | undefined,\n): ServedModelCheck {\n const requestedFamily = judgeFamily(requested)\n if (served === null || served === undefined || served.trim() === '') {\n return {\n requested,\n served: null,\n requestedFamily,\n servedFamily: null,\n verdict: 'unreported',\n substituted: true,\n }\n }\n const servedFamily = judgeFamily(served)\n if (requested.trim().toLowerCase() === served.trim().toLowerCase()) {\n return {\n requested,\n served,\n requestedFamily,\n servedFamily,\n verdict: 'exact',\n substituted: false,\n }\n }\n if (normalizeModelId(requested) === normalizeModelId(served)) {\n return {\n requested,\n served,\n requestedFamily,\n servedFamily,\n verdict: 'alias',\n substituted: false,\n }\n }\n return {\n requested,\n served,\n requestedFamily,\n servedFamily,\n verdict:\n requestedFamily === servedFamily ? 'substituted-within-family' : 'substituted-cross-family',\n substituted: true,\n }\n}\n\nexport class ModelSubstitutionError extends AgentEvalError {\n constructor(\n message: string,\n public readonly checks: ReadonlyArray<ServedModelCheck>,\n ) {\n super('model_substitution', message)\n this.name = 'ModelSubstitutionError'\n }\n}\n\nexport interface AssertServedModelOptions {\n /**\n * Accept a different model of the same provider family (e.g. requested\n * `deepseek-v3.2`, served `deepseek-v4-flash`). Default false. Setting this\n * keeps family-level claims valid and forfeits per-model claims.\n */\n allowWithinFamily?: boolean\n /**\n * Accept a response that carried no model id. Default false — an\n * unidentified response cannot support a per-model or per-family claim.\n */\n allowUnreported?: boolean\n /** Prefixed to the thrown message, e.g. the judge or campaign cell name. */\n context?: string\n}\n\n/**\n * The one place the accept/reject policy lives, so a caller that reports\n * substitution (a preflight table, a run record) and a caller that throws on it\n * can never drift apart. A cross-family substitution is never acceptable.\n */\nexport function servedModelAcceptable(\n check: ServedModelCheck,\n opts: AssertServedModelOptions = {},\n): boolean {\n switch (check.verdict) {\n case 'exact':\n case 'alias':\n return true\n case 'unreported':\n return opts.allowUnreported === true\n case 'substituted-within-family':\n return opts.allowWithinFamily === true\n default:\n return false\n }\n}\n\nfunction describe(check: ServedModelCheck): string {\n if (check.verdict === 'unreported')\n return `${check.requested}: response carried no model id (identity unproven)`\n return (\n `${check.requested} (${check.requestedFamily}) → served ${check.served} ` +\n `(${check.servedFamily}) [${check.verdict}]`\n )\n}\n\n/**\n * Throw `ModelSubstitutionError` unless the served id is the requested model.\n * Returns the check on success so callers can record the served id alongside\n * the result.\n */\nexport function assertServedModel(\n requested: string,\n served: string | null | undefined,\n opts: AssertServedModelOptions = {},\n): ServedModelCheck {\n const check = checkServedModel(requested, served)\n if (servedModelAcceptable(check, opts)) return check\n const prefix = opts.context ? `${opts.context}: ` : ''\n throw new ModelSubstitutionError(\n `${prefix}model substitution — ${describe(check)}. The measurement is of the SERVED model, ` +\n 'not the requested one; any per-model or per-family claim from this call is invalid.',\n [check],\n )\n}\n\n/**\n * Batch form: check every pair and throw naming EVERY substitution, so one\n * failure does not hide the rest. Returns all checks on success.\n */\nexport function assertServedModels(\n pairs: ReadonlyArray<{ requested: string; served: string | null | undefined }>,\n opts: AssertServedModelOptions = {},\n): ServedModelCheck[] {\n const checks = pairs.map((p) => checkServedModel(p.requested, p.served))\n const bad = checks.filter((c) => !servedModelAcceptable(c, opts))\n if (bad.length > 0) {\n const prefix = opts.context ? `${opts.context}: ` : ''\n throw new ModelSubstitutionError(\n `${prefix}${bad.length}/${checks.length} call(s) were answered by a different model than ` +\n `requested — ${bad.map(describe).join('; ')}. Per-model and per-family claims from this ` +\n 'run are invalid until the ids are re-measured.',\n checks,\n )\n }\n return checks\n}\n\nexport interface AssertCrossFamilyServedOptions extends AssertServedModelOptions {\n /** Minimum distinct SERVED families required. Default 2. */\n minFamilies?: number\n /** Count `unknown`-family served ids toward the total. Default false. */\n allowUnknown?: boolean\n}\n\nexport class ServedCrossFamilyError extends AgentEvalError {\n constructor(\n message: string,\n public readonly families: JudgeFamily[],\n public readonly checks: ReadonlyArray<ServedModelCheck>,\n ) {\n super('model_substitution', message)\n this.name = 'ServedCrossFamilyError'\n }\n}\n\n/**\n * Family-diversity rule over the models that actually ANSWERED.\n *\n * `assertCrossFamily` (../judge-families) reads the requested ids and so\n * cannot see a gateway that answers three \"different\" requests from one\n * provider. This one asserts no substitution first, then counts families from\n * the served ids — a panel that collapsed to one family under the hood fails\n * here even though its request list looked diverse.\n */\nexport function assertCrossFamilyServed(\n pairs: ReadonlyArray<{ requested: string; served: string | null | undefined }>,\n opts: AssertCrossFamilyServedOptions = {},\n): JudgeFamily[] {\n const checks = assertServedModels(pairs, opts)\n const families = new Set<JudgeFamily>()\n for (const check of checks) {\n const family = check.servedFamily\n if (family === null) continue\n if (family === 'unknown' && !opts.allowUnknown) continue\n families.add(family)\n }\n const list = [...families].sort()\n const minFamilies = opts.minFamilies ?? 2\n if (list.length < minFamilies) {\n const prefix = opts.context ? `${opts.context}: ` : ''\n throw new ServedCrossFamilyError(\n `${prefix}the models that ANSWERED span ${list.length} provider famil` +\n `${list.length === 1 ? 'y' : 'ies'} (${list.join(', ') || 'none'}) but ${minFamilies} ` +\n `required — served ids: ${checks.map((c) => c.served ?? 'unreported').join(', ')}`,\n list,\n checks,\n )\n }\n return list\n}\n","/**\n * LLM client with graceful degrade.\n *\n * OpenAI-compatible `/v1/chat/completions` client with:\n * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).\n * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).\n * - One retry at temperature 1 when a model explicitly requires it.\n * - Graceful json_schema → json_object degrade on 400 with schema-reject body.\n * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.\n * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI\n * directly, cli-bridge subscriptions, and any router that speaks the spec.\n *\n * Usage:\n * const { value, result } = await callLlmJson<MyType>(\n * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },\n * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },\n * )\n *\n * `createChatClient` wraps this implementation for provider-neutral package\n * entry points. Direct callers can use `callLlm` or `callLlmJson`.\n */\n\nimport {\n type CostReceiptInput,\n type CustomTokenPricing,\n costForTokenPricing,\n type MaximumCharge,\n} from './cost-ledger'\nimport { AgentEvalError, CaptureIntegrityError } from './errors'\nimport {\n type AssertServedModelOptions,\n assertServedModel as assertServedModelIdentity,\n checkServedModel,\n PROBE_MAX_TOKENS,\n} from './integrity/served-model'\nimport {\n defaultProviderRedactor,\n type ProviderRedactor,\n providerFromBaseUrl,\n type RawProviderEvent,\n type RawProviderSink,\n} from './trace/raw-provider-sink'\n\n// ─── Types ──────────────────────────────────────────────────────────────\n\nexport interface LlmMessage {\n role: 'system' | 'user' | 'assistant'\n /**\n * Either a plain text content string OR a multimodal content array\n * (text + image_url parts) for vision-capable models.\n */\n content:\n | string\n | Array<\n | { type: 'text'; text: string }\n | { type: 'image_url'; image_url: { url: string; detail?: 'auto' | 'low' | 'high' } }\n >\n}\n\nexport type LlmThinkingMode = 'enabled' | 'disabled'\n\nexport interface LlmCallRequest {\n model: string\n messages: LlmMessage[]\n /** Optional JSON-mode response format (response_format: json_object). */\n jsonMode?: boolean\n /** Optional structured output via JSON Schema. Falls back to json_object on 400. */\n jsonSchema?: { name: string; schema: Record<string, unknown> }\n temperature?: number\n maxTokens?: number\n /** OpenAI-compatible reasoning mode. Omitted when the provider default should apply. */\n thinking?: LlmThinkingMode\n /** Per-call timeout, default 300s. */\n timeoutMs?: number\n}\n\n/** Conservative priced bound for the exact text request sent to a provider.\n * Returns undefined when output or multimodal input is not bounded, causing a\n * capped CostLedger to reject the call before execution. Pass\n * `customTokenPricing` when package pricing does not cover the model or endpoint. */\nexport function maximumChargeForLlmRequest(\n request: Pick<LlmCallRequest, 'model' | 'messages' | 'jsonSchema' | 'maxTokens' | 'thinking'>,\n options: LlmClientOptions = {},\n): MaximumCharge | undefined {\n if (request.maxTokens === undefined) return undefined\n if (!Number.isInteger(request.maxTokens) || request.maxTokens <= 0) {\n throw new RangeError(`maximumChargeForLlmRequest: maxTokens must be a positive integer`)\n }\n if (\n request.messages.some(\n (message) =>\n Array.isArray(message.content) && message.content.some((part) => part.type === 'image_url'),\n )\n ) {\n return undefined\n }\n\n const attempts = resolveMaximumAttempts(options.maximumAttempts)\n const forceJsonObject = options.jsonSchemaTransport === 'json-object'\n // A byte-level tokenizer cannot emit more input tokens than request bytes.\n // Pricing the complete body also covers role/schema framing omitted from content-only estimates.\n const requestBytes = new TextEncoder().encode(\n JSON.stringify(buildBody(request, forceJsonObject, options.thinking)),\n ).byteLength\n // A rejected response schema can trigger one JSON-mode batch with the same output limit.\n const batches = request.jsonSchema && !forceJsonObject ? 2 : 1\n const usage = {\n inputTokens: requestBytes * attempts * batches,\n outputTokens: request.maxTokens * attempts * batches,\n }\n return options.customTokenPricing\n ? { customTokenPricing: options.customTokenPricing, ...usage }\n : { model: request.model, ...usage }\n}\n\nexport interface LlmUsage {\n promptTokens: number\n completionTokens: number\n totalTokens: number\n /** False when the provider omitted or malformed prompt/completion usage. */\n captured?: boolean\n /** Reasoning-token subset of completionTokens, when reported. */\n reasoningTokens?: number\n /** Proxies populate this when prompt caching is on. */\n cachedPromptTokens?: number\n}\n\nexport interface LlmCallResult {\n /** The text content of the first choice. Empty string if none. */\n content: string\n usage: LlmUsage\n /**\n * Cost in USD. Uses the provider's reported cost when present, otherwise\n * caller-supplied token pricing. `null` when neither is available.\n */\n costUsd: number | null\n /**\n * Model id used for attribution (cost, pricing, log lines). The response's\n * echoed id when the provider sent one, else the requested id.\n *\n * NOT evidence of which model answered — read `servedModel` for that. A\n * provider that omits `model` makes this equal to the request, which is\n * exactly the case an identity check must be able to distinguish.\n */\n model: string\n /**\n * The model id the provider echoed on the response, verbatim; `null` when\n * the body carried none. This is the only field that can witness a gateway\n * substituting a different model for the one requested — compare it with\n * `assertServedModel` / `checkServedModel` (src/integrity/served-model.ts).\n *\n * Optional so hand-built results (mock/custom transports) still typecheck,\n * but omitting it is not a pass: the identity checks read `undefined` as\n * `unreported` and reject it by default. A transport that knows which model\n * answered should say so.\n */\n servedModel?: string | null\n /** Wall-clock duration of the HTTP call (last attempt, if retried). */\n durationMs: number\n /**\n * `finish_reason` echoed from the first choice (`stop`, `length`,\n * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.\n * Exposed so a free-form `callLlm` caller CAN detect a truncated answer\n * (`length`) instead of treating a cut-off completion as complete. Note:\n * `callLlm` does not itself reject on it — acting on this signal is the\n * caller's responsibility (in-repo free-form drivers do not yet enforce it).\n */\n finishReason?: string | null\n /**\n * True when `content.trim()` is empty. An empty completion is a silent zero\n * for free-form `callLlm` callers; this flag is the signal a caller can\n * inspect to fail loud rather than proceed on an empty string. `callLlm`\n * surfaces it but does not throw on it.\n */\n contentEmpty?: boolean\n /** Raw response body. */\n raw: Record<string, unknown>\n}\n\nexport type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>\n\n/** Convert a provider result into the canonical paid-call receipt input. */\nexport function costReceiptFromLlm(\n result: LlmCallResult,\n customTokenPricing?: CustomTokenPricing,\n): CostReceiptInput {\n const cachedTokens = result.usage.cachedPromptTokens ?? 0\n const inputTokens = Math.max(0, result.usage.promptTokens - cachedTokens)\n const providerCostUsd = providerReportedCost(result.raw)\n return {\n model: result.model,\n inputTokens,\n outputTokens: result.usage.completionTokens,\n reasoningTokens: result.usage.reasoningTokens,\n cachedTokens: cachedTokens > 0 ? cachedTokens : undefined,\n ...(providerCostUsd === undefined\n ? customTokenPricing && result.usage.captured !== false\n ? { customTokenPricing }\n : result.costUsd === null\n ? {}\n : { estimatedCostUsd: result.costUsd }\n : { actualCostUsd: providerCostUsd }),\n usageUnknown: result.usage.captured === false,\n }\n}\n\nfunction providerReportedCost(raw: Record<string, unknown>): number | undefined {\n const value = raw._response_cost ?? raw.cost_usd\n return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined\n}\n\n/** Structured-response failures retain their completed provider receipt. */\nexport function costReceiptFromLlmError(\n error: Error,\n customTokenPricing?: CustomTokenPricing,\n): CostReceiptInput | undefined {\n return error instanceof LlmResponseError\n ? costReceiptFromLlm(error.result, customTokenPricing)\n : undefined\n}\n\nexport class LlmCallError extends AgentEvalError {\n constructor(\n message: string,\n public readonly status: number,\n public readonly body: string,\n public readonly model: string,\n ) {\n super('judge', message)\n }\n}\n\n/** A provider response completed and incurred measurable usage, but its content\n * could not satisfy the caller's response contract. The response envelope is\n * retained so accounting can commit the receipt before the error propagates. */\nexport class LlmResponseError extends AgentEvalError {\n constructor(\n message: string,\n public readonly result: LlmCallResult,\n options?: { cause?: unknown },\n ) {\n super('judge', message, options)\n }\n}\n\nexport interface LlmClientOptions {\n /** Base URL (without trailing slash). Must end at the `/v1` prefix. */\n baseUrl?: string\n /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */\n apiKey?: string\n bearer?: string\n /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */\n authHeader?: { name: string; value: string }\n /** Stable provider idempotency key, reused across retries of this logical call. */\n idempotencyKey?: string\n /** Default timeout in ms. Per-call can override. */\n defaultTimeoutMs?: number\n /**\n * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to\n * each attempt's per-attempt timeout controller, so aborting it cancels\n * the in-flight fetch. A caller abort is FATAL: it is not retried even\n * though an AbortError otherwise matches the transient patterns.\n */\n signal?: AbortSignal\n /**\n * Cross-attempt wall-clock budget in ms, measured from the first attempt.\n * Before launching each attempt the loop checks the remaining budget and\n * stops retrying once it is exhausted, rather than waiting the full\n * per-attempt timeout on every retry. Bounds total time independent of\n * total attempts × `timeoutMs`.\n */\n deadlineMs?: number\n /** Total provider attempts. Default 3. */\n maximumAttempts?: number\n /** Token rates used when the provider omits cost or package pricing does not cover the model. */\n customTokenPricing?: CustomTokenPricing\n /**\n * Transport for requests that declare `jsonSchema`. `native` sends\n * `response_format: json_schema`; `json-object` sends the broadly supported\n * JSON mode and relies on the caller to include the schema in model-visible\n * instructions. Default: `native`.\n */\n jsonSchemaTransport?: 'native' | 'json-object'\n /**\n * JSON payload parsing policy. `extract` accepts fenced or prose-prefixed JSON.\n * `exact` requires the complete response content to be one JSON value.\n * Default: `extract`.\n */\n jsonPayloadMode?: 'extract' | 'exact'\n /** Default provider reasoning mode. A per-call request value takes precedence. */\n thinking?: LlmThinkingMode\n /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */\n fetch?: typeof fetch\n /**\n * Optional raw HTTP capture sink. When provided, every request, response,\n * and error (across all retry attempts) is recorded to the sink, with auth\n * headers and credential-shaped body fields redacted by default. This is\n * the layer-1 forensics primitive: structured `LlmSpan`s record intent,\n * raw events record what actually crossed the wire.\n */\n rawSink?: RawProviderSink\n /**\n * Logical provider id attached to raw events. When omitted, derived from\n * `baseUrl` via `providerFromBaseUrl`.\n */\n provider?: string\n /** Trace context attached to raw events; populated by emitter-aware callers. */\n traceContext?: { runId?: string; spanId?: string }\n /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */\n redactor?: ProviderRedactor\n /**\n * Reject a response whose echoed model is not the model that was requested.\n * A routing gateway can accept one id and answer from another, which\n * silently invalidates every per-model and per-family claim downstream.\n * `true` uses the strict default (aliases pass, substitutions and\n * unidentified responses throw `ModelSubstitutionError`); pass an options\n * object to relax a specific case. Off by default — turning it on for a\n * measurement run is the point.\n */\n assertServedModel?: boolean | AssertServedModelOptions\n}\n\n// ─── Internals ──────────────────────────────────────────────────────────\n\nconst DEFAULT_BASE_URL = 'https://router.tangle.tools/v1'\n// Flagship / reasoning models routinely take several minutes on large prompts (a\n// reflection over many failures, a long tool transcript). A tight cap aborts a\n// legitimately-slow but healthy call — and because every retry attempt re-uses\n// the same window, such a model aborts on ALL attempts and the loop throws. The\n// default is generous enough to let those complete, bounded enough that a truly\n// hung call still fails over after retries, and tunable per deployment via\n// TANGLE_LLM_TIMEOUT_MS. Per-call `req.timeoutMs` / `opts.defaultTimeoutMs`\n// still win for callers that know their model's latency.\nconst DEFAULT_TIMEOUT_MS = Number(process.env.TANGLE_LLM_TIMEOUT_MS) || 300_000\nconst DEFAULT_MAXIMUM_ATTEMPTS =\n process.env.TANGLE_LLM_MAXIMUM_ATTEMPTS === undefined\n ? 3\n : Number(process.env.TANGLE_LLM_MAXIMUM_ATTEMPTS)\n\nfunction resolveMaximumAttempts(configured: number | undefined): number {\n const attempts = configured ?? DEFAULT_MAXIMUM_ATTEMPTS\n if (!Number.isInteger(attempts) || attempts <= 0) {\n throw new RangeError('LLM maximum attempts must be a positive integer')\n }\n return attempts\n}\n\nfunction providerTokenCount(value: unknown): number | undefined {\n return typeof value === 'number' && Number.isSafeInteger(value) && value >= 0 ? value : undefined\n}\n\nconst RETRYABLE_STATUS = new Set([429, 502, 503, 504])\n\n/**\n * Transient transport/network error signatures, matched against an error's\n * name, message, and `code`. Covers fetch/undici network failures, aborts\n * and timeouts, and — critically — HTTP/2 transport faults a keep-alive\n * connection raises mid-response: `terminated`, `NGHTTP2_INTERNAL_ERROR`,\n * `UND_ERR_*`, `other side closed`. Those last ones carry no clean HTTP\n * status; unrecognised, they escape the retry loop and surface as an\n * uncaught rejection.\n */\nconst TRANSIENT_ERROR_PATTERNS: readonly RegExp[] = [\n /AbortError/i,\n /TimeoutError/i,\n /this operation was aborted/i,\n /fetch failed/i,\n /ECONNRESET/i,\n /ETIMEDOUT/i,\n /EAI_AGAIN/i,\n /socket hang up/i,\n /stream.*ended.*unexpectedly/i,\n /terminated/i,\n /other side closed/i,\n /NGHTTP2/i,\n /UND_ERR/i,\n]\n\n/**\n * True when an error is a transient transport/network fault worth retrying,\n * as opposed to a deterministic failure (4xx schema reject, JSON parse) that\n * a retry cannot fix. Inspects `LlmCallError.status`, then the error's\n * name/message/code, then recurses into `error.cause` — undici nests the\n * real socket fault one or more levels under `.cause`.\n *\n * This is the retry classifier for the package: `callLlm` and\n * `withJudgeRetry` both route through it, so connection failures are treated\n * consistently across transports.\n */\nexport function isTransientLlmError(err: unknown): boolean {\n return classifyTransient(err, 0)\n}\n\nfunction classifyTransient(err: unknown, depth: number): boolean {\n if (err instanceof LlmCallError) return RETRYABLE_STATUS.has(err.status)\n if (!(err instanceof Error)) return false\n // Foreign transport errors can carry a numeric HTTP status without being an\n // LlmCallError. A retryable status is decisive.\n const status = (err as { status?: unknown }).status\n if (typeof status === 'number' && RETRYABLE_STATUS.has(status)) return true\n const code = (err as { code?: unknown }).code\n const haystack = `${err.name}\\n${err.message}\\n${typeof code === 'string' ? code : ''}`\n if (TRANSIENT_ERROR_PATTERNS.some((p) => p.test(haystack))) return true\n const cause = (err as { cause?: unknown }).cause\n if (depth < 4 && cause instanceof Error && cause !== err) {\n return classifyTransient(cause, depth + 1)\n }\n return false\n}\n\nfunction parseRetryAfter(headers: Headers): number | null {\n const h = headers.get('retry-after')\n if (!h) return null\n const asNumber = Number(h)\n if (Number.isFinite(asNumber) && asNumber > 0) return asNumber * 1000\n const asDate = Date.parse(h)\n if (Number.isFinite(asDate)) return Math.max(0, asDate - Date.now())\n return null\n}\n\n/** Exponential backoff: 500ms, 1s, 2s, 4s, ... capped at 16s. Attempt is 0-indexed. */\nexport function backoffMs(attempt: number): number {\n return Math.min(500 * 2 ** attempt, 16_000)\n}\n\nfunction buildHeaders(opts: LlmClientOptions): Record<string, string> {\n const headers: Record<string, string> = {\n 'Content-Type': 'application/json',\n Accept: 'application/json',\n }\n if (opts.authHeader) {\n headers[opts.authHeader.name] = opts.authHeader.value\n } else if (opts.bearer || opts.apiKey) {\n headers.Authorization = `Bearer ${opts.bearer ?? opts.apiKey}`\n }\n if (opts.idempotencyKey) headers['Idempotency-Key'] = opts.idempotencyKey\n return headers\n}\n\nfunction isSchemaRejection(status: number, body: string): boolean {\n if (status !== 400) return false\n const lower = body.toLowerCase()\n return (\n lower.includes('response_format') ||\n lower.includes('json_schema') ||\n lower.includes('is unavailable') ||\n lower.includes('not supported')\n )\n}\n\nfunction isTemperatureOneRejection(status: number, body: string): boolean {\n if (status !== 400 || !/temperature/i.test(body)) return false\n return (\n /temperature[^.\\n]{0,120}\\b(?:only|must|should|required|requires?)\\b[^.\\n]{0,40}\\b1(?:\\.0+)?\\b/i.test(\n body,\n ) || /\\bonly\\s+1(?:\\.0+)?\\s+is\\s+allowed\\b[^.\\n]{0,120}\\btemperature\\b/i.test(body)\n )\n}\n\nfunction buildBody(\n req: LlmCallRequest,\n forceJsonObject: boolean,\n defaultThinking?: LlmThinkingMode,\n): Record<string, unknown> {\n const body: Record<string, unknown> = {\n model: req.model,\n messages: req.messages,\n temperature: req.temperature ?? 0,\n }\n if (req.maxTokens != null) {\n if (usesMaxCompletionTokens(req.model)) body.max_completion_tokens = req.maxTokens\n else body.max_tokens = req.maxTokens\n }\n const thinking = req.thinking ?? defaultThinking\n if (thinking !== undefined) {\n body.thinking = { type: thinking }\n }\n\n if (req.jsonSchema && !forceJsonObject) {\n body.response_format = {\n type: 'json_schema',\n json_schema: { name: req.jsonSchema.name, schema: req.jsonSchema.schema, strict: true },\n }\n } else if (req.jsonMode || req.jsonSchema) {\n body.response_format = { type: 'json_object' }\n }\n\n return body\n}\n\nfunction usesMaxCompletionTokens(model: string): boolean {\n return /^gpt-5(?:[.-]|$)/i.test(model)\n}\n\nasync function sleep(ms: number): Promise<void> {\n return new Promise((resolve) => setTimeout(resolve, ms))\n}\n\n/**\n * Combine the per-attempt timeout signal with an optional caller signal into\n * one signal the fetch listens on. Prefers the native `AbortSignal.any`; falls\n * back to manual wiring on runtimes that predate it. The caller signal is also\n * propagated to the timeout controller so aborting it cancels the in-flight\n * fetch immediately.\n */\nfunction linkSignals(timeoutController: AbortController, caller?: AbortSignal): AbortSignal {\n if (!caller) return timeoutController.signal\n if (typeof (AbortSignal as { any?: unknown }).any === 'function') {\n return AbortSignal.any([timeoutController.signal, caller])\n }\n if (caller.aborted) {\n timeoutController.abort()\n } else {\n caller.addEventListener('abort', () => timeoutController.abort(), { once: true })\n }\n return timeoutController.signal\n}\n\n/** True once the cross-attempt wall-clock budget (if any) is exhausted. */\nfunction deadlineExceeded(start: number, deadlineMs: number | undefined): boolean {\n return deadlineMs != null && Date.now() - start >= deadlineMs\n}\n\n// ─── Public API ─────────────────────────────────────────────────────────\n\n/**\n * Strip a ```json / ``` code fence if the model emitted one.\n * Idempotent for naked JSON. Some models (claude-code via router, certain\n * deepseek models) wrap output even under json_object.\n */\nexport function stripFencedJson(raw: string): string {\n const trimmed = raw.trim()\n const m = trimmed.match(/^```(?:json)?\\s*\\n?([\\s\\S]*?)\\n?```\\s*$/)\n return m ? m[1]!.trim() : trimmed\n}\n\nexport function extractJsonPayload(raw: string): string {\n const stripped = stripFencedJson(raw)\n try {\n JSON.parse(stripped)\n return stripped\n } catch {\n // A response that declares a JSON root must parse as that complete root.\n // Scanning onward could turn a truncated object into one of its valid nested\n // arrays or objects and silently change the response schema.\n if (stripped.startsWith('{') || stripped.startsWith('[')) return stripped\n }\n\n // Only prose-leading responses may contain a recoverable JSON payload.\n const starts = [...stripped.matchAll(/[[{]/g)]\n .map((match) => match.index)\n .filter((index) => index != null)\n for (const start of starts) {\n const candidate = extractBalancedJson(stripped, start)\n if (!candidate) continue\n try {\n JSON.parse(candidate)\n return candidate\n } catch {\n // Keep scanning; earlier braces may belong to prose.\n }\n }\n\n return stripped\n}\n\nfunction extractBalancedJson(input: string, start: number): string | null {\n const opener = input[start]\n const closer = opener === '{' ? '}' : opener === '[' ? ']' : null\n if (!closer) return null\n\n const stack: string[] = [closer]\n let isInString = false\n let isEscaped = false\n\n for (let i = start + 1; i < input.length; i++) {\n const char = input[i]!\n if (isEscaped) {\n isEscaped = false\n continue\n }\n if (char === '\\\\') {\n isEscaped = isInString\n continue\n }\n if (char === '\"') {\n isInString = !isInString\n continue\n }\n if (isInString) continue\n\n if (char === '{') stack.push('}')\n else if (char === '[') stack.push(']')\n else if (char === stack[stack.length - 1]) {\n stack.pop()\n if (stack.length === 0) return input.slice(start, i + 1)\n }\n }\n\n return null\n}\n\n/**\n * Low-level call. Returns raw content + usage + cost. Retries on transient\n * failures; does NOT degrade schema here — callers that want graceful\n * degrade use `callLlmJson`.\n */\nexport async function callLlm(\n req: LlmCallRequest,\n opts: LlmClientOptions = {},\n): Promise<LlmCallResult> {\n const baseUrl = (opts.baseUrl ?? DEFAULT_BASE_URL).replace(/\\/+$/, '')\n const url = `${baseUrl}/chat/completions`\n const endpoint = '/chat/completions'\n const timeoutMs = req.timeoutMs ?? opts.defaultTimeoutMs ?? DEFAULT_TIMEOUT_MS\n const maximumAttempts = resolveMaximumAttempts(opts.maximumAttempts)\n const fetchFn = opts.fetch ?? globalThis.fetch\n const headers = buildHeaders(opts)\n const provider = opts.provider ?? providerFromBaseUrl(baseUrl)\n const sink = opts.rawSink\n const redactor = opts.redactor ?? defaultProviderRedactor\n const traceContext = opts.traceContext\n const callerSignal = opts.signal\n const deadlineMs = opts.deadlineMs\n const deadlineStart = Date.now()\n if (opts.customTokenPricing) {\n costForTokenPricing(opts.customTokenPricing, { inputTokens: 0, outputTokens: 0 })\n }\n\n let lastErr: unknown\n let effectiveRequest = req\n for (let attempt = 0; attempt < maximumAttempts; attempt++) {\n // A caller cancel is fatal — never retried. Checking before each attempt\n // means an already-aborted signal short-circuits without firing fetch.\n if (callerSignal?.aborted) {\n throw new DOMException('callLlm aborted by caller signal', 'AbortError')\n }\n // Stop retrying once the cross-attempt budget is spent rather than burning\n // a full per-attempt timeout on each remaining retry.\n if (attempt > 0 && deadlineExceeded(deadlineStart, deadlineMs)) {\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n }\n const controller = new AbortController()\n const attemptSignal = linkSignals(controller, callerSignal)\n const timeoutHandle = setTimeout(() => controller.abort(), timeoutMs)\n const started = Date.now()\n const requestBody = buildBody(\n effectiveRequest,\n opts.jsonSchemaTransport === 'json-object',\n opts.thinking,\n )\n let attemptErrorRecorded = false\n if (sink) {\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'request',\n timestamp: started,\n requestHeaders: headers,\n requestBody,\n redactedFields: [],\n })\n }\n\n try {\n const res = await fetchFn(url, {\n method: 'POST',\n headers,\n body: JSON.stringify(requestBody),\n signal: attemptSignal,\n })\n clearTimeout(timeoutHandle)\n const responseHeaders = sink ? headersToObject(res.headers) : undefined\n\n if (!res.ok) {\n const body = await res.text()\n if (sink) {\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n statusCode: res.status,\n responseHeaders,\n responseBody: body,\n errorMessage: `HTTP ${res.status}`,\n redactedFields: [],\n })\n attemptErrorRecorded = true\n }\n const err = new LlmCallError(\n `LLM call failed with HTTP ${res.status}`,\n res.status,\n body,\n req.model,\n )\n if (\n isTemperatureOneRejection(res.status, body) &&\n effectiveRequest.temperature !== 1 &&\n attempt < maximumAttempts - 1 &&\n !deadlineExceeded(deadlineStart, deadlineMs)\n ) {\n lastErr = err\n effectiveRequest = { ...effectiveRequest, temperature: 1 }\n continue\n }\n if (\n RETRYABLE_STATUS.has(res.status) &&\n attempt < maximumAttempts - 1 &&\n !deadlineExceeded(deadlineStart, deadlineMs)\n ) {\n lastErr = err\n const retryAfter = parseRetryAfter(res.headers)\n await sleep(retryAfter ?? backoffMs(attempt))\n continue\n }\n throw err\n }\n\n const text = await res.text()\n let json: Record<string, unknown>\n try {\n json = JSON.parse(text) as Record<string, unknown>\n } catch (parseErr) {\n if (sink) {\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n statusCode: res.status,\n responseHeaders,\n responseBody: text,\n errorMessage: `non-JSON response: ${parseErr instanceof Error ? parseErr.message : String(parseErr)}`,\n redactedFields: [],\n })\n attemptErrorRecorded = true\n }\n throw parseErr\n }\n if (sink) {\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'response',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n statusCode: res.status,\n responseHeaders,\n responseBody: json,\n redactedFields: [],\n })\n }\n const choice = (\n json.choices as\n | Array<{ message?: { content?: string }; finish_reason?: string | null }>\n | undefined\n )?.[0]\n const usageRaw =\n json.usage && typeof json.usage === 'object' && !Array.isArray(json.usage)\n ? (json.usage as Record<string, unknown>)\n : undefined\n const promptTokens = providerTokenCount(usageRaw?.prompt_tokens)\n const completionTokens = providerTokenCount(usageRaw?.completion_tokens)\n const totalTokens = providerTokenCount(usageRaw?.total_tokens)\n const completionDetails =\n usageRaw?.completion_tokens_details &&\n typeof usageRaw.completion_tokens_details === 'object' &&\n !Array.isArray(usageRaw.completion_tokens_details)\n ? (usageRaw.completion_tokens_details as Record<string, unknown>)\n : undefined\n const reasoningRaw = completionDetails?.reasoning_tokens\n const reasoningTokens =\n reasoningRaw === undefined ? undefined : providerTokenCount(reasoningRaw)\n const cachedRaw =\n usageRaw?.prompt_tokens_details &&\n typeof usageRaw.prompt_tokens_details === 'object' &&\n !Array.isArray(usageRaw.prompt_tokens_details)\n ? (usageRaw.prompt_tokens_details as Record<string, unknown>).cached_tokens\n : undefined\n const cachedPromptTokens = cachedRaw === undefined ? undefined : providerTokenCount(cachedRaw)\n const usageCaptured =\n promptTokens !== undefined &&\n completionTokens !== undefined &&\n (reasoningRaw === undefined ||\n (reasoningTokens !== undefined && reasoningTokens <= completionTokens)) &&\n (cachedRaw === undefined ||\n (cachedPromptTokens !== undefined && cachedPromptTokens <= promptTokens)) &&\n (totalTokens === undefined || totalTokens === promptTokens + completionTokens)\n const costFromProxy = (json._response_cost ?? json.cost_usd) as number | undefined\n const content = choice?.message?.content ?? ''\n\n const configuredCost =\n typeof costFromProxy !== 'number' && usageCaptured && opts.customTokenPricing\n ? costForTokenPricing(opts.customTokenPricing, {\n inputTokens: promptTokens! - (cachedPromptTokens ?? 0),\n ...(cachedPromptTokens ? { cachedTokens: cachedPromptTokens } : {}),\n outputTokens: completionTokens!,\n })\n : undefined\n\n // The echoed id, kept separate from the attribution id: a provider that\n // omits it must read as \"unproven\", never as \"the model I asked for\".\n const servedModel =\n typeof json.model === 'string' && json.model.trim() !== '' ? json.model : null\n if (opts.assertServedModel) {\n assertServedModelIdentity(\n req.model,\n servedModel,\n opts.assertServedModel === true ? {} : opts.assertServedModel,\n )\n }\n\n return {\n content,\n finishReason: choice?.finish_reason ?? null,\n contentEmpty: content.trim().length === 0,\n usage: {\n promptTokens: promptTokens ?? 0,\n completionTokens: completionTokens ?? 0,\n totalTokens: totalTokens ?? (promptTokens ?? 0) + (completionTokens ?? 0),\n captured: usageCaptured,\n reasoningTokens,\n cachedPromptTokens,\n },\n costUsd: typeof costFromProxy === 'number' ? costFromProxy : (configuredCost ?? null),\n model: servedModel ?? req.model,\n servedModel,\n durationMs: Date.now() - started,\n raw: json,\n }\n } catch (err) {\n clearTimeout(timeoutHandle)\n lastErr = err\n // A caller cancel is fatal even though an AbortError matches the\n // transient patterns — a cancelled call must surface immediately, not\n // be retried against the same dead intent.\n if (callerSignal?.aborted) {\n if (sink && !attemptErrorRecorded) {\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n }\n throw err\n }\n if (sink && !attemptErrorRecorded) {\n // Record only if neither the !res.ok branch nor the JSON.parse catch\n // already produced an error event for this attempt. Covers network\n // failures, timeouts, and aborts.\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n }\n if (\n attempt < maximumAttempts - 1 &&\n isTransientLlmError(err) &&\n !deadlineExceeded(deadlineStart, deadlineMs)\n ) {\n await sleep(backoffMs(attempt))\n continue\n }\n throw err\n }\n }\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n}\n\nasync function recordRaw(\n sink: RawProviderSink,\n redactor: ProviderRedactor,\n event: RawProviderEvent,\n): Promise<void> {\n // Errors from sinks must not crash the LLM call. Forensic capture is\n // best-effort; the structured trace is the system of record.\n try {\n await sink.record(redactor(event))\n } catch {\n // Intentionally swallowed.\n }\n}\n\nfunction headersToObject(h: Headers): Record<string, string> {\n const out: Record<string, string> = {}\n h.forEach((value, key) => {\n out[key] = value\n })\n return out\n}\n\nfunction cryptoEventId(): string {\n if (typeof globalThis.crypto?.randomUUID === 'function') return globalThis.crypto.randomUUID()\n return `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 10)}`\n}\n\n/**\n * Structured-output call. Returns parsed JSON plus the raw result envelope.\n * Degrades `jsonSchema` → `jsonMode` on a 400 that names the schema param —\n * critical for deepseek-v3/v4, kimi-k2.6, and other models that don't accept\n * the `response_format.json_schema` shape but DO accept `json_object`.\n */\nexport async function callLlmJson<T = unknown>(\n req: LlmCallRequest,\n opts: LlmClientOptions = {},\n): Promise<{ value: T; result: LlmCallResult }> {\n const result = await callLlmStructured(req, opts)\n const value = parseJsonResult<T>(result, opts.jsonPayloadMode ?? 'extract')\n return { value, result }\n}\n\n/** Shared schema-to-JSON-mode fallback that preserves the raw result. */\nasync function callLlmStructured(\n req: LlmCallRequest,\n opts: LlmClientOptions = {},\n): Promise<LlmCallResult> {\n try {\n return await callLlm({ ...req, jsonMode: req.jsonMode ?? !req.jsonSchema }, opts)\n } catch (err) {\n if (\n opts.jsonSchemaTransport !== 'json-object' &&\n err instanceof LlmCallError &&\n isSchemaRejection(err.status, err.body) &&\n req.jsonSchema\n ) {\n const degradedReq: LlmCallRequest = { ...req, jsonMode: true, jsonSchema: undefined }\n return await callLlm(degradedReq, opts)\n }\n throw err\n }\n}\n\nfunction parseJsonResult<T>(\n result: LlmCallResult,\n jsonPayloadMode: NonNullable<LlmClientOptions['jsonPayloadMode']>,\n): T {\n try {\n if (result.finishReason === 'length') {\n throw new Error(\n `LLM returned truncated JSON content (model=${result.model}, finishReason=length)`,\n )\n }\n return parseJsonSafely<T>(result.content, result.model, jsonPayloadMode)\n } catch (error) {\n if (error instanceof LlmResponseError) throw error\n const cause = error instanceof Error ? error : new Error(String(error))\n throw new LlmResponseError(cause.message, result, { cause })\n }\n}\n\nfunction parseJsonSafely<T>(\n content: string,\n model: string,\n jsonPayloadMode: NonNullable<LlmClientOptions['jsonPayloadMode']>,\n): T {\n const payload = jsonPayloadMode === 'exact' ? content : extractJsonPayload(content)\n try {\n return JSON.parse(payload) as T\n } catch {\n throw new Error(`LLM returned non-JSON content (model=${model})`)\n }\n}\n\n// ─── Route assertion ────────────────────────────────────────────────────\n\nexport type LlmRouteAssertionReason =\n | 'no_explicit_base_url'\n | 'base_url_blocked'\n | 'base_url_not_allowed'\n | 'no_auth'\n | 'wrong_provider'\n\nexport class LlmRouteAssertionError extends CaptureIntegrityError {\n constructor(\n message: string,\n public readonly reason: LlmRouteAssertionReason,\n public readonly baseUrl: string,\n ) {\n super(message)\n }\n}\n\nexport interface LlmRouteRequirements {\n /**\n * Throw if `opts.baseUrl` is undefined, i.e. the call would fall back to\n * `DEFAULT_BASE_URL`. Set this for evaluation runs where silently using\n * the public/free-tier router is a defect — the launch reviewer needs to\n * know exactly which provider answered.\n */\n requireExplicitBaseUrl?: boolean\n /**\n * Allowlist of acceptable base URLs. Strings match by prefix\n * (case-insensitive); RegExps test against the full base URL.\n */\n allowedBaseUrls?: Array<string | RegExp>\n /** Blocklist that takes precedence over `allowedBaseUrls`. */\n blockedBaseUrls?: Array<string | RegExp>\n /** Throw if no auth header / api key is configured. */\n requireAuth?: boolean\n /**\n * Logical provider id the configured `baseUrl` is expected to match (via\n * `providerFromBaseUrl`). Mainly useful when paired with `requireExplicitBaseUrl`.\n */\n expectedProvider?: string\n}\n\n/**\n * Fail-loud assertion that the configured LLM client points at the route\n * the caller intends. Designed for the matrix-runner preflight: invoke\n * once before any LLM call to catch misconfiguration before a sweep burns\n * dollars on the wrong provider.\n *\n * Throws `LlmRouteAssertionError`. Pure — no I/O — so it's safe to call\n * from constructors and CI gates.\n */\nexport function assertLlmRoute(opts: LlmClientOptions, req: LlmRouteRequirements = {}): void {\n const baseUrlExplicit = opts.baseUrl !== undefined\n const baseUrl = (opts.baseUrl ?? DEFAULT_BASE_URL).replace(/\\/+$/, '')\n\n if (req.requireExplicitBaseUrl && !baseUrlExplicit) {\n throw new LlmRouteAssertionError(\n `assertLlmRoute: requireExplicitBaseUrl set but opts.baseUrl is undefined; would fall back to ${DEFAULT_BASE_URL}.`,\n 'no_explicit_base_url',\n baseUrl,\n )\n }\n\n if (req.blockedBaseUrls?.some((p) => matchUrl(baseUrl, p))) {\n throw new LlmRouteAssertionError(\n `assertLlmRoute: baseUrl ${baseUrl} matches a blocked pattern.`,\n 'base_url_blocked',\n baseUrl,\n )\n }\n\n if (req.allowedBaseUrls && req.allowedBaseUrls.length > 0) {\n const ok = req.allowedBaseUrls.some((p) => matchUrl(baseUrl, p))\n if (!ok) {\n throw new LlmRouteAssertionError(\n `assertLlmRoute: baseUrl ${baseUrl} is not in the allowed list (${req.allowedBaseUrls.map(describePattern).join(', ')}).`,\n 'base_url_not_allowed',\n baseUrl,\n )\n }\n }\n\n if (req.requireAuth && !opts.apiKey && !opts.bearer && !opts.authHeader) {\n throw new LlmRouteAssertionError(\n `assertLlmRoute: requireAuth set but no apiKey, bearer, or authHeader was supplied.`,\n 'no_auth',\n baseUrl,\n )\n }\n\n if (req.expectedProvider) {\n const actual = opts.provider ?? providerFromBaseUrl(baseUrl)\n if (actual !== req.expectedProvider) {\n throw new LlmRouteAssertionError(\n `assertLlmRoute: expected provider ${req.expectedProvider} but baseUrl ${baseUrl} resolves to ${actual}.`,\n 'wrong_provider',\n baseUrl,\n )\n }\n }\n}\n\nfunction matchUrl(url: string, pattern: string | RegExp): boolean {\n if (pattern instanceof RegExp) return pattern.test(url)\n return url.toLowerCase().startsWith(pattern.toLowerCase())\n}\n\nfunction describePattern(p: string | RegExp): string {\n return p instanceof RegExp ? p.source : p\n}\n\n/**\n * Probe whether a model is reachable. Returns latency + null error on\n * success; `ok=false` + error message on any failure (HTTP, timeout,\n * network, parse). Designed for sweep preflights — fail loud at the\n * boundary before burning a 30-leaf run on a misconfigured router.\n *\n * Sends a tiny `ping` message with `maxTokens = PROBE_MAX_TOKENS`. Reasoning\n * models (glm-5.1, deepseek-v4) can burn the entire budget on internal\n * reasoning for short prompts, so don't tighten this further — the shared\n * constant keeps this probe and `preflightModels` on one answer. We don't\n * validate content.\n *\n * Reachability and identity are separate answers: `ok` means the route\n * answered, `servedModel` / `substituted` say WHICH model answered. A gateway\n * that serves another provider's model returns `ok: true` with\n * `substituted: true` — inspect both before treating the id as measured.\n */\nexport async function probeLlm(\n model: string,\n opts: LlmClientOptions & { timeoutMs?: number } = {},\n): Promise<{\n ok: boolean\n latencyMs: number\n error: string | null\n /** Id echoed by the provider; `null` when it sent none or the probe failed. */\n servedModel: string | null\n /** True when the echoed id is a different model than `model` (or absent). */\n substituted: boolean\n}> {\n const start = Date.now()\n try {\n const result = await callLlm(\n {\n model,\n messages: [{ role: 'user', content: 'ping' }],\n maxTokens: PROBE_MAX_TOKENS,\n timeoutMs: opts.timeoutMs ?? 30_000,\n },\n opts,\n )\n return {\n ok: true,\n latencyMs: Date.now() - start,\n error: null,\n servedModel: result.servedModel ?? null,\n substituted: checkServedModel(model, result.servedModel).substituted,\n }\n } catch (err) {\n return {\n ok: false,\n latencyMs: Date.now() - start,\n error: err instanceof Error ? err.message : String(err),\n servedModel: null,\n substituted: false,\n }\n }\n}\n\n/**\n * Stateful client — construct once with defaults, call many times.\n * Thin wrapper around the free functions; exists for callers that want\n * to inject a single configured instance into multiple primitives.\n */\nexport class LlmClient {\n readonly maximumAttempts: number\n private readonly opts: LlmClientOptions\n\n constructor(opts: LlmClientOptions = {}) {\n this.opts = opts\n this.maximumAttempts = resolveMaximumAttempts(opts.maximumAttempts)\n }\n\n call(req: LlmCallRequest, per?: LlmClientOptions): Promise<LlmCallResult> {\n const options = { ...this.opts, ...per }\n return req.jsonSchema ? callLlmStructured(req, options) : callLlm(req, options)\n }\n\n callJson<T = unknown>(\n req: LlmCallRequest,\n per?: LlmClientOptions,\n ): Promise<{ value: T; result: LlmCallResult }> {\n return callLlmJson<T>(req, { ...this.opts, ...per })\n }\n}\n"],"mappings":";;;;;AA2BA,MAAM,kBAA+C;CACnD,WAAW;CACX,QAAQ;CACR,gBAAgB;CAChB,QAAQ;CACR,iBAAiB;CACjB,MAAM;CACN,cAAc;CACd,SAAS;CACT,WAAW;CACX,UAAU;CACV,KAAK;CACL,MAAM;CACN,SAAS;CACT,QAAQ;CACR,QAAQ;CACR,SAAS;CACT,UAAU;CACV,YAAY;CACZ,MAAM;CACN,aAAa;CACb,OAAO;CACP,SAAS;CACT,KAAK;CACL,QAAQ;CACR,KAAK;AACP;;AAGA,MAAM,gBAA8C;CAClD,CAAC,WAAW,WAAW;CACvB,CAAC,0DAA0D,QAAQ;CACnE,CAAC,4BAA4B,QAAQ;CACrC,CAAC,UAAU,MAAM;CACjB,CAAC,oCAAoC,SAAS;CAC9C,CAAC,aAAa,UAAU;CACxB,CAAC,SAAS,KAAK;CACf,CAAC,SAAS,MAAM;CAChB,CAAC,oBAAoB,QAAQ;CAC7B,CAAC,qBAAqB,QAAQ;CAC9B,CAAC,sBAAsB,UAAU;CACjC,CAAC,4BAA4B,OAAO;AACtC;;;;;;;AAQA,SAAgB,YAAY,SAA8B;CACxD,MAAM,KAAK,QAAQ,KAAK,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,EAAE,CAAE,YAAY;CACrD,MAAM,QAAQ,GAAG,QAAQ,GAAG;CAC5B,IAAI,QAAQ,GAAG;EACb,MAAM,SAAS,GAAG,MAAM,GAAG,KAAK;EAChC,MAAM,SAAS,gBAAgB;EAC/B,IAAI,QAAQ,OAAO;CACrB;CACA,KAAK,MAAM,CAAC,SAAS,WAAW,eAC9B,IAAI,QAAQ,KAAK,EAAE,GAAG,OAAO;CAE/B,OAAO;AACT;AAWA,IAAa,mBAAb,cAAsC,MAAM;CAGxB;CACA;CAHlB,YACE,SACA,UACA,QACA;EACA,MAAM,OAAO;EAHG,KAAA,WAAA;EACA,KAAA,SAAA;EAGhB,KAAK,OAAO;CACd;AACF;;;;;;;;;;;;;;AAeA,SAAgB,kBACd,QACA,OAAiC,CAAC,GACnB;CACf,MAAM,cAAc,KAAK,eAAe;CACxC,MAAM,2BAAW,IAAI,IAAiB;CACtC,KAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,IAAI,YAAY,CAAC;EACvB,IAAI,MAAM,aAAa,CAAC,KAAK,cAAc;EAC3C,SAAS,IAAI,CAAC;CAChB;CACA,MAAM,OAAO,CAAC,GAAG,QAAQ,CAAC,CAAC,KAAK;CAChC,IAAI,KAAK,SAAS,aAChB,MAAM,IAAI,iBACR,wBAAwB,KAAK,OAAO,iBAAiB,KAAK,WAAW,IAAI,MAAM,MAAM,IAC/E,KAAK,KAAK,IAAI,KAAK,OAAO,QAAQ,YAAY,kFAEpD,MACA,MACF;CAEF,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACzGA,MAAa,mBAAmB;;;;;;;;;;;;;;;;AA2ChC,SAAgB,iBAAiB,SAAyB;CACxD,IAAI,KAAK,QAAQ,KAAK,CAAC,CAAC,YAAY;CACpC,MAAM,KAAK,GAAG,QAAQ,GAAG;CACzB,IAAI,KAAK,GAAG,KAAK,GAAG,MAAM,GAAG,EAAE;CAC/B,MAAM,QAAQ,GAAG,QAAQ,GAAG;CAC5B,IAAI,QAAQ,GAAG,KAAK,GAAG,MAAM,GAAG,KAAK;CACrC,MAAM,QAAQ,GAAG,YAAY,GAAG;CAChC,IAAI,SAAS,GAAG,KAAK,GAAG,MAAM,QAAQ,CAAC;CACvC,OAAO,GACJ,QAAQ,uBAAuB,EAAE,CAAC,CAClC,QAAQ,WAAW,EAAE,CAAC,CACtB,QAAQ,SAAS,GAAG,CAAC,CACrB,QAAQ,OAAO,EAAE,CAAC,CAClB,KAAK;AACV;;;;;;;;;;;AAYA,SAAgB,iBACd,WACA,QACkB;CAClB,MAAM,kBAAkB,YAAY,SAAS;CAC7C,IAAI,WAAW,QAAQ,WAAW,KAAA,KAAa,OAAO,KAAK,MAAM,IAC/D,OAAO;EACL;EACA,QAAQ;EACR;EACA,cAAc;EACd,SAAS;EACT,aAAa;CACf;CAEF,MAAM,eAAe,YAAY,MAAM;CACvC,IAAI,UAAU,KAAK,CAAC,CAAC,YAAY,MAAM,OAAO,KAAK,CAAC,CAAC,YAAY,GAC/D,OAAO;EACL;EACA;EACA;EACA;EACA,SAAS;EACT,aAAa;CACf;CAEF,IAAI,iBAAiB,SAAS,MAAM,iBAAiB,MAAM,GACzD,OAAO;EACL;EACA;EACA;EACA;EACA,SAAS;EACT,aAAa;CACf;CAEF,OAAO;EACL;EACA;EACA;EACA;EACA,SACE,oBAAoB,eAAe,8BAA8B;EACnE,aAAa;CACf;AACF;AAEA,IAAa,yBAAb,cAA4C,eAAe;CAGvC;CAFlB,YACE,SACA,QACA;EACA,MAAM,sBAAsB,OAAO;EAFnB,KAAA,SAAA;EAGhB,KAAK,OAAO;CACd;AACF;;;;;;AAuBA,SAAgB,sBACd,OACA,OAAiC,CAAC,GACzB;CACT,QAAQ,MAAM,SAAd;EACE,KAAK;EACL,KAAK,SACH,OAAO;EACT,KAAK,cACH,OAAO,KAAK,oBAAoB;EAClC,KAAK,6BACH,OAAO,KAAK,sBAAsB;EACpC,SACE,OAAO;CACX;AACF;AAEA,SAAS,SAAS,OAAiC;CACjD,IAAI,MAAM,YAAY,cACpB,OAAO,GAAG,MAAM,UAAU;CAC5B,OACE,GAAG,MAAM,UAAU,IAAI,MAAM,gBAAgB,aAAa,MAAM,OAAO,IACnE,MAAM,aAAa,KAAK,MAAM,QAAQ;AAE9C;;;;;;AAOA,SAAgB,kBACd,WACA,QACA,OAAiC,CAAC,GAChB;CAClB,MAAM,QAAQ,iBAAiB,WAAW,MAAM;CAChD,IAAI,sBAAsB,OAAO,IAAI,GAAG,OAAO;CAE/C,MAAM,IAAI,uBACR,GAFa,KAAK,UAAU,GAAG,KAAK,QAAQ,MAAM,GAExC,uBAAuB,SAAS,KAAK,EAAE,gIAEjD,CAAC,KAAK,CACR;AACF;;;;;AAMA,SAAgB,mBACd,OACA,OAAiC,CAAC,GACd;CACpB,MAAM,SAAS,MAAM,KAAK,MAAM,iBAAiB,EAAE,WAAW,EAAE,MAAM,CAAC;CACvE,MAAM,MAAM,OAAO,QAAQ,MAAM,CAAC,sBAAsB,GAAG,IAAI,CAAC;CAChE,IAAI,IAAI,SAAS,GAEf,MAAM,IAAI,uBACR,GAFa,KAAK,UAAU,GAAG,KAAK,QAAQ,MAAM,KAEtC,IAAI,OAAO,GAAG,OAAO,OAAO,+DACvB,IAAI,IAAI,QAAQ,CAAC,CAAC,KAAK,IAAI,EAAE,6FAE9C,MACF;CAEF,OAAO;AACT;AASA,IAAa,yBAAb,cAA4C,eAAe;CAGvC;CACA;CAHlB,YACE,SACA,UACA,QACA;EACA,MAAM,sBAAsB,OAAO;EAHnB,KAAA,WAAA;EACA,KAAA,SAAA;EAGhB,KAAK,OAAO;CACd;AACF;;;;;;;;;;AAWA,SAAgB,wBACd,OACA,OAAuC,CAAC,GACzB;CACf,MAAM,SAAS,mBAAmB,OAAO,IAAI;CAC7C,MAAM,2BAAW,IAAI,IAAiB;CACtC,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,SAAS,MAAM;EACrB,IAAI,WAAW,MAAM;EACrB,IAAI,WAAW,aAAa,CAAC,KAAK,cAAc;EAChD,SAAS,IAAI,MAAM;CACrB;CACA,MAAM,OAAO,CAAC,GAAG,QAAQ,CAAC,CAAC,KAAK;CAChC,MAAM,cAAc,KAAK,eAAe;CACxC,IAAI,KAAK,SAAS,aAEhB,MAAM,IAAI,uBACR,GAFa,KAAK,UAAU,GAAG,KAAK,QAAQ,MAAM,GAExC,gCAAgC,KAAK,OAAO,iBACjD,KAAK,WAAW,IAAI,MAAM,MAAM,IAAI,KAAK,KAAK,IAAI,KAAK,OAAO,QAAQ,YAAY,0BAC3D,OAAO,KAAK,MAAM,EAAE,UAAU,YAAY,CAAC,CAAC,KAAK,IAAI,KACjF,MACA,MACF;CAEF,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACpOA,SAAgB,2BACd,SACA,UAA4B,CAAC,GACF;CAC3B,IAAI,QAAQ,cAAc,KAAA,GAAW,OAAO,KAAA;CAC5C,IAAI,CAAC,OAAO,UAAU,QAAQ,SAAS,KAAK,QAAQ,aAAa,GAC/D,MAAM,IAAI,WAAW,kEAAkE;CAEzF,IACE,QAAQ,SAAS,MACd,YACC,MAAM,QAAQ,QAAQ,OAAO,KAAK,QAAQ,QAAQ,MAAM,SAAS,KAAK,SAAS,WAAW,CAC9F,GAEA;CAGF,MAAM,WAAW,uBAAuB,QAAQ,eAAe;CAC/D,MAAM,kBAAkB,QAAQ,wBAAwB;CAGxD,MAAM,eAAe,IAAI,YAAY,CAAC,CAAC,OACrC,KAAK,UAAU,UAAU,SAAS,iBAAiB,QAAQ,QAAQ,CAAC,CACtE,CAAC,CAAC;CAEF,MAAM,UAAU,QAAQ,cAAc,CAAC,kBAAkB,IAAI;CAC7D,MAAM,QAAQ;EACZ,aAAa,eAAe,WAAW;EACvC,cAAc,QAAQ,YAAY,WAAW;CAC/C;CACA,OAAO,QAAQ,qBACX;EAAE,oBAAoB,QAAQ;EAAoB,GAAG;CAAM,IAC3D;EAAE,OAAO,QAAQ;EAAO,GAAG;CAAM;AACvC;;AAqEA,SAAgB,mBACd,QACA,oBACkB;CAClB,MAAM,eAAe,OAAO,MAAM,sBAAsB;CACxD,MAAM,cAAc,KAAK,IAAI,GAAG,OAAO,MAAM,eAAe,YAAY;CACxE,MAAM,kBAAkB,qBAAqB,OAAO,GAAG;CACvD,OAAO;EACL,OAAO,OAAO;EACd;EACA,cAAc,OAAO,MAAM;EAC3B,iBAAiB,OAAO,MAAM;EAC9B,cAAc,eAAe,IAAI,eAAe,KAAA;EAChD,GAAI,oBAAoB,KAAA,IACpB,sBAAsB,OAAO,MAAM,aAAa,QAC9C,EAAE,mBAAmB,IACrB,OAAO,YAAY,OACjB,CAAC,IACD,EAAE,kBAAkB,OAAO,QAAQ,IACvC,EAAE,eAAe,gBAAgB;EACrC,cAAc,OAAO,MAAM,aAAa;CAC1C;AACF;AAEA,SAAS,qBAAqB,KAAkD;CAC9E,MAAM,QAAQ,IAAI,kBAAkB,IAAI;CACxC,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,KAAK,SAAS,IAAI,QAAQ,KAAA;AACrF;;AAGA,SAAgB,wBACd,OACA,oBAC8B;CAC9B,OAAO,iBAAiB,mBACpB,mBAAmB,MAAM,QAAQ,kBAAkB,IACnD,KAAA;AACN;AAEA,IAAa,eAAb,cAAkC,eAAe;CAG7B;CACA;CACA;CAJlB,YACE,SACA,QACA,MACA,OACA;EACA,MAAM,SAAS,OAAO;EAJN,KAAA,SAAA;EACA,KAAA,OAAA;EACA,KAAA,QAAA;CAGlB;AACF;;;;AAKA,IAAa,mBAAb,cAAsC,eAAe;CAGjC;CAFlB,YACE,SACA,QACA,SACA;EACA,MAAM,SAAS,SAAS,OAAO;EAHf,KAAA,SAAA;CAIlB;AACF;AAiFA,MAAM,mBAAmB;AASzB,MAAM,qBAAqB,OAAO,QAAQ,IAAI,qBAAqB,KAAK;AACxE,MAAM,2BACJ,QAAQ,IAAI,gCAAgC,KAAA,IACxC,IACA,OAAO,QAAQ,IAAI,2BAA2B;AAEpD,SAAS,uBAAuB,YAAwC;CACtE,MAAM,WAAW,cAAc;CAC/B,IAAI,CAAC,OAAO,UAAU,QAAQ,KAAK,YAAY,GAC7C,MAAM,IAAI,WAAW,iDAAiD;CAExE,OAAO;AACT;AAEA,SAAS,mBAAmB,OAAoC;CAC9D,OAAO,OAAO,UAAU,YAAY,OAAO,cAAc,KAAK,KAAK,SAAS,IAAI,QAAQ,KAAA;AAC1F;AAEA,MAAM,mCAAmB,IAAI,IAAI;CAAC;CAAK;CAAK;CAAK;AAAG,CAAC;;;;;;;;;;AAWrD,MAAM,2BAA8C;CAClD;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;;;;;;;;;;;;AAaA,SAAgB,oBAAoB,KAAuB;CACzD,OAAO,kBAAkB,KAAK,CAAC;AACjC;AAEA,SAAS,kBAAkB,KAAc,OAAwB;CAC/D,IAAI,eAAe,cAAc,OAAO,iBAAiB,IAAI,IAAI,MAAM;CACvE,IAAI,EAAE,eAAe,QAAQ,OAAO;CAGpC,MAAM,SAAU,IAA6B;CAC7C,IAAI,OAAO,WAAW,YAAY,iBAAiB,IAAI,MAAM,GAAG,OAAO;CACvE,MAAM,OAAQ,IAA2B;CACzC,MAAM,WAAW,GAAG,IAAI,KAAK,IAAI,IAAI,QAAQ,IAAI,OAAO,SAAS,WAAW,OAAO;CACnF,IAAI,yBAAyB,MAAM,MAAM,EAAE,KAAK,QAAQ,CAAC,GAAG,OAAO;CACnE,MAAM,QAAS,IAA4B;CAC3C,IAAI,QAAQ,KAAK,iBAAiB,SAAS,UAAU,KACnD,OAAO,kBAAkB,OAAO,QAAQ,CAAC;CAE3C,OAAO;AACT;AAEA,SAAS,gBAAgB,SAAiC;CACxD,MAAM,IAAI,QAAQ,IAAI,aAAa;CACnC,IAAI,CAAC,GAAG,OAAO;CACf,MAAM,WAAW,OAAO,CAAC;CACzB,IAAI,OAAO,SAAS,QAAQ,KAAK,WAAW,GAAG,OAAO,WAAW;CACjE,MAAM,SAAS,KAAK,MAAM,CAAC;CAC3B,IAAI,OAAO,SAAS,MAAM,GAAG,OAAO,KAAK,IAAI,GAAG,SAAS,KAAK,IAAI,CAAC;CACnE,OAAO;AACT;;AAGA,SAAgB,UAAU,SAAyB;CACjD,OAAO,KAAK,IAAI,MAAM,KAAK,SAAS,IAAM;AAC5C;AAEA,SAAS,aAAa,MAAgD;CACpE,MAAM,UAAkC;EACtC,gBAAgB;EAChB,QAAQ;CACV;CACA,IAAI,KAAK,YACP,QAAQ,KAAK,WAAW,QAAQ,KAAK,WAAW;MAC3C,IAAI,KAAK,UAAU,KAAK,QAC7B,QAAQ,gBAAgB,UAAU,KAAK,UAAU,KAAK;CAExD,IAAI,KAAK,gBAAgB,QAAQ,qBAAqB,KAAK;CAC3D,OAAO;AACT;AAEA,SAAS,kBAAkB,QAAgB,MAAuB;CAChE,IAAI,WAAW,KAAK,OAAO;CAC3B,MAAM,QAAQ,KAAK,YAAY;CAC/B,OACE,MAAM,SAAS,iBAAiB,KAChC,MAAM,SAAS,aAAa,KAC5B,MAAM,SAAS,gBAAgB,KAC/B,MAAM,SAAS,eAAe;AAElC;AAEA,SAAS,0BAA0B,QAAgB,MAAuB;CACxE,IAAI,WAAW,OAAO,CAAC,eAAe,KAAK,IAAI,GAAG,OAAO;CACzD,OACE,iGAAiG,KAC/F,IACF,KAAK,oEAAoE,KAAK,IAAI;AAEtF;AAEA,SAAS,UACP,KACA,iBACA,iBACyB;CACzB,MAAM,OAAgC;EACpC,OAAO,IAAI;EACX,UAAU,IAAI;EACd,aAAa,IAAI,eAAe;CAClC;CACA,IAAI,IAAI,aAAa,MACnB,IAAI,wBAAwB,IAAI,KAAK,GAAG,KAAK,wBAAwB,IAAI;MACpE,KAAK,aAAa,IAAI;CAE7B,MAAM,WAAW,IAAI,YAAY;CACjC,IAAI,aAAa,KAAA,GACf,KAAK,WAAW,EAAE,MAAM,SAAS;CAGnC,IAAI,IAAI,cAAc,CAAC,iBACrB,KAAK,kBAAkB;EACrB,MAAM;EACN,aAAa;GAAE,MAAM,IAAI,WAAW;GAAM,QAAQ,IAAI,WAAW;GAAQ,QAAQ;EAAK;CACxF;MACK,IAAI,IAAI,YAAY,IAAI,YAC7B,KAAK,kBAAkB,EAAE,MAAM,cAAc;CAG/C,OAAO;AACT;AAEA,SAAS,wBAAwB,OAAwB;CACvD,OAAO,oBAAoB,KAAK,KAAK;AACvC;AAEA,eAAe,MAAM,IAA2B;CAC9C,OAAO,IAAI,SAAS,YAAY,WAAW,SAAS,EAAE,CAAC;AACzD;;;;;;;;AASA,SAAS,YAAY,mBAAoC,QAAmC;CAC1F,IAAI,CAAC,QAAQ,OAAO,kBAAkB;CACtC,IAAI,OAAQ,YAAkC,QAAQ,YACpD,OAAO,YAAY,IAAI,CAAC,kBAAkB,QAAQ,MAAM,CAAC;CAE3D,IAAI,OAAO,SACT,kBAAkB,MAAM;MAExB,OAAO,iBAAiB,eAAe,kBAAkB,MAAM,GAAG,EAAE,MAAM,KAAK,CAAC;CAElF,OAAO,kBAAkB;AAC3B;;AAGA,SAAS,iBAAiB,OAAe,YAAyC;CAChF,OAAO,cAAc,QAAQ,KAAK,IAAI,IAAI,SAAS;AACrD;;;;;;AASA,SAAgB,gBAAgB,KAAqB;CACnD,MAAM,UAAU,IAAI,KAAK;CACzB,MAAM,IAAI,QAAQ,MAAM,yCAAyC;CACjE,OAAO,IAAI,EAAE,EAAE,CAAE,KAAK,IAAI;AAC5B;AAEA,SAAgB,mBAAmB,KAAqB;CACtD,MAAM,WAAW,gBAAgB,GAAG;CACpC,IAAI;EACF,KAAK,MAAM,QAAQ;EACnB,OAAO;CACT,QAAQ;EAIN,IAAI,SAAS,WAAW,GAAG,KAAK,SAAS,WAAW,GAAG,GAAG,OAAO;CACnE;CAGA,MAAM,SAAS,CAAC,GAAG,SAAS,SAAS,OAAO,CAAC,CAAC,CAC3C,KAAK,UAAU,MAAM,KAAK,CAAC,CAC3B,QAAQ,UAAU,SAAS,IAAI;CAClC,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,YAAY,oBAAoB,UAAU,KAAK;EACrD,IAAI,CAAC,WAAW;EAChB,IAAI;GACF,KAAK,MAAM,SAAS;GACpB,OAAO;EACT,QAAQ,CAER;CACF;CAEA,OAAO;AACT;AAEA,SAAS,oBAAoB,OAAe,OAA8B;CACxE,MAAM,SAAS,MAAM;CACrB,MAAM,SAAS,WAAW,MAAM,MAAM,WAAW,MAAM,MAAM;CAC7D,IAAI,CAAC,QAAQ,OAAO;CAEpB,MAAM,QAAkB,CAAC,MAAM;CAC/B,IAAI,aAAa;CACjB,IAAI,YAAY;CAEhB,KAAK,IAAI,IAAI,QAAQ,GAAG,IAAI,MAAM,QAAQ,KAAK;EAC7C,MAAM,OAAO,MAAM;EACnB,IAAI,WAAW;GACb,YAAY;GACZ;EACF;EACA,IAAI,SAAS,MAAM;GACjB,YAAY;GACZ;EACF;EACA,IAAI,SAAS,MAAK;GAChB,aAAa,CAAC;GACd;EACF;EACA,IAAI,YAAY;EAEhB,IAAI,SAAS,KAAK,MAAM,KAAK,GAAG;OAC3B,IAAI,SAAS,KAAK,MAAM,KAAK,GAAG;OAChC,IAAI,SAAS,MAAM,MAAM,SAAS,IAAI;GACzC,MAAM,IAAI;GACV,IAAI,MAAM,WAAW,GAAG,OAAO,MAAM,MAAM,OAAO,IAAI,CAAC;EACzD;CACF;CAEA,OAAO;AACT;;;;;;AAOA,eAAsB,QACpB,KACA,OAAyB,CAAC,GACF;CACxB,MAAM,WAAW,KAAK,WAAW,iBAAA,CAAkB,QAAQ,QAAQ,EAAE;CACrE,MAAM,MAAM,GAAG,QAAQ;CACvB,MAAM,WAAW;CACjB,MAAM,YAAY,IAAI,aAAa,KAAK,oBAAoB;CAC5D,MAAM,kBAAkB,uBAAuB,KAAK,eAAe;CACnE,MAAM,UAAU,KAAK,SAAS,WAAW;CACzC,MAAM,UAAU,aAAa,IAAI;CACjC,MAAM,WAAW,KAAK,YAAY,oBAAoB,OAAO;CAC7D,MAAM,OAAO,KAAK;CAClB,MAAM,WAAW,KAAK,YAAY;CAClC,MAAM,eAAe,KAAK;CAC1B,MAAM,eAAe,KAAK;CAC1B,MAAM,aAAa,KAAK;CACxB,MAAM,gBAAgB,KAAK,IAAI;CAC/B,IAAI,KAAK,oBACP,oBAAoB,KAAK,oBAAoB;EAAE,aAAa;EAAG,cAAc;CAAE,CAAC;CAGlF,IAAI;CACJ,IAAI,mBAAmB;CACvB,KAAK,IAAI,UAAU,GAAG,UAAU,iBAAiB,WAAW;EAG1D,IAAI,cAAc,SAChB,MAAM,IAAI,aAAa,oCAAoC,YAAY;EAIzE,IAAI,UAAU,KAAK,iBAAiB,eAAe,UAAU,GAC3D,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;EAEtE,MAAM,aAAa,IAAI,gBAAgB;EACvC,MAAM,gBAAgB,YAAY,YAAY,YAAY;EAC1D,MAAM,gBAAgB,iBAAiB,WAAW,MAAM,GAAG,SAAS;EACpE,MAAM,UAAU,KAAK,IAAI;EACzB,MAAM,cAAc,UAClB,kBACA,KAAK,wBAAwB,eAC7B,KAAK,QACP;EACA,IAAI,uBAAuB;EAC3B,IAAI,MACF,MAAM,UAAU,MAAM,UAAU;GAC9B,SAAS,cAAc;GACvB,OAAO,cAAc;GACrB,QAAQ,cAAc;GACtB;GACA,OAAO,IAAI;GACX;GACA;GACA,cAAc;GACd,WAAW;GACX,WAAW;GACX,gBAAgB;GAChB;GACA,gBAAgB,CAAC;EACnB,CAAC;EAGH,IAAI;GACF,MAAM,MAAM,MAAM,QAAQ,KAAK;IAC7B,QAAQ;IACR;IACA,MAAM,KAAK,UAAU,WAAW;IAChC,QAAQ;GACV,CAAC;GACD,aAAa,aAAa;GAC1B,MAAM,kBAAkB,OAAO,gBAAgB,IAAI,OAAO,IAAI,KAAA;GAE9D,IAAI,CAAC,IAAI,IAAI;IACX,MAAM,OAAO,MAAM,IAAI,KAAK;IAC5B,IAAI,MAAM;KACR,MAAM,UAAU,MAAM,UAAU;MAC9B,SAAS,cAAc;MACvB,OAAO,cAAc;MACrB,QAAQ,cAAc;MACtB;MACA,OAAO,IAAI;MACX;MACA;MACA,cAAc;MACd,WAAW;MACX,WAAW,KAAK,IAAI;MACpB,YAAY,KAAK,IAAI,IAAI;MACzB,YAAY,IAAI;MAChB;MACA,cAAc;MACd,cAAc,QAAQ,IAAI;MAC1B,gBAAgB,CAAC;KACnB,CAAC;KACD,uBAAuB;IACzB;IACA,MAAM,MAAM,IAAI,aACd,6BAA6B,IAAI,UACjC,IAAI,QACJ,MACA,IAAI,KACN;IACA,IACE,0BAA0B,IAAI,QAAQ,IAAI,KAC1C,iBAAiB,gBAAgB,KACjC,UAAU,kBAAkB,KAC5B,CAAC,iBAAiB,eAAe,UAAU,GAC3C;KACA,UAAU;KACV,mBAAmB;MAAE,GAAG;MAAkB,aAAa;KAAE;KACzD;IACF;IACA,IACE,iBAAiB,IAAI,IAAI,MAAM,KAC/B,UAAU,kBAAkB,KAC5B,CAAC,iBAAiB,eAAe,UAAU,GAC3C;KACA,UAAU;KAEV,MAAM,MADa,gBAAgB,IAAI,OAClB,KAAK,UAAU,OAAO,CAAC;KAC5C;IACF;IACA,MAAM;GACR;GAEA,MAAM,OAAO,MAAM,IAAI,KAAK;GAC5B,IAAI;GACJ,IAAI;IACF,OAAO,KAAK,MAAM,IAAI;GACxB,SAAS,UAAU;IACjB,IAAI,MAAM;KACR,MAAM,UAAU,MAAM,UAAU;MAC9B,SAAS,cAAc;MACvB,OAAO,cAAc;MACrB,QAAQ,cAAc;MACtB;MACA,OAAO,IAAI;MACX;MACA;MACA,cAAc;MACd,WAAW;MACX,WAAW,KAAK,IAAI;MACpB,YAAY,KAAK,IAAI,IAAI;MACzB,YAAY,IAAI;MAChB;MACA,cAAc;MACd,cAAc,sBAAsB,oBAAoB,QAAQ,SAAS,UAAU,OAAO,QAAQ;MAClG,gBAAgB,CAAC;KACnB,CAAC;KACD,uBAAuB;IACzB;IACA,MAAM;GACR;GACA,IAAI,MACF,MAAM,UAAU,MAAM,UAAU;IAC9B,SAAS,cAAc;IACvB,OAAO,cAAc;IACrB,QAAQ,cAAc;IACtB;IACA,OAAO,IAAI;IACX;IACA;IACA,cAAc;IACd,WAAW;IACX,WAAW,KAAK,IAAI;IACpB,YAAY,KAAK,IAAI,IAAI;IACzB,YAAY,IAAI;IAChB;IACA,cAAc;IACd,gBAAgB,CAAC;GACnB,CAAC;GAEH,MAAM,SACJ,KAAK,UAGH;GACJ,MAAM,WACJ,KAAK,SAAS,OAAO,KAAK,UAAU,YAAY,CAAC,MAAM,QAAQ,KAAK,KAAK,IACpE,KAAK,QACN,KAAA;GACN,MAAM,eAAe,mBAAmB,UAAU,aAAa;GAC/D,MAAM,mBAAmB,mBAAmB,UAAU,iBAAiB;GACvE,MAAM,cAAc,mBAAmB,UAAU,YAAY;GAO7D,MAAM,gBALJ,UAAU,6BACV,OAAO,SAAS,8BAA8B,YAC9C,CAAC,MAAM,QAAQ,SAAS,yBAAyB,IAC5C,SAAS,4BACV,KAAA,EAAA,EACkC;GACxC,MAAM,kBACJ,iBAAiB,KAAA,IAAY,KAAA,IAAY,mBAAmB,YAAY;GAC1E,MAAM,YACJ,UAAU,yBACV,OAAO,SAAS,0BAA0B,YAC1C,CAAC,MAAM,QAAQ,SAAS,qBAAqB,IACxC,SAAS,sBAAkD,gBAC5D,KAAA;GACN,MAAM,qBAAqB,cAAc,KAAA,IAAY,KAAA,IAAY,mBAAmB,SAAS;GAC7F,MAAM,gBACJ,iBAAiB,KAAA,KACjB,qBAAqB,KAAA,MACpB,iBAAiB,KAAA,KACf,oBAAoB,KAAA,KAAa,mBAAmB,sBACtD,cAAc,KAAA,KACZ,uBAAuB,KAAA,KAAa,sBAAsB,kBAC5D,gBAAgB,KAAA,KAAa,gBAAgB,eAAe;GAC/D,MAAM,gBAAiB,KAAK,kBAAkB,KAAK;GACnD,MAAM,UAAU,QAAQ,SAAS,WAAW;GAE5C,MAAM,iBACJ,OAAO,kBAAkB,YAAY,iBAAiB,KAAK,qBACvD,oBAAoB,KAAK,oBAAoB;IAC3C,aAAa,gBAAiB,sBAAsB;IACpD,GAAI,qBAAqB,EAAE,cAAc,mBAAmB,IAAI,CAAC;IACjE,cAAc;GAChB,CAAC,IACD,KAAA;GAIN,MAAM,cACJ,OAAO,KAAK,UAAU,YAAY,KAAK,MAAM,KAAK,MAAM,KAAK,KAAK,QAAQ;GAC5E,IAAI,KAAK,mBACP,kBACE,IAAI,OACJ,aACA,KAAK,sBAAsB,OAAO,CAAC,IAAI,KAAK,iBAC9C;GAGF,OAAO;IACL;IACA,cAAc,QAAQ,iBAAiB;IACvC,cAAc,QAAQ,KAAK,CAAC,CAAC,WAAW;IACxC,OAAO;KACL,cAAc,gBAAgB;KAC9B,kBAAkB,oBAAoB;KACtC,aAAa,gBAAgB,gBAAgB,MAAM,oBAAoB;KACvE,UAAU;KACV;KACA;IACF;IACA,SAAS,OAAO,kBAAkB,WAAW,gBAAiB,kBAAkB;IAChF,OAAO,eAAe,IAAI;IAC1B;IACA,YAAY,KAAK,IAAI,IAAI;IACzB,KAAK;GACP;EACF,SAAS,KAAK;GACZ,aAAa,aAAa;GAC1B,UAAU;GAIV,IAAI,cAAc,SAAS;IACzB,IAAI,QAAQ,CAAC,sBACX,MAAM,UAAU,MAAM,UAAU;KAC9B,SAAS,cAAc;KACvB,OAAO,cAAc;KACrB,QAAQ,cAAc;KACtB;KACA,OAAO,IAAI;KACX;KACA;KACA,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;KAC7D,gBAAgB,CAAC;IACnB,CAAC;IAEH,MAAM;GACR;GACA,IAAI,QAAQ,CAAC,sBAIX,MAAM,UAAU,MAAM,UAAU;IAC9B,SAAS,cAAc;IACvB,OAAO,cAAc;IACrB,QAAQ,cAAc;IACtB;IACA,OAAO,IAAI;IACX;IACA;IACA,cAAc;IACd,WAAW;IACX,WAAW,KAAK,IAAI;IACpB,YAAY,KAAK,IAAI,IAAI;IACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAC7D,gBAAgB,CAAC;GACnB,CAAC;GAEH,IACE,UAAU,kBAAkB,KAC5B,oBAAoB,GAAG,KACvB,CAAC,iBAAiB,eAAe,UAAU,GAC3C;IACA,MAAM,MAAM,UAAU,OAAO,CAAC;IAC9B;GACF;GACA,MAAM;EACR;CACF;CACA,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;AACtE;AAEA,eAAe,UACb,MACA,UACA,OACe;CAGf,IAAI;EACF,MAAM,KAAK,OAAO,SAAS,KAAK,CAAC;CACnC,QAAQ,CAER;AACF;AAEA,SAAS,gBAAgB,GAAoC;CAC3D,MAAM,MAA8B,CAAC;CACrC,EAAE,SAAS,OAAO,QAAQ;EACxB,IAAI,OAAO;CACb,CAAC;CACD,OAAO;AACT;AAEA,SAAS,gBAAwB;CAC/B,IAAI,OAAO,WAAW,QAAQ,eAAe,YAAY,OAAO,WAAW,OAAO,WAAW;CAC7F,OAAO,GAAG,KAAK,IAAI,CAAC,CAAC,SAAS,EAAE,EAAE,GAAG,KAAK,OAAO,CAAC,CAAC,SAAS,EAAE,CAAC,CAAC,MAAM,GAAG,EAAE;AAC7E;;;;;;;AAQA,eAAsB,YACpB,KACA,OAAyB,CAAC,GACoB;CAC9C,MAAM,SAAS,MAAM,kBAAkB,KAAK,IAAI;CAEhD,OAAO;EAAE,OADK,gBAAmB,QAAQ,KAAK,mBAAmB,SACpD;EAAG;CAAO;AACzB;;AAGA,eAAe,kBACb,KACA,OAAyB,CAAC,GACF;CACxB,IAAI;EACF,OAAO,MAAM,QAAQ;GAAE,GAAG;GAAK,UAAU,IAAI,YAAY,CAAC,IAAI;EAAW,GAAG,IAAI;CAClF,SAAS,KAAK;EACZ,IACE,KAAK,wBAAwB,iBAC7B,eAAe,gBACf,kBAAkB,IAAI,QAAQ,IAAI,IAAI,KACtC,IAAI,YAGJ,OAAO,MAAM,QAAQ;GADiB,GAAG;GAAK,UAAU;GAAM,YAAY,KAAA;EAC3C,GAAG,IAAI;EAExC,MAAM;CACR;AACF;AAEA,SAAS,gBACP,QACA,iBACG;CACH,IAAI;EACF,IAAI,OAAO,iBAAiB,UAC1B,MAAM,IAAI,MACR,8CAA8C,OAAO,MAAM,uBAC7D;EAEF,OAAO,gBAAmB,OAAO,SAAS,OAAO,OAAO,eAAe;CACzE,SAAS,OAAO;EACd,IAAI,iBAAiB,kBAAkB,MAAM;EAC7C,MAAM,QAAQ,iBAAiB,QAAQ,QAAQ,IAAI,MAAM,OAAO,KAAK,CAAC;EACtE,MAAM,IAAI,iBAAiB,MAAM,SAAS,QAAQ,EAAE,MAAM,CAAC;CAC7D;AACF;AAEA,SAAS,gBACP,SACA,OACA,iBACG;CACH,MAAM,UAAU,oBAAoB,UAAU,UAAU,mBAAmB,OAAO;CAClF,IAAI;EACF,OAAO,KAAK,MAAM,OAAO;CAC3B,QAAQ;EACN,MAAM,IAAI,MAAM,wCAAwC,MAAM,EAAE;CAClE;AACF;AAWA,IAAa,yBAAb,cAA4C,sBAAsB;CAG9C;CACA;CAHlB,YACE,SACA,QACA,SACA;EACA,MAAM,OAAO;EAHG,KAAA,SAAA;EACA,KAAA,UAAA;CAGlB;AACF;;;;;;;;;;AAmCA,SAAgB,eAAe,MAAwB,MAA4B,CAAC,GAAS;CAC3F,MAAM,kBAAkB,KAAK,YAAY,KAAA;CACzC,MAAM,WAAW,KAAK,WAAW,iBAAA,CAAkB,QAAQ,QAAQ,EAAE;CAErE,IAAI,IAAI,0BAA0B,CAAC,iBACjC,MAAM,IAAI,uBACR,gGAAgG,iBAAiB,IACjH,wBACA,OACF;CAGF,IAAI,IAAI,iBAAiB,MAAM,MAAM,SAAS,SAAS,CAAC,CAAC,GACvD,MAAM,IAAI,uBACR,2BAA2B,QAAQ,8BACnC,oBACA,OACF;CAGF,IAAI,IAAI,mBAAmB,IAAI,gBAAgB,SAAS,GAElD;MAAA,CADO,IAAI,gBAAgB,MAAM,MAAM,SAAS,SAAS,CAAC,CACxD,GACJ,MAAM,IAAI,uBACR,2BAA2B,QAAQ,+BAA+B,IAAI,gBAAgB,IAAI,eAAe,CAAC,CAAC,KAAK,IAAI,EAAE,KACtH,wBACA,OACF;CAAA;CAIJ,IAAI,IAAI,eAAe,CAAC,KAAK,UAAU,CAAC,KAAK,UAAU,CAAC,KAAK,YAC3D,MAAM,IAAI,uBACR,sFACA,WACA,OACF;CAGF,IAAI,IAAI,kBAAkB;EACxB,MAAM,SAAS,KAAK,YAAY,oBAAoB,OAAO;EAC3D,IAAI,WAAW,IAAI,kBACjB,MAAM,IAAI,uBACR,qCAAqC,IAAI,iBAAiB,eAAe,QAAQ,eAAe,OAAO,IACvG,kBACA,OACF;CAEJ;AACF;AAEA,SAAS,SAAS,KAAa,SAAmC;CAChE,IAAI,mBAAmB,QAAQ,OAAO,QAAQ,KAAK,GAAG;CACtD,OAAO,IAAI,YAAY,CAAC,CAAC,WAAW,QAAQ,YAAY,CAAC;AAC3D;AAEA,SAAS,gBAAgB,GAA4B;CACnD,OAAO,aAAa,SAAS,EAAE,SAAS;AAC1C;;;;;;;;;;;;;;;;;;AAmBA,eAAsB,SACpB,OACA,OAAkD,CAAC,GASlD;CACD,MAAM,QAAQ,KAAK,IAAI;CACvB,IAAI;EACF,MAAM,SAAS,MAAM,QACnB;GACE;GACA,UAAU,CAAC;IAAE,MAAM;IAAQ,SAAS;GAAO,CAAC;GAC5C,WAAA;GACA,WAAW,KAAK,aAAa;EAC/B,GACA,IACF;EACA,OAAO;GACL,IAAI;GACJ,WAAW,KAAK,IAAI,IAAI;GACxB,OAAO;GACP,aAAa,OAAO,eAAe;GACnC,aAAa,iBAAiB,OAAO,OAAO,WAAW,CAAC,CAAC;EAC3D;CACF,SAAS,KAAK;EACZ,OAAO;GACL,IAAI;GACJ,WAAW,KAAK,IAAI,IAAI;GACxB,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;GACtD,aAAa;GACb,aAAa;EACf;CACF;AACF;;;;;;AAOA,IAAa,YAAb,MAAuB;CACrB;CACA;CAEA,YAAY,OAAyB,CAAC,GAAG;EACvC,KAAK,OAAO;EACZ,KAAK,kBAAkB,uBAAuB,KAAK,eAAe;CACpE;CAEA,KAAK,KAAqB,KAAgD;EACxE,MAAM,UAAU;GAAE,GAAG,KAAK;GAAM,GAAG;EAAI;EACvC,OAAO,IAAI,aAAa,kBAAkB,KAAK,OAAO,IAAI,QAAQ,KAAK,OAAO;CAChF;CAEA,SACE,KACA,KAC8C;EAC9C,OAAO,YAAe,KAAK;GAAE,GAAG,KAAK;GAAM,GAAG;EAAI,CAAC;CACrD;AACF"}
@@ -1,3 +1,3 @@
1
- import { t as DefaultVerdict } from "../verdict-Dps8_okt.js";
2
- import { a as MatrixCell, i as MatrixAxis, n as AxisSummary, o as MatrixResult, r as CellResult, s as RunAgentMatrixOptions, t as runAgentMatrix } from "../index-DSC51roc.js";
1
+ import { t as DefaultVerdict } from "../verdict-DExhxfgR.js";
2
+ import { a as MatrixCell, i as MatrixAxis, n as AxisSummary, o as MatrixResult, r as CellResult, s as RunAgentMatrixOptions, t as runAgentMatrix } from "../index-C3ssXVLv.js";
3
3
  export { type AxisSummary, type CellResult, type DefaultVerdict, type MatrixAxis, type MatrixCell, type MatrixResult, type RunAgentMatrixOptions, runAgentMatrix };
@@ -1,6 +1,6 @@
1
1
  import { f as Run } from "../schema-BtVldJ3T.js";
2
2
  import { s as TraceStore } from "../store-CT9YIIve.js";
3
- import { Ct as GoldenItem, St as ContinuousCalibrationResult, a as CorpusAgreementReport, bt as ContinuousAgreement, vt as CalibrationResult, yt as CandidateScore } from "../statistics-C-dm-J6H.js";
3
+ import { Ct as GoldenItem, St as ContinuousCalibrationResult, a as CorpusAgreementReport, bt as ContinuousAgreement, vt as CalibrationResult, yt as CandidateScore } from "../statistics-D6Uebe_4.js";
4
4
  import { a as OutcomeFilter, i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "../outcome-store-BYHIuO0e.js";
5
5
  import { n as SeriesConvergenceResult, t as SeriesConvergenceOptions } from "../series-convergence-ofsqPWhs.js";
6
6
  import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "../rubric-predictive-validity-9qAwzkZm.js";
@@ -1,7 +1,7 @@
1
1
  import { c as ValidationError } from "./errors-D-LKuDhb.js";
2
2
  import { i as ROLLOUT_SCHEMA, s as assertMinted } from "./schema-C6DW4ZHR.js";
3
3
  import { a as scoreOrigin, i as rolloutRewardFields } from "./reward-nw2xZGZG.js";
4
- import { o as runTaskScore } from "./run-record-CWN8-VsV.js";
4
+ import { o as runTaskScore } from "./run-record-DqOw5X6_.js";
5
5
  import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
6
6
  //#region src/rollout/mint.ts
7
7
  /**
@@ -315,4 +315,4 @@ async function mintRolloutRows(records, store, options = {}) {
315
315
  //#endregion
316
316
  export { unmintableReasons as n, mintRolloutRows as t };
317
317
 
318
- //# sourceMappingURL=mint-DD-0oQTA.js.map
318
+ //# sourceMappingURL=mint-CGEkzPLf.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"mint-DD-0oQTA.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
1
+ {"version":3,"file":"mint-CGEkzPLf.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
@@ -1,4 +1,4 @@
1
- import { t as DefaultVerdict } from "./verdict-Dps8_okt.js";
1
+ import { t as DefaultVerdict } from "./verdict-DExhxfgR.js";
2
2
  //#region src/multi-layer-verifier.d.ts
3
3
  type LayerStatus = 'pass' | 'fail' | 'skipped' | 'error' | 'timeout';
4
4
  type Severity = 'critical' | 'major' | 'minor' | 'info';
@@ -135,4 +135,4 @@ declare class MultiLayerVerifier<Env = unknown> {
135
135
  }
136
136
  //#endregion
137
137
  export { MultiLayerVerifier as a, VerifyContext as c, LayerStatus as i, VerifyOptions as l, Layer as n, Severity as o, LayerResult as r, VerificationReport as s, Finding as t, gradeSemanticStatus as u };
138
- //# sourceMappingURL=multi-layer-verifier-BHY1gWAc.d.ts.map
138
+ //# sourceMappingURL=multi-layer-verifier-DnAqwl0h.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"multi-layer-verifier-BHY1gWAc.d.ts","names":[],"sources":["../src/multi-layer-verifier.ts"],"mappings":";;KA4BY;KAEA;UAEK;EACf,UAAU;EACV;EACA;;EAEA;;;;;;EAMA,SAAS;;UAGM;EACf;EACA,QAAQ;;EAER;;EAEA;EACA;EACA,UAAU;;EAEV;;;;;;;;;EASA,cAAc;;EAEd,SAAS;;UAGM,cAAc;;EAE7B,KAAK;;EAEL,OAAO,eAAe;;EAEtB,QAAQ;;UAGO,MAAM;EACrB;;EAEA;;EAEA;;;;;EAKA;;;;;;EAMA;;EAEA;EACA,MAAM,KAAK,cAAc,SAAS,QAAQ,eAAe;;UAG1C,cAAc;EAC7B,KAAK;;;;;EAKL;;EAEA,WAAW,QAAQ;;;;UAKJ,2BAA2B;EAC1C,QAAQ;EACR;EACA;EACA;EACA;;EAEA;;;;;EAKA;;;;;;;;EAQA;EACA;EACA;EACA;;;;;;;;;;;;;iBAgBc,oBAAoB;EAClC;EACA,UAAU;IAAQ,UAAU;IAAU;IAAmB;;EACzD;EACA;IACE;;;;cAcS,mBAAmB;mBACD;EAA7B,YAA6B,QAAQ,MAAM;EAiBrC,IAAI,MAAM,cAAc,OAAO,QAAQ"}
1
+ {"version":3,"file":"multi-layer-verifier-DnAqwl0h.d.ts","names":[],"sources":["../src/multi-layer-verifier.ts"],"mappings":";;KA4BY;KAEA;UAEK;EACf,UAAU;EACV;EACA;;EAEA;;;;;;EAMA,SAAS;;UAGM;EACf;EACA,QAAQ;;EAER;;EAEA;EACA;EACA,UAAU;;EAEV;;;;;;;;;EASA,cAAc;;EAEd,SAAS;;UAGM,cAAc;;EAE7B,KAAK;;EAEL,OAAO,eAAe;;EAEtB,QAAQ;;UAGO,MAAM;EACrB;;EAEA;;EAEA;;;;;EAKA;;;;;;EAMA;;EAEA;EACA,MAAM,KAAK,cAAc,SAAS,QAAQ,eAAe;;UAG1C,cAAc;EAC7B,KAAK;;;;;EAKL;;EAEA,WAAW,QAAQ;;;;UAKJ,2BAA2B;EAC1C,QAAQ;EACR;EACA;EACA;EACA;;EAEA;;;;;EAKA;;;;;;;;EAQA;EACA;EACA;EACA;;;;;;;;;;;;;iBAgBc,oBAAoB;EAClC;EACA,UAAU;IAAQ,UAAU;IAAU;IAAmB;;EACzD;EACA;IACE;;;;cAcS,mBAAmB;mBACD;EAA7B,YAA6B,QAAQ,MAAM;EAiBrC,IAAI,MAAM,cAAc,OAAO,QAAQ"}
@@ -1,6 +1,6 @@
1
1
  import { p as CostProvenance } from "../cost-ledger-Bv_e8XHY.js";
2
- import { w as JudgeScore } from "../types-DOZyvsFU.js";
3
- import { o as MatrixResult } from "../index-DSC51roc.js";
2
+ import { w as JudgeScore } from "../types-DYuNHo9R.js";
3
+ import { o as MatrixResult } from "../index-C3ssXVLv.js";
4
4
  import { AgentProfile } from "@tangle-network/agent-interface";
5
5
  //#region src/multishot/types.d.ts
6
6
  interface MultishotMessage {
package/dist/openapi.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "openapi": "3.1.0",
3
3
  "info": {
4
4
  "title": "@tangle-network/agent-eval — wire protocol",
5
- "version": "0.144.6",
5
+ "version": "0.144.7",
6
6
  "description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
7
7
  "contact": {
8
8
  "name": "Tangle Network",
@@ -0,0 +1,114 @@
1
+ import { y as PairedBootstrapResult } from "./statistics-D6Uebe_4.js";
2
+ //#region src/paired-promotion-decision.d.ts
3
+ /** Which paired estimator produced the deciding interval. */
4
+ type PairedDecisionStatistic = 'paired_risk_difference' | 'mean_bootstrap' | 'median_bootstrap';
5
+ /** Which test carried the decision, given the estimator and the sample size. */
6
+ type PairedDecisionMethod = 'score-interval' | 'bootstrap-ci' | 'exact-sign';
7
+ /** McNemar's exact paired-binary evidence, on the two-point path only. */
8
+ interface PairedMcNemarEvidence {
9
+ /** Discordant pairs the treatment won. */
10
+ b: number;
11
+ /** Discordant pairs the control won. */
12
+ c: number;
13
+ /** b + c — the only pairs carrying information. */
14
+ nDiscordant: number;
15
+ /** Two-sided exact p-value. */
16
+ pValue: number;
17
+ }
18
+ interface PairedPromotionDecisionOptions {
19
+ /** Smallest candidate-minus-baseline delta that counts as improvement, in the
20
+ * caller's native units. May be negative (a noninferiority margin). Default 0. */
21
+ threshold?: number;
22
+ /** Confidence level. Default 0.95. */
23
+ confidence?: number;
24
+ /** Bootstrap resamples, on the paths where a bootstrap decides. Default 2000. */
25
+ resamples?: number;
26
+ /** Deterministic bootstrap seed. Omitted ⇒ derived from the deltas. */
27
+ seed?: number;
28
+ /** Caller-required paired observations. The exact test may impose a higher
29
+ * minimum; the effective one is reported as `minimumPairs`. */
30
+ minPairs?: number;
31
+ /**
32
+ * `'mean'` (default) routes by SHAPE: a two-point (pass/fail) outcome on any
33
+ * encoding decides on the score interval, everything else on the mean
34
+ * bootstrap. `'median'` forces the median bootstrap on every input, including
35
+ * shapes where it is structurally blind — kept for callers who want outlier
36
+ * robustness on genuinely continuous outcomes and accept that cost.
37
+ */
38
+ statistic?: 'mean' | 'median';
39
+ }
40
+ interface PairedPromotionDecision {
41
+ /** Paired observations supplied. */
42
+ n: number;
43
+ /** Threshold the interval was judged against, native units. */
44
+ threshold: number;
45
+ confidence: number;
46
+ statistic: PairedDecisionStatistic;
47
+ method: PairedDecisionMethod;
48
+ /** Common positive level of a two-point outcome ({0,1} ⇒ 1, {0,100} ⇒ 100),
49
+ * or null when the outcome is not two-point. Non-null is exactly the
50
+ * condition for the `paired_risk_difference` path, and it is the factor
51
+ * `delta` / `low` / `high` were rescaled by. */
52
+ binaryScale: number | null;
53
+ /** Exact-tie fraction over the paired deltas; null when there are no pairs. */
54
+ tieFraction: number | null;
55
+ /** Point estimate of the DECIDING statistic, in the caller's native units. */
56
+ delta: number;
57
+ /** Lower bound of the DECIDING interval, native units. */
58
+ low: number;
59
+ /** Upper bound of the DECIDING interval, native units. */
60
+ high: number;
61
+ /** The bootstrap that decided, or null when the score interval did. Callers
62
+ * that need a bootstrap as a diagnostic on the two-point path compute their
63
+ * own — it is not computed here, so the binary path costs no resamples. */
64
+ bootstrap: PairedBootstrapResult | null;
65
+ /** McNemar's exact evidence, or null off the two-point path. */
66
+ mcnemar: PairedMcNemarEvidence | null;
67
+ /** Exact one-sided sign-test p-value on the small-sample bootstrap path;
68
+ * null otherwise. */
69
+ pValue: number | null;
70
+ /** Effective observation minimum after accounting for confidence. */
71
+ minimumPairs: number;
72
+ /** n >= minimumPairs. */
73
+ sufficient: boolean;
74
+ /** The deciding interval is zero-width or non-finite — no evidence in either
75
+ * direction, so it cannot clear any threshold on evidence. */
76
+ indeterminate: boolean;
77
+ /** McNemar's exact test refuses at a non-negative threshold. */
78
+ exactTestVetoes: boolean;
79
+ /** The deciding interval clears the threshold, ignoring the other two guards. */
80
+ clearsThreshold: boolean;
81
+ /** `sufficient && !indeterminate && clearsThreshold && !exactTestVetoes` —
82
+ * the whole rule. */
83
+ promote: boolean;
84
+ /** What `delta` measures, for a reason string. */
85
+ label: 'success-rate' | 'mean' | 'median';
86
+ /** Why a zero-width interval is zero-width; empty when it is not. */
87
+ indeterminateCause: string;
88
+ /** Sentence naming the test when the exact sign test decided; else empty. */
89
+ methodDetail: string;
90
+ }
91
+ /** The shape facts that pick the estimator, without computing an interval. */
92
+ interface PairedDecisionShape {
93
+ statistic: PairedDecisionStatistic;
94
+ /** Common positive level of a two-point outcome; null when not two-point. */
95
+ binaryScale: number | null;
96
+ /** Exact-tie fraction over the paired deltas; null when there are no pairs. */
97
+ tieFraction: number | null;
98
+ }
99
+ /**
100
+ * Which estimator {@link decidePairedPromotion} would use on this data, and the
101
+ * shape facts behind it — for callers that must report the shape on a path
102
+ * where no interval is computed at all (an early rejection, or zero pairs).
103
+ * Cheap: no bootstrap, no interval.
104
+ */
105
+ declare function pairedDecisionShape(before: number[], after: number[], statistic?: 'mean' | 'median'): PairedDecisionShape;
106
+ /**
107
+ * Decide whether a paired candidate-minus-baseline delta clears a promotion
108
+ * threshold. `before` is the baseline arm, `after` the candidate arm, paired by
109
+ * position. Throws on unequal lengths.
110
+ */
111
+ declare function decidePairedPromotion(before: number[], after: number[], options?: PairedPromotionDecisionOptions): PairedPromotionDecision;
112
+ //#endregion
113
+ export { PairedPromotionDecision as a, pairedDecisionShape as c, PairedMcNemarEvidence as i, PairedDecisionShape as n, PairedPromotionDecisionOptions as o, PairedDecisionStatistic as r, decidePairedPromotion as s, PairedDecisionMethod as t };
114
+ //# sourceMappingURL=paired-promotion-decision-B6zJ3gYM.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"paired-promotion-decision-B6zJ3gYM.d.ts","names":[],"sources":["../src/paired-promotion-decision.ts"],"mappings":";;;KA2DY;;KAMA;;UAGK;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;;;EAGf;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;;EAQA;;UAGe;;EAEf;;EAEA;EACA;EACA,WAAW;EACX,QAAQ;;;;;EAKR;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,WAAW;;EAEX,SAAS;;;EAGT;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;;UAIe;EACf,WAAW;;EAEX;;EAEA;;;;;;;;iBASc,oBACd,kBACA,iBACA,gCACC;;;;;;iBAiBa,sBACd,kBACA,iBACA,UAAS,iCACR"}
@@ -2,7 +2,8 @@ import { S as ToolSpan, f as Run, r as BudgetSpec } from "../schema-BtVldJ3T.js"
2
2
  import { a as RunFilter, s as TraceStore } from "../store-CT9YIIve.js";
3
3
  import { o as llmSpans } from "../query-CJ_DX8vl.js";
4
4
  import { n as FailureClusterReport, r as failureClusterView, t as FailureCluster } from "../failure-cluster-CqcvCcdR.js";
5
- import { d as TrajectoryStep, g as computeToolUseMetrics, n as BaselineReport, t as BaselineOptions } from "../baseline-D_fT6277.js";
5
+ import { n as BaselineReport, p as computeToolUseMetrics, t as BaselineOptions } from "../baseline-CavEbRyH.js";
6
+ import { n as TrajectoryStep } from "../trajectory-YC15QDYQ.js";
6
7
  //#region src/pipelines/budget-breach.d.ts
7
8
  interface BudgetBreachFinding {
8
9
  runId: string;
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","names":[],"sources":["../../src/pipelines/budget-breach.ts","../../src/pipelines/first-divergence.ts","../../src/pipelines/judge-agreement.ts","../../src/pipelines/regression.ts","../../src/pipelines/stuck-loop.ts","../../src/pipelines/tool-waste.ts"],"mappings":";;;;;;UAUiB;EACf;EACA;EACA;EACA,iBAAiB;EACjB;EACA;EACA;EACA;;UAGe;EACf,UAAU;EACV,aAAa;EACb,YAAY;EACZ,WAAW;EACX;EACA;;iBAGoB,iBACpB,OAAO,YACP;EAAW;EAAqB;IAC/B,QAAQ;;;UCrBM;EACf;EACA;EACA;EACA,QAAQ;EACR,QAAQ;EACR;;EAEA;;UAGe;;EAEf,cAAc,GAAG,gBAAgB,GAAG;;iBAGhB,oBACpB,OAAO,YACP,cACA,cACA,UAAS,oBACR,QAAQ;;;UCnBM;EACf;EACA;EACA;;EAEA;EACA;EACA;;UAGe;EACf,OAAO;EACP;EACA;;iBAGoB,mBAAmB,OAAO,aAAa,QAAQ;;;UChBpD;EACf;EACA;;EAEA,WAAW,KAAK,KAAK,OAAO,eAAe;;UAG5B,0BAA0B;EACzC,UAAU;EACV,WAAW;;iBAGS,eACpB,OAAO,YACP,SAAS,kBACT,SAAS,oBACR,QAAQ;;;UCDM;EACf;EACA;EACA;;EAEA;EACA;;EAEA;;EAEA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;;EAEf;;EAEA;;;;;EAKA;;EAEA;;iBAGoB,cACpB,OAAO,YACP,UAAS,mBACR,QAAQ;;;UChDM;EACf;EACA;EACA;EACA;;UAGe;EACf,OAAO;EACP;;UAGe;EACf;EACA,eAAe,MAAM,UAAU;IAAS,KAAK,QAAQ,kBAAkB;;;iBAGnD,cACpB,OAAO,YACP,UAAS,mBACR,QAAQ"}
1
+ {"version":3,"file":"index.d.ts","names":[],"sources":["../../src/pipelines/budget-breach.ts","../../src/pipelines/first-divergence.ts","../../src/pipelines/judge-agreement.ts","../../src/pipelines/regression.ts","../../src/pipelines/stuck-loop.ts","../../src/pipelines/tool-waste.ts"],"mappings":";;;;;;;UAUiB;EACf;EACA;EACA;EACA,iBAAiB;EACjB;EACA;EACA;EACA;;UAGe;EACf,UAAU;EACV,aAAa;EACb,YAAY;EACZ,WAAW;EACX;EACA;;iBAGoB,iBACpB,OAAO,YACP;EAAW;EAAqB;IAC/B,QAAQ;;;UCrBM;EACf;EACA;EACA;EACA,QAAQ;EACR,QAAQ;EACR;;EAEA;;UAGe;;EAEf,cAAc,GAAG,gBAAgB,GAAG;;iBAGhB,oBACpB,OAAO,YACP,cACA,cACA,UAAS,oBACR,QAAQ;;;UCnBM;EACf;EACA;EACA;;EAEA;EACA;EACA;;UAGe;EACf,OAAO;EACP;EACA;;iBAGoB,mBAAmB,OAAO,aAAa,QAAQ;;;UChBpD;EACf;EACA;;EAEA,WAAW,KAAK,KAAK,OAAO,eAAe;;UAG5B,0BAA0B;EACzC,UAAU;EACV,WAAW;;iBAGS,eACpB,OAAO,YACP,SAAS,kBACT,SAAS,oBACR,QAAQ;;;UCDM;EACf;EACA;EACA;;EAEA;EACA;;EAEA;;EAEA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;;EAEf;;EAEA;;;;;EAKA;;EAEA;;iBAGoB,cACpB,OAAO,YACP,UAAS,mBACR,QAAQ;;;UChDM;EACf;EACA;EACA;EACA;;UAGe;EACf,OAAO;EACP;;UAGe;EACf;EACA,eAAe,MAAM,UAAU;IAAS,KAAK,QAAQ,kBAAkB;;;iBAGnD,cACpB,OAAO,YACP,UAAS,mBACR,QAAQ"}