@tangle-network/agent-eval 0.172.1 → 0.173.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (219) hide show
  1. package/CHANGELOG.md +19 -0
  2. package/README.md +17 -2
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/analyst/index.d.ts +13 -13
  5. package/dist/analyst/index.js +7 -6
  6. package/dist/analyst/index.js.map +1 -1
  7. package/dist/{attestation-CJBGmMVh.d.ts → attestation-c1QvaBdX.d.ts} +2 -2
  8. package/dist/{attestation-CJBGmMVh.d.ts.map → attestation-c1QvaBdX.d.ts.map} +1 -1
  9. package/dist/{backend-integrity-e79K3UPD.d.ts → backend-integrity-CeuTgqsd.d.ts} +3 -4
  10. package/dist/backend-integrity-CeuTgqsd.d.ts.map +1 -0
  11. package/dist/{benchmark-h-h4bfqj.d.ts → benchmark-BjLGkfnN.d.ts} +3 -3
  12. package/dist/{benchmark-h-h4bfqj.d.ts.map → benchmark-BjLGkfnN.d.ts.map} +1 -1
  13. package/dist/{benchmark-command--qeZUHbu.js → benchmark-command-9S20PRel.js} +9 -9
  14. package/dist/{benchmark-command--qeZUHbu.js.map → benchmark-command-9S20PRel.js.map} +1 -1
  15. package/dist/benchmarks/index.d.ts +5 -5
  16. package/dist/benchmarks/index.js +3 -3
  17. package/dist/{bounded-process-VIi0KSL2.js → bounded-process-BBZob7vl.js} +30 -7
  18. package/dist/bounded-process-BBZob7vl.js.map +1 -0
  19. package/dist/builder-eval/index.d.ts +3 -3
  20. package/dist/builder-eval/index.js +2 -2
  21. package/dist/campaign/index.d.ts +9 -9
  22. package/dist/campaign/index.js +6 -6
  23. package/dist/{campaign-Dp35pBbS.js → campaign-pxS0wmo4.js} +8 -8
  24. package/dist/{campaign-Dp35pBbS.js.map → campaign-pxS0wmo4.js.map} +1 -1
  25. package/dist/chat-client-Db4bqYfA.js +115 -0
  26. package/dist/chat-client-Db4bqYfA.js.map +1 -0
  27. package/dist/{chat-json-call-5Jxna-aV.js → chat-json-call-C26igCih.js} +16 -6
  28. package/dist/chat-json-call-C26igCih.js.map +1 -0
  29. package/dist/cli.js +31 -17
  30. package/dist/cli.js.map +1 -1
  31. package/dist/{client-BvwNkIRN.js → client-BlLY6o2w.js} +2 -2
  32. package/dist/{client-BvwNkIRN.js.map → client-BlLY6o2w.js.map} +1 -1
  33. package/dist/{client-Df7wdslk.d.ts → client-DlqdbM7n.d.ts} +4 -4
  34. package/dist/{client-Df7wdslk.d.ts.map → client-DlqdbM7n.d.ts.map} +1 -1
  35. package/dist/contract/index.d.ts +13 -13
  36. package/dist/contract/index.js +202 -10
  37. package/dist/contract/index.js.map +1 -1
  38. package/dist/{counterfactual-Bee5_BIn.d.ts → counterfactual-CLgrwhkY.d.ts} +4 -4
  39. package/dist/{counterfactual-Bee5_BIn.d.ts.map → counterfactual-CLgrwhkY.d.ts.map} +1 -1
  40. package/dist/{chat-client-DI79OPye.js → default-registry-B0bKikCb.js} +3 -39
  41. package/dist/default-registry-B0bKikCb.js.map +1 -0
  42. package/dist/{default-registry-XxedTLwu.d.ts → default-registry-BKwc8bN5.d.ts} +6 -6
  43. package/dist/{default-registry-XxedTLwu.d.ts.map → default-registry-BKwc8bN5.d.ts.map} +1 -1
  44. package/dist/{define-agent-eval-0wW7gFhr.d.ts → define-agent-eval-CY6qdlGV.d.ts} +6 -6
  45. package/dist/{define-agent-eval-0wW7gFhr.d.ts.map → define-agent-eval-CY6qdlGV.d.ts.map} +1 -1
  46. package/dist/{define-agent-eval-jS8xj_Q_.js → define-agent-eval-D_i_s69h.js} +7 -7
  47. package/dist/{define-agent-eval-jS8xj_Q_.js.map → define-agent-eval-D_i_s69h.js.map} +1 -1
  48. package/dist/{dspy-rlm-engine-CS3qcCEk.js → dspy-rlm-engine-D5byiHn9.js} +5 -6
  49. package/dist/dspy-rlm-engine-D5byiHn9.js.map +1 -0
  50. package/dist/{emitter-Bvnu0VzL.d.ts → emitter-Cs0egaFd.d.ts} +3 -3
  51. package/dist/{emitter-Bvnu0VzL.d.ts.map → emitter-Cs0egaFd.d.ts.map} +1 -1
  52. package/dist/{engine-BfRay1qD.d.ts → engine-DhFir3Ys.d.ts} +23 -8
  53. package/dist/{engine-BfRay1qD.d.ts.map → engine-DhFir3Ys.d.ts.map} +1 -1
  54. package/dist/{eval-campaign-JDTeE6Pl.js → eval-campaign-BeAjdhzC.js} +2 -2
  55. package/dist/{eval-campaign-JDTeE6Pl.js.map → eval-campaign-BeAjdhzC.js.map} +1 -1
  56. package/dist/{exact-types-BEecmnWm.d.ts → exact-types-BKOEILRP.d.ts} +2 -2
  57. package/dist/{exact-types-BEecmnWm.d.ts.map → exact-types-BKOEILRP.d.ts.map} +1 -1
  58. package/dist/experiment/index.d.ts +4 -4
  59. package/dist/{external-optimizer-contracts-szBJ_1vh.d.ts → external-optimizer-contracts-CQCpyrIL.d.ts} +2 -2
  60. package/dist/{external-optimizer-contracts-szBJ_1vh.d.ts.map → external-optimizer-contracts-CQCpyrIL.d.ts.map} +1 -1
  61. package/dist/{external-optimizer-process-CQxylYeG.js → external-optimizer-process-Cq_Pg15r.js} +2 -2
  62. package/dist/{external-optimizer-process-CQxylYeG.js.map → external-optimizer-process-Cq_Pg15r.js.map} +1 -1
  63. package/dist/{external-optimizer-subprocess-Cex8Da2i.js → external-optimizer-subprocess-DgNebftP.js} +2 -2
  64. package/dist/{external-optimizer-subprocess-Cex8Da2i.js.map → external-optimizer-subprocess-DgNebftP.js.map} +1 -1
  65. package/dist/failure-cluster-OldNRoAt.d.ts +154 -0
  66. package/dist/failure-cluster-OldNRoAt.d.ts.map +1 -0
  67. package/dist/{feedback-trajectory-DIqpCyF0.d.ts → feedback-trajectory-CMnv_uYs.d.ts} +6 -6
  68. package/dist/{feedback-trajectory-DIqpCyF0.d.ts.map → feedback-trajectory-CMnv_uYs.d.ts.map} +1 -1
  69. package/dist/{heldout-gate-JgNRDZwZ.d.ts → heldout-gate-Df5hsqmm.d.ts} +7 -7
  70. package/dist/{heldout-gate-JgNRDZwZ.d.ts.map → heldout-gate-Df5hsqmm.d.ts.map} +1 -1
  71. package/dist/hosted/index.d.ts +2 -2
  72. package/dist/hosted/index.js +1 -1
  73. package/dist/{index-DnglhM0A.d.ts → index-BQqOjerE.d.ts} +94 -13
  74. package/dist/index-BQqOjerE.d.ts.map +1 -0
  75. package/dist/{index-_vPrVMRX.d.ts → index-CFDffsKz.d.ts} +11 -11
  76. package/dist/{index-_vPrVMRX.d.ts.map → index-CFDffsKz.d.ts.map} +1 -1
  77. package/dist/{index-DDAPhUJJ.d.ts → index-D0Db5X-4.d.ts} +45 -9
  78. package/dist/index-D0Db5X-4.d.ts.map +1 -0
  79. package/dist/{index-DMoxLG8P.d.ts → index-e7LXeRVa.d.ts} +3 -3
  80. package/dist/{index-DMoxLG8P.d.ts.map → index-e7LXeRVa.d.ts.map} +1 -1
  81. package/dist/index.d.ts +130 -42
  82. package/dist/index.d.ts.map +1 -1
  83. package/dist/index.js +64 -41
  84. package/dist/index.js.map +1 -1
  85. package/dist/{insight-report-08F022xN.d.ts → insight-report-DETqPc_A.d.ts} +4 -4
  86. package/dist/{insight-report-08F022xN.d.ts.map → insight-report-DETqPc_A.d.ts.map} +1 -1
  87. package/dist/{integrity-B_EDELom.d.ts → integrity-BKTcA-HP.d.ts} +3 -3
  88. package/dist/{integrity-B_EDELom.d.ts.map → integrity-BKTcA-HP.d.ts.map} +1 -1
  89. package/dist/{kind-factory-DMeEoMQZ.js → kind-factory-gP6lDySe.js} +164 -149
  90. package/dist/kind-factory-gP6lDySe.js.map +1 -0
  91. package/dist/{llm-client-BFMRpmqb.js → llm-client-CxQtdtd6.js} +12 -5
  92. package/dist/llm-client-CxQtdtd6.js.map +1 -0
  93. package/dist/{llm-judge-aQHIk5_-.js → llm-judge-B2YxbAJb.js} +81 -7
  94. package/dist/{llm-judge-aQHIk5_-.js.map → llm-judge-B2YxbAJb.js.map} +1 -1
  95. package/dist/{matrix-Ch8JO1pG.d.ts → matrix-DGu8KhSs.d.ts} +2 -2
  96. package/dist/{matrix-Ch8JO1pG.d.ts.map → matrix-DGu8KhSs.d.ts.map} +1 -1
  97. package/dist/meta-eval/index.d.ts +100 -5
  98. package/dist/meta-eval/index.d.ts.map +1 -1
  99. package/dist/meta-eval/index.js +200 -2
  100. package/dist/meta-eval/index.js.map +1 -1
  101. package/dist/{mint-DjfDUMHr.js → mint-vWOdD8Ae.js} +2 -2
  102. package/dist/{mint-DjfDUMHr.js.map → mint-vWOdD8Ae.js.map} +1 -1
  103. package/dist/multishot/golden/index.d.ts +1 -1
  104. package/dist/multishot/index.d.ts +2 -2
  105. package/dist/openapi.json +1 -1
  106. package/dist/pipelines/index.d.ts +5 -5
  107. package/dist/pipelines/index.js +3 -3
  108. package/dist/{pre-registration-DHz6P_6f.d.ts → pre-registration-BoI4ucR3.d.ts} +2 -2
  109. package/dist/{pre-registration-DHz6P_6f.d.ts.map → pre-registration-BoI4ucR3.d.ts.map} +1 -1
  110. package/dist/{produced-state-CxmbFxFd.js → produced-state-7VYDwtkk.js} +3 -3
  111. package/dist/{produced-state-CxmbFxFd.js.map → produced-state-7VYDwtkk.js.map} +1 -1
  112. package/dist/{promotion-policy-BBBcz5_3.d.ts → promotion-policy-CvMda3kU.d.ts} +2 -2
  113. package/dist/{promotion-policy-BBBcz5_3.d.ts.map → promotion-policy-CvMda3kU.d.ts.map} +1 -1
  114. package/dist/{provenance-Dp-vvyrU.d.ts → provenance-CRY67X50.d.ts} +39 -165
  115. package/dist/provenance-CRY67X50.d.ts.map +1 -0
  116. package/dist/{query-BPGMVlbM.js → query-D1nLIKt7.js} +2 -2
  117. package/dist/{query-BPGMVlbM.js.map → query-D1nLIKt7.js.map} +1 -1
  118. package/dist/{query-Na5gEIGd.d.ts → query-D6W6MaGx.d.ts} +3 -3
  119. package/dist/{query-Na5gEIGd.d.ts.map → query-D6W6MaGx.d.ts.map} +1 -1
  120. package/dist/{registry-xEb_xfns.d.ts → registry-7pOUBrtX.d.ts} +4 -4
  121. package/dist/{registry-xEb_xfns.d.ts.map → registry-7pOUBrtX.d.ts.map} +1 -1
  122. package/dist/{release-confidence-D6lQw_o7.d.ts → release-confidence-BAcNYOf1.d.ts} +4 -4
  123. package/dist/{release-confidence-D6lQw_o7.d.ts.map → release-confidence-BAcNYOf1.d.ts.map} +1 -1
  124. package/dist/{release-confidence-CzUHc4z4.js → release-confidence-BsGEg_xg.js} +3 -3
  125. package/dist/{release-confidence-CzUHc4z4.js.map → release-confidence-BsGEg_xg.js.map} +1 -1
  126. package/dist/reporting.d.ts +3 -3
  127. package/dist/reporting.js +1 -1
  128. package/dist/{researcher-CMUTQXD7.d.ts → researcher-jsW1X94L.d.ts} +7 -8
  129. package/dist/researcher-jsW1X94L.d.ts.map +1 -0
  130. package/dist/{reward-hacking-SkxYgT0x.js → reward-hacking-CKW4teig.js} +2 -2
  131. package/dist/{reward-hacking-SkxYgT0x.js.map → reward-hacking-CKW4teig.js.map} +1 -1
  132. package/dist/{reward-hacking-CgPRUesA.d.ts → reward-hacking-ZXEi9VCq.d.ts} +2 -2
  133. package/dist/{reward-hacking-CgPRUesA.d.ts.map → reward-hacking-ZXEi9VCq.d.ts.map} +1 -1
  134. package/dist/rl.d.ts +7 -7
  135. package/dist/rl.js +4 -4
  136. package/dist/rollout/index.d.ts +1 -1
  137. package/dist/rollout/index.js +2 -2
  138. package/dist/{rollout-Crypdx8s.js → rollout-C-znbbYg.js} +2 -2
  139. package/dist/{rollout-Crypdx8s.js.map → rollout-C-znbbYg.js.map} +1 -1
  140. package/dist/{rubric-predictive-validity-DluJLCKQ.d.ts → rubric-predictive-validity-Dl1dvKCv.d.ts} +2 -2
  141. package/dist/{rubric-predictive-validity-DluJLCKQ.d.ts.map → rubric-predictive-validity-Dl1dvKCv.d.ts.map} +1 -1
  142. package/dist/{run-record-DQjRcYwA.d.ts → run-record-DTv1MdjK.d.ts} +2 -2
  143. package/dist/{run-record-DQjRcYwA.d.ts.map → run-record-DTv1MdjK.d.ts.map} +1 -1
  144. package/dist/{run-record-DLORoL7t.js → run-record-ZIsR9Fif.js} +2 -2
  145. package/dist/{run-record-DLORoL7t.js.map → run-record-ZIsR9Fif.js.map} +1 -1
  146. package/dist/{schema-DID1Cqct.d.ts → schema-CR5cpjQ3.d.ts} +57 -2
  147. package/dist/{schema-DID1Cqct.d.ts.map → schema-CR5cpjQ3.d.ts.map} +1 -1
  148. package/dist/{schema-CdIX2aHu.js → schema-CSf6qWgZ.js} +35 -1
  149. package/dist/{schema-CdIX2aHu.js.map → schema-CSf6qWgZ.js.map} +1 -1
  150. package/dist/semantic-concept-judge-Ct3QU7t5.js +780 -0
  151. package/dist/semantic-concept-judge-Ct3QU7t5.js.map +1 -0
  152. package/dist/{series-convergence-D9WgpXGi.d.ts → series-convergence-DeG33RpC.d.ts} +2 -2
  153. package/dist/{series-convergence-D9WgpXGi.d.ts.map → series-convergence-DeG33RpC.d.ts.map} +1 -1
  154. package/dist/{server-CCEnywOR.js → server-BR6onwZB.js} +2 -2
  155. package/dist/{server-CCEnywOR.js.map → server-BR6onwZB.js.map} +1 -1
  156. package/dist/{skillopt-optimization-method-LHi02MzH.js → skillopt-optimization-method-BzdphODy.js} +6 -6
  157. package/dist/{skillopt-optimization-method-LHi02MzH.js.map → skillopt-optimization-method-BzdphODy.js.map} +1 -1
  158. package/dist/{statistical-heldout-Yldkntvy.d.ts → statistical-heldout-DTyB_6-1.d.ts} +3 -3
  159. package/dist/{statistical-heldout-Yldkntvy.d.ts.map → statistical-heldout-DTyB_6-1.d.ts.map} +1 -1
  160. package/dist/{store-Cq9oOrI1.d.ts → store-BErPvYBr.d.ts} +2 -2
  161. package/dist/{store-Cq9oOrI1.d.ts.map → store-BErPvYBr.d.ts.map} +1 -1
  162. package/dist/{store-otlp-CHjBvWQY.js → store-otlp-Dow0pk_5.js} +2 -2
  163. package/dist/{store-otlp-CHjBvWQY.js.map → store-otlp-Dow0pk_5.js.map} +1 -1
  164. package/dist/{store-tool-spans-BvdUbeOB.d.ts → store-tool-spans-CCZNsihA.d.ts} +8 -8
  165. package/dist/{store-tool-spans-BvdUbeOB.d.ts.map → store-tool-spans-CCZNsihA.d.ts.map} +1 -1
  166. package/dist/{store-tool-spans-B9o6tU8f.js → store-tool-spans-CeNj_m2L.js} +3 -3
  167. package/dist/{store-tool-spans-B9o6tU8f.js.map → store-tool-spans-CeNj_m2L.js.map} +1 -1
  168. package/dist/storyboard/index.d.ts +1 -1
  169. package/dist/{summary-report-DRstQNBX.d.ts → summary-report-gMrbYawB.d.ts} +3 -3
  170. package/dist/{summary-report-DRstQNBX.d.ts.map → summary-report-gMrbYawB.d.ts.map} +1 -1
  171. package/dist/{task-failure-attributes-CBGtLS_H.js → task-failure-attributes-CZjZeBsY.js} +3 -3
  172. package/dist/{task-failure-attributes-CBGtLS_H.js.map → task-failure-attributes-CZjZeBsY.js.map} +1 -1
  173. package/dist/{tool-groups-DjwlMBvW.d.ts → tool-groups-Cp4Xdzrp.d.ts} +3 -3
  174. package/dist/tool-groups-Cp4Xdzrp.d.ts.map +1 -0
  175. package/dist/{tool-waste-CwGHzBzX.js → tool-waste-B9tdWV6g.js} +253 -6
  176. package/dist/tool-waste-B9tdWV6g.js.map +1 -0
  177. package/dist/{tool-waste-BrmLKxMw.d.ts → tool-waste-D23I0zWm.d.ts} +4 -4
  178. package/dist/{tool-waste-BrmLKxMw.d.ts.map → tool-waste-D23I0zWm.d.ts.map} +1 -1
  179. package/dist/trace-repair/index.d.ts +2 -2
  180. package/dist/traces.d.ts +10 -10
  181. package/dist/traces.js +7 -7
  182. package/dist/{trajectory-r1bQqvBQ.d.ts → trajectory-D7qrNvaN.d.ts} +3 -3
  183. package/dist/{trajectory-r1bQqvBQ.d.ts.map → trajectory-D7qrNvaN.d.ts.map} +1 -1
  184. package/dist/trajectory-replay/index.d.ts +3 -3
  185. package/dist/{types-CCZ34qmV.d.ts → types-BDV4PiMR.d.ts} +3 -3
  186. package/dist/{types-CCZ34qmV.d.ts.map → types-BDV4PiMR.d.ts.map} +1 -1
  187. package/dist/{types-nokrtr7M.d.ts → types-Ba5UQyVD.d.ts} +4 -4
  188. package/dist/{types-nokrtr7M.d.ts.map → types-Ba5UQyVD.d.ts.map} +1 -1
  189. package/dist/{types-DMoNFDWi.d.ts → types-DN2WdT5S.d.ts} +3 -3
  190. package/dist/{types-DMoNFDWi.d.ts.map → types-DN2WdT5S.d.ts.map} +1 -1
  191. package/dist/types-gvRsyJLh.d.ts +831 -0
  192. package/dist/types-gvRsyJLh.d.ts.map +1 -0
  193. package/dist/wire/index.d.ts +3 -3
  194. package/dist/wire/index.js +1 -1
  195. package/docs/plants.md +69 -1
  196. package/docs/public-api.md +6 -4
  197. package/docs/trace-analysis.md +55 -8
  198. package/package.json +1 -1
  199. package/dist/backend-integrity-e79K3UPD.d.ts.map +0 -1
  200. package/dist/bounded-process-VIi0KSL2.js.map +0 -1
  201. package/dist/chat-client-DI79OPye.js.map +0 -1
  202. package/dist/chat-json-call-5Jxna-aV.js.map +0 -1
  203. package/dist/dspy-rlm-engine-CS3qcCEk.js.map +0 -1
  204. package/dist/failure-cluster-6YSvsKlp.d.ts +0 -58
  205. package/dist/failure-cluster-6YSvsKlp.d.ts.map +0 -1
  206. package/dist/index-DDAPhUJJ.d.ts.map +0 -1
  207. package/dist/index-DnglhM0A.d.ts.map +0 -1
  208. package/dist/kind-factory-DMeEoMQZ.js.map +0 -1
  209. package/dist/llm-client-BFMRpmqb.js.map +0 -1
  210. package/dist/provenance-Dp-vvyrU.d.ts.map +0 -1
  211. package/dist/raw-provider-sink-BU29Sh8h.d.ts +0 -134
  212. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +0 -1
  213. package/dist/researcher-CMUTQXD7.d.ts.map +0 -1
  214. package/dist/semantic-concept-judge-I36eejJx.js +0 -382
  215. package/dist/semantic-concept-judge-I36eejJx.js.map +0 -1
  216. package/dist/tool-groups-DjwlMBvW.d.ts.map +0 -1
  217. package/dist/tool-waste-CwGHzBzX.js.map +0 -1
  218. package/dist/types-Bfk0uxRj.d.ts +0 -443
  219. package/dist/types-Bfk0uxRj.d.ts.map +0 -1
@@ -0,0 +1,831 @@
1
+ import { t as AgentEvalError } from "./errors-DEE6u6ot.js";
2
+ import { b as CustomTokenPricing, c as CostLedgerHandle, g as CostReceiptInput, x as MaximumCharge } from "./cost-ledger-DbQdN3nO.js";
3
+ //#region src/judge-families.d.ts
4
+ /**
5
+ * Judge model-family classification + cross-family enforcement.
6
+ *
7
+ * A judge ensemble built entirely from one provider family shares that
8
+ * family's blind spots and self-preference — its "agreement" is correlated
9
+ * bias, not independent signal. `assertCrossFamily` makes the consumer prove
10
+ * the ensemble spans ≥2 families; `judgeFamily` is the single regex map that
11
+ * replaces the per-consumer copies (tax/legal/creative/gtm each ship one).
12
+ */
13
+ /** Provider family a model belongs to. `unknown` when no rule matches. */
14
+ type JudgeFamily = 'anthropic' | 'openai' | 'google' | 'meta' | 'mistral' | 'deepseek' | 'xai' | 'qwen' | 'cohere' | 'amazon' | 'moonshot' | 'zhipu' | 'unknown';
15
+ /**
16
+ * Classify a model id into its provider family. Strips a `@snapshot` suffix
17
+ * and prefers an explicit `provider/...` prefix; otherwise matches the model
18
+ * name. Returns `unknown` when nothing matches (callers decide whether that's
19
+ * acceptable — `assertCrossFamily` counts it as its own family).
20
+ */
21
+ declare function judgeFamily(modelId: string): JudgeFamily;
22
+ interface AssertCrossFamilyOptions {
23
+ /** Minimum number of distinct families the ensemble must span. Default 2. */
24
+ minFamilies?: number;
25
+ /** When false (default), `unknown`-family models do NOT count toward the
26
+ * family total — an ensemble of all-unclassifiable models is not provably
27
+ * cross-family. Set true to count `unknown` as one shared family. */
28
+ allowUnknown?: boolean;
29
+ }
30
+ declare class CrossFamilyError extends Error {
31
+ readonly families: JudgeFamily[];
32
+ readonly models: string[];
33
+ constructor(message: string, families: JudgeFamily[], models: string[]);
34
+ }
35
+ /**
36
+ * Throw unless the judge models span at least `minFamilies` distinct provider
37
+ * families. Pass the model ids backing your judge ensemble. Fail-loud by
38
+ * design — a correlated single-family ensemble silently inflates agreement.
39
+ *
40
+ * Scope: this reads the ids you REQUEST. It proves the panel was configured
41
+ * across families; it cannot prove the panel RAN across families, because a
42
+ * routing gateway may answer several different ids from one provider. Where
43
+ * the diversity claim is load-bearing (a published leaderboard, a
44
+ * certification, a non-self-judging exclusion), assert on the ids the
45
+ * provider echoed instead: `assertCrossFamilyServed` in
46
+ * ./integrity/served-model.
47
+ */
48
+ declare function assertCrossFamily(models: string[], opts?: AssertCrossFamilyOptions): JudgeFamily[];
49
+ //#endregion
50
+ //#region src/integrity/served-model.d.ts
51
+ /** How a served id relates to the id that was requested. */
52
+ type ServedModelVerdict =
53
+ /** Byte-identical after normalisation — the requested model answered. */
54
+ 'exact' |
55
+ /** Same model, different spelling (provider prefix, snapshot, tier suffix). */
56
+ 'alias' |
57
+ /** A different model of the SAME provider family answered. */
58
+ 'substituted-within-family' |
59
+ /** A different provider's model answered. */
60
+ 'substituted-cross-family' |
61
+ /** The response carried no model id — identity is unproven either way. */
62
+ 'unreported';
63
+ interface ServedModelCheck {
64
+ /** The id the caller asked for. */
65
+ requested: string;
66
+ /** The id echoed on the response; `null` when the response omitted it. */
67
+ served: string | null;
68
+ requestedFamily: JudgeFamily;
69
+ /** `null` when `served` is null. */
70
+ servedFamily: JudgeFamily | null;
71
+ verdict: ServedModelVerdict;
72
+ /** True for every verdict except `exact` and `alias`. */
73
+ substituted: boolean;
74
+ }
75
+ /**
76
+ * Classify one requested/served pair. Pure — no I/O — so it is safe inside
77
+ * response handlers, reducers, and CI gates.
78
+ *
79
+ * `served` is the id echoed by the provider (OpenAI-compatible bodies put it
80
+ * at `model`). `null`/`undefined` means the body omitted it; that is
81
+ * `unreported`, NOT a pass — a provider that does not name what answered has
82
+ * not proven identity, and a transport that drops the field must not read as
83
+ * agreement.
84
+ */
85
+ declare function checkServedModel(requested: string, served: string | null | undefined): ServedModelCheck;
86
+ declare class ModelSubstitutionError extends AgentEvalError {
87
+ readonly checks: ReadonlyArray<ServedModelCheck>;
88
+ constructor(message: string, checks: ReadonlyArray<ServedModelCheck>);
89
+ }
90
+ /**
91
+ * Consumer-facing name for the substitution policy a metered surface applies.
92
+ * `'exact'` rejects every substitution. `'allow-within-family'` accepts a
93
+ * different model of the same provider family; it keeps family-level claims
94
+ * valid and forfeits per-model claims. Maps to
95
+ * `AssertServedModelOptions.allowWithinFamily`.
96
+ */
97
+ type ServedModelPolicy = 'exact' | 'allow-within-family';
98
+ interface AssertServedModelOptions {
99
+ /**
100
+ * Accept a different model of the same provider family (e.g. requested
101
+ * `deepseek-v3.2`, served `deepseek-v4-flash`). Default false. Setting this
102
+ * keeps family-level claims valid and forfeits per-model claims.
103
+ */
104
+ allowWithinFamily?: boolean;
105
+ /**
106
+ * Accept a response that carried no model id. Default false — an
107
+ * unidentified response cannot support a per-model or per-family claim.
108
+ */
109
+ allowUnreported?: boolean;
110
+ /** Prefixed to the thrown message, e.g. the judge or campaign cell name. */
111
+ context?: string;
112
+ }
113
+ /**
114
+ * The one place the accept/reject policy lives, so a caller that reports
115
+ * substitution (a preflight table, a run record) and a caller that throws on it
116
+ * can never drift apart. A cross-family substitution is never acceptable.
117
+ */
118
+ declare function servedModelAcceptable(check: ServedModelCheck, opts?: AssertServedModelOptions): boolean;
119
+ /**
120
+ * Throw `ModelSubstitutionError` unless the served id is the requested model.
121
+ * Returns the check on success so callers can record the served id alongside
122
+ * the result.
123
+ */
124
+ declare function assertServedModel(requested: string, served: string | null | undefined, opts?: AssertServedModelOptions): ServedModelCheck;
125
+ /**
126
+ * Batch form: check every pair and throw naming EVERY substitution, so one
127
+ * failure does not hide the rest. Returns all checks on success.
128
+ */
129
+ declare function assertServedModels(pairs: ReadonlyArray<{
130
+ requested: string;
131
+ served: string | null | undefined;
132
+ }>, opts?: AssertServedModelOptions): ServedModelCheck[];
133
+ interface AssertCrossFamilyServedOptions extends AssertServedModelOptions {
134
+ /** Minimum distinct SERVED families required. Default 2. */
135
+ minFamilies?: number;
136
+ /** Count `unknown`-family served ids toward the total. Default false. */
137
+ allowUnknown?: boolean;
138
+ }
139
+ declare class ServedCrossFamilyError extends AgentEvalError {
140
+ readonly families: JudgeFamily[];
141
+ readonly checks: ReadonlyArray<ServedModelCheck>;
142
+ constructor(message: string, families: JudgeFamily[], checks: ReadonlyArray<ServedModelCheck>);
143
+ }
144
+ /**
145
+ * Family-diversity rule over the models that actually ANSWERED.
146
+ *
147
+ * `assertCrossFamily` (../judge-families) reads the requested ids and so
148
+ * cannot see a gateway that answers three "different" requests from one
149
+ * provider. This one asserts no substitution first, then counts families from
150
+ * the served ids — a panel that collapsed to one family under the hood fails
151
+ * here even though its request list looked diverse.
152
+ */
153
+ declare function assertCrossFamilyServed(pairs: ReadonlyArray<{
154
+ requested: string;
155
+ served: string | null | undefined;
156
+ }>, opts?: AssertCrossFamilyServedOptions): JudgeFamily[];
157
+ //#endregion
158
+ //#region src/trace/raw-provider-sink.d.ts
159
+ /**
160
+ * RawProviderSink — first-class persistence for the actual HTTP-level
161
+ * request/response bodies of every LLM provider call.
162
+ *
163
+ * Why this is a separate sink from the structured `LlmSpan`:
164
+ *
165
+ * - `LlmSpan` records the *intent* — model name, messages, output text,
166
+ * usage. It's what dashboards read; it's NOT enough for forensics.
167
+ * - When a downstream consumer reports "the verifier used the wrong route"
168
+ * or "tokens look right but reasoning was missing," the only way to
169
+ * answer is the raw HTTP body. Span fields can lie (a proxy can echo
170
+ * a different `model` value than what actually answered); the raw
171
+ * response is ground truth.
172
+ *
173
+ * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
174
+ * matrix runner / BuilderSession sets it up automatically) and every
175
+ * request, response, and error is recorded — including retries, with the
176
+ * attempt index attached so a flaky call's full event chain is recoverable.
177
+ *
178
+ * Redaction is enforced at sink time. The default redactor strips
179
+ * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
180
+ * payload field whose key matches `apiKey | api_key | bearer | password |
181
+ * secret | token` (case-insensitive). Override via the sink constructor or
182
+ * the per-call `redactor`. The `redactedFields` array on the persisted
183
+ * event lets a reviewer see what was stripped without exposing the values.
184
+ */
185
+ type RawProviderDirection = 'request' | 'response' | 'error';
186
+ interface RawProviderEvent {
187
+ /** Stable id. Generated by the sink if omitted. */
188
+ eventId: string;
189
+ /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
190
+ runId?: string;
191
+ spanId?: string;
192
+ /**
193
+ * Logical provider name. Free-form so callers can use whatever id matches
194
+ * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
195
+ * omitted, derived from `baseUrl` in `LlmClientOptions`.
196
+ */
197
+ provider: string;
198
+ model: string;
199
+ /** Endpoint path, e.g. `'/v1/chat/completions'`. */
200
+ endpoint: string;
201
+ /** Base URL used for the call (already-normalised — no trailing slash). */
202
+ baseUrl: string;
203
+ /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
204
+ attemptIndex: number;
205
+ direction: RawProviderDirection;
206
+ /** Unix ms. */
207
+ timestamp: number;
208
+ /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
209
+ durationMs?: number;
210
+ statusCode?: number;
211
+ requestHeaders?: Record<string, string>;
212
+ requestBody?: unknown;
213
+ responseHeaders?: Record<string, string>;
214
+ responseBody?: unknown;
215
+ /** Set on `direction: 'error'` events. */
216
+ errorMessage?: string;
217
+ /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
218
+ redactedFields: string[];
219
+ }
220
+ interface RawProviderSinkFilter {
221
+ runId?: string;
222
+ spanId?: string;
223
+ direction?: RawProviderDirection;
224
+ attemptIndex?: number;
225
+ }
226
+ interface RawProviderSink {
227
+ record(event: RawProviderEvent): Promise<void>;
228
+ /** Optional listing — implementations that durably persist (file, db) should support this. */
229
+ list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
230
+ /** Optional teardown for backed implementations. */
231
+ close?(): Promise<void>;
232
+ }
233
+ type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
234
+ /**
235
+ * Default redactor — strips well-known auth headers and any body field whose
236
+ * key matches the credential pattern. Records every redacted path on
237
+ * `event.redactedFields` so a downstream reviewer can see what was removed.
238
+ */
239
+ declare function defaultProviderRedactor(event: RawProviderEvent): RawProviderEvent;
240
+ interface InMemoryRawProviderSinkOptions {
241
+ redactor?: ProviderRedactor;
242
+ }
243
+ declare class InMemoryRawProviderSink implements RawProviderSink {
244
+ private events;
245
+ private redactor;
246
+ constructor(opts?: InMemoryRawProviderSinkOptions);
247
+ record(event: RawProviderEvent): Promise<void>;
248
+ list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
249
+ size(): number;
250
+ }
251
+ declare class NoopRawProviderSink implements RawProviderSink {
252
+ record(): Promise<void>;
253
+ /**
254
+ * Returns an empty array. Implemented so `assertRunCaptured` does not
255
+ * trip the `no_raw_sink` issue when a caller explicitly opts out of
256
+ * capture by passing this sink — opt-out is a deliberate choice, not a
257
+ * misconfiguration.
258
+ */
259
+ list(): Promise<RawProviderEvent[]>;
260
+ }
261
+ interface FileSystemRawProviderSinkOptions {
262
+ /** Directory the NDJSON file is written into. Created if missing. */
263
+ dir: string;
264
+ /** File name; default `'raw-provider-events.ndjson'`. */
265
+ fileName?: string;
266
+ /** Bytes after which the writer rolls over to a new file (default 32 MiB). */
267
+ rollAtBytes?: number;
268
+ redactor?: ProviderRedactor;
269
+ }
270
+ declare class FileSystemRawProviderSink implements RawProviderSink {
271
+ private dir;
272
+ private fileName;
273
+ private rollAtBytes;
274
+ private redactor;
275
+ private bytesWritten;
276
+ private rollIndex;
277
+ private initPromise;
278
+ constructor(opts: FileSystemRawProviderSinkOptions);
279
+ private ensureInit;
280
+ private currentPath;
281
+ record(event: RawProviderEvent): Promise<void>;
282
+ list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
283
+ }
284
+ /**
285
+ * Best-effort provider id from a base URL. Falls back to the URL host when
286
+ * none of the well-known patterns match.
287
+ */
288
+ declare function providerFromBaseUrl(baseUrl: string): string;
289
+ //#endregion
290
+ //#region src/llm-client.d.ts
291
+ interface LlmMessage {
292
+ role: 'system' | 'user' | 'assistant' | 'tool';
293
+ /**
294
+ * Either a plain text content string OR a multimodal content array
295
+ * (text + image_url parts) for vision-capable models.
296
+ */
297
+ content: string | Array<{
298
+ type: 'text';
299
+ text: string;
300
+ } | {
301
+ type: 'image_url';
302
+ image_url: {
303
+ url: string;
304
+ detail?: 'auto' | 'low' | 'high';
305
+ };
306
+ }>;
307
+ /** Tool invocations made by an `assistant` message earlier in the turn. */
308
+ toolCalls?: LlmToolCall[];
309
+ /** The invocation a `tool` message answers. Required when role is `tool`. */
310
+ toolCallId?: string;
311
+ }
312
+ type LlmThinkingMode = 'enabled' | 'disabled';
313
+ /** Canonical function-tool definition offered to the model. */
314
+ interface LlmToolDefinition {
315
+ type: 'function';
316
+ function: {
317
+ name: string;
318
+ description?: string;
319
+ parameters: Record<string, unknown>;
320
+ };
321
+ }
322
+ /** Canonical tool-choice policy; meaningful only when `tools` is present. */
323
+ type LlmToolChoice = 'auto' | 'none' | 'required' | {
324
+ type: 'function';
325
+ function: {
326
+ name: string;
327
+ };
328
+ };
329
+ /** One tool invocation on a response or an assistant history message. */
330
+ interface LlmToolCall {
331
+ id: string;
332
+ name: string;
333
+ /** JSON-encoded arguments, exactly as the provider produced them. */
334
+ argumentsJson: string;
335
+ }
336
+ interface LlmCallRequest {
337
+ model: string;
338
+ messages: LlmMessage[];
339
+ /** Optional JSON-mode response format (response_format: json_object). */
340
+ jsonMode?: boolean;
341
+ /** Optional structured output via JSON Schema. Falls back to json_object on 400. */
342
+ jsonSchema?: {
343
+ name: string;
344
+ schema: Record<string, unknown>;
345
+ };
346
+ /** Function tools offered to the model for this call. */
347
+ tools?: LlmToolDefinition[];
348
+ /** Tool-choice policy for `tools`. */
349
+ toolChoice?: LlmToolChoice;
350
+ temperature?: number;
351
+ maxTokens?: number;
352
+ /** Ask the provider for the log-probability of each sampled token and its
353
+ * `topLogprobs` most likely alternatives. Sends `logprobs: true` with
354
+ * `top_logprobs`. A provider that ignores the field returns
355
+ * `LlmCallResult.logprobs === null`; nothing is inferred from its absence. */
356
+ logprobs?: {
357
+ topLogprobs: number;
358
+ };
359
+ /** OpenAI-compatible reasoning mode. Omitted when the provider default should apply. */
360
+ thinking?: LlmThinkingMode;
361
+ /** Per-call timeout, default 300s. */
362
+ timeoutMs?: number;
363
+ }
364
+ /** Conservative priced bound for the exact text request sent to a provider.
365
+ * Returns undefined when output or multimodal input is not bounded, causing a
366
+ * capped CostLedger to reject the call before execution. Pass
367
+ * `customTokenPricing` when package pricing does not cover the model or endpoint. */
368
+ interface LlmChargeBounds {
369
+ /** Total provider attempts the transport may make for this call. Default 3. */
370
+ maximumAttempts?: number;
371
+ /** The transport sends JSON mode instead of a response schema. */
372
+ jsonSchemaTransport?: 'native' | 'json-object';
373
+ /** Default provider reasoning mode the transport applies. */
374
+ thinking?: LlmThinkingMode;
375
+ /** Token rates used when the provider omits cost or package pricing does not cover the model. */
376
+ customTokenPricing?: CustomTokenPricing;
377
+ }
378
+ declare function maximumChargeForLlmRequest(request: Pick<LlmCallRequest, 'model' | 'messages' | 'jsonSchema' | 'tools' | 'toolChoice' | 'maxTokens' | 'thinking'>, options?: LlmChargeBounds): MaximumCharge | undefined;
379
+ interface LlmUsage {
380
+ promptTokens: number;
381
+ completionTokens: number;
382
+ totalTokens: number;
383
+ /** False when the provider omitted or malformed prompt/completion usage. */
384
+ captured?: boolean;
385
+ /** Reasoning-token subset of completionTokens, when reported. */
386
+ reasoningTokens?: number;
387
+ /** Proxies populate this when prompt caching is on. */
388
+ cachedPromptTokens?: number;
389
+ }
390
+ /** One sampled token with its own log probability and the alternatives the
391
+ * provider ranked at that position. */
392
+ interface LlmTokenLogprob {
393
+ token: string;
394
+ logprob: number;
395
+ top: ReadonlyArray<{
396
+ token: string;
397
+ logprob: number;
398
+ }>;
399
+ }
400
+ interface LlmCallResult {
401
+ /** The text content of the first choice. Empty string if none. */
402
+ content: string;
403
+ /** Tool invocations from the first choice, when the model called tools. */
404
+ toolCalls?: LlmToolCall[];
405
+ usage: LlmUsage;
406
+ /**
407
+ * Cost in USD. Uses the provider's reported cost when present, otherwise
408
+ * caller-supplied token pricing. `null` when neither is available.
409
+ */
410
+ costUsd: number | null;
411
+ /**
412
+ * Model id used for attribution (cost, pricing, log lines). The response's
413
+ * echoed id when the provider sent one, else the requested id.
414
+ *
415
+ * NOT evidence of which model answered — read `servedModel` for that. A
416
+ * provider that omits `model` makes this equal to the request, which is
417
+ * exactly the case an identity check must be able to distinguish.
418
+ */
419
+ model: string;
420
+ /**
421
+ * The model id the provider echoed on the response, verbatim; `null` when
422
+ * the body carried none. This is the only field that can witness a gateway
423
+ * substituting a different model for the one requested — compare it with
424
+ * `assertServedModel` / `checkServedModel` (src/integrity/served-model.ts).
425
+ *
426
+ * Optional so hand-built results (mock/custom transports) still typecheck,
427
+ * but omitting it is not a pass: the identity checks read `undefined` as
428
+ * `unreported` and reject it by default. A transport that knows which model
429
+ * answered should say so.
430
+ */
431
+ servedModel?: string | null;
432
+ /** Wall-clock duration of the HTTP call (last attempt, if retried). */
433
+ durationMs: number;
434
+ /**
435
+ * `finish_reason` echoed from the first choice (`stop`, `length`,
436
+ * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
437
+ * Exposed so a free-form `callLlm` caller CAN detect a truncated answer
438
+ * (`length`) instead of treating a cut-off completion as complete. Note:
439
+ * `callLlm` does not itself reject on it — acting on this signal is the
440
+ * caller's responsibility (in-repo free-form drivers do not yet enforce it).
441
+ */
442
+ finishReason?: string | null;
443
+ /**
444
+ * True when `content.trim()` is empty. An empty completion is a silent zero
445
+ * for free-form `callLlm` callers; this flag is the signal a caller can
446
+ * inspect to fail loud rather than proceed on an empty string. `callLlm`
447
+ * surfaces it but does not throw on it.
448
+ */
449
+ contentEmpty?: boolean;
450
+ /**
451
+ * Per-token log probabilities for the first choice, in emission order, when
452
+ * the request asked for them and the provider returned
453
+ * `choices[0].logprobs.content`. `null` when the provider returned none, so a
454
+ * caller can tell "not requested or not supported" from "requested and
455
+ * empty". Absent when the request did not ask for logprobs.
456
+ */
457
+ logprobs?: ReadonlyArray<LlmTokenLogprob> | null;
458
+ /** Raw response body. */
459
+ raw: Record<string, unknown>;
460
+ }
461
+ type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
462
+ /** Convert a provider result into the canonical paid-call receipt input.
463
+ * The receipt is JSON-clean: absent optional fields are omitted, never carried
464
+ * as explicit-undefined keys. Receipts cross JSON boundaries (the external
465
+ * optimizer's loopback proxy validates them with `assertJsonValue`), where an
466
+ * explicit-undefined value is rejected as non-serializable. */
467
+ declare function costReceiptFromLlm(result: LlmCallResult, customTokenPricing?: CustomTokenPricing): CostReceiptInput;
468
+ /** Structured-response failures retain their completed provider receipt. */
469
+ declare function costReceiptFromLlmError(error: Error, customTokenPricing?: CustomTokenPricing): CostReceiptInput | undefined;
470
+ declare class LlmCallError extends AgentEvalError {
471
+ readonly status: number;
472
+ readonly body: string;
473
+ readonly model: string;
474
+ constructor(message: string, status: number, body: string, model: string);
475
+ }
476
+ /** A provider response completed and incurred measurable usage, but its content
477
+ * could not satisfy the caller's response contract. The response envelope is
478
+ * retained so accounting can commit the receipt before the error propagates. */
479
+ declare class LlmResponseError extends AgentEvalError {
480
+ readonly result: LlmCallResult;
481
+ constructor(message: string, result: LlmCallResult, options?: {
482
+ cause?: unknown;
483
+ });
484
+ }
485
+ interface LlmClientOptions extends LlmChargeBounds {
486
+ /** Base URL (without trailing slash), ending at the `/v1` prefix. Required: there is no default endpoint. */
487
+ baseUrl?: string;
488
+ /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
489
+ apiKey?: string;
490
+ bearer?: string;
491
+ /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
492
+ authHeader?: {
493
+ name: string;
494
+ value: string;
495
+ };
496
+ /** Stable provider idempotency key, reused across retries of this logical call. */
497
+ idempotencyKey?: string;
498
+ /** Default timeout in ms. Per-call can override. */
499
+ defaultTimeoutMs?: number;
500
+ /**
501
+ * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
502
+ * each attempt's per-attempt timeout controller, so aborting it cancels
503
+ * the in-flight fetch. A caller abort is FATAL: it is not retried even
504
+ * though an AbortError otherwise matches the transient patterns.
505
+ */
506
+ signal?: AbortSignal;
507
+ /**
508
+ * Cross-attempt wall-clock budget in ms, measured from the first attempt.
509
+ * Before launching each attempt the loop checks the remaining budget and
510
+ * stops retrying once it is exhausted, rather than waiting the full
511
+ * per-attempt timeout on every retry. Bounds total time independent of
512
+ * total attempts × `timeoutMs`.
513
+ */
514
+ deadlineMs?: number;
515
+ /**
516
+ * JSON payload parsing policy. `extract` accepts fenced or prose-prefixed JSON.
517
+ * `exact` requires the complete response content to be one JSON value.
518
+ * Default: `extract`.
519
+ */
520
+ jsonPayloadMode?: 'extract' | 'exact';
521
+ /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
522
+ fetch?: typeof fetch;
523
+ /**
524
+ * Optional raw HTTP capture sink. When provided, every request, response,
525
+ * and error (across all retry attempts) is recorded to the sink, with auth
526
+ * headers and credential-shaped body fields redacted by default. This is
527
+ * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
528
+ * raw events record what actually crossed the wire.
529
+ */
530
+ rawSink?: RawProviderSink;
531
+ /**
532
+ * Logical provider id attached to raw events. When omitted, derived from
533
+ * `baseUrl` via `providerFromBaseUrl`.
534
+ */
535
+ provider?: string;
536
+ /** Trace context attached to raw events; populated by emitter-aware callers. */
537
+ traceContext?: {
538
+ runId?: string;
539
+ spanId?: string;
540
+ };
541
+ /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
542
+ redactor?: ProviderRedactor;
543
+ /**
544
+ * Reject a response whose echoed model is not the model that was requested.
545
+ * A routing gateway can accept one id and answer from another, which
546
+ * silently invalidates every per-model and per-family claim downstream.
547
+ * `true` uses the strict default (aliases pass, substitutions and
548
+ * unidentified responses throw `ModelSubstitutionError`); pass an options
549
+ * object to relax a specific case. Off by default — turning it on for a
550
+ * measurement run is the point.
551
+ */
552
+ assertServedModel?: boolean | AssertServedModelOptions;
553
+ }
554
+ /**
555
+ * True when an error is a transient transport/network fault worth retrying,
556
+ * as opposed to a deterministic failure (4xx schema reject, JSON parse) that
557
+ * a retry cannot fix. Inspects `LlmCallError.status`, then the error's
558
+ * name/message/code, then recurses into `error.cause` — undici nests the
559
+ * real socket fault one or more levels under `.cause`.
560
+ *
561
+ * This is the retry classifier for the package: `callLlm` and
562
+ * `withJudgeRetry` both route through it, so connection failures are treated
563
+ * consistently across transports.
564
+ */
565
+ declare function isTransientLlmError(err: unknown): boolean;
566
+ /**
567
+ * Strip a ```json / ``` code fence if the model emitted one.
568
+ * Idempotent for naked JSON. Some models (claude-code via router, certain
569
+ * deepseek models) wrap output even under json_object.
570
+ */
571
+ declare function stripFencedJson(raw: string): string;
572
+ //#endregion
573
+ //#region src/analyst/chat-client.d.ts
574
+ /**
575
+ * Unified chat interface using the package's canonical LLM request and result.
576
+ */
577
+ interface ChatClient {
578
+ /** Display name of the bound transport, included in telemetry. */
579
+ readonly transport: ChatTransport;
580
+ /** Default model when the caller omits one. */
581
+ readonly defaultModel?: string;
582
+ /** Total provider attempts this transport can make for one chat call. */
583
+ readonly maximumAttempts?: number;
584
+ /** Implementations must enforce `req.maxTokens` when it is present. */
585
+ chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
586
+ }
587
+ type ChatTransport = 'sandbox-sdk' | 'custom' | 'openai-compatible' | 'mock';
588
+ interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
589
+ /** Optional — falls back to ChatClient.defaultModel. */
590
+ model?: string;
591
+ }
592
+ type ChatResponse = LlmCallResult;
593
+ interface ChatCallOpts {
594
+ /** Cancel the in-flight request. */
595
+ signal?: AbortSignal;
596
+ /** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
597
+ maxCostUsd?: number;
598
+ /** Correlation tag carried into request headers when the transport allows. */
599
+ correlationId?: string;
600
+ /** Stable provider idempotency key for retries/redrives of one paid call. */
601
+ idempotencyKey?: string;
602
+ }
603
+ type CreateChatClientOpts = SandboxSdkTransportOpts | CustomTransportOpts | OpenAiCompatibleTransportOpts | MockTransportOpts;
604
+ interface BaseTransportOpts {
605
+ defaultModel?: string;
606
+ /** Total provider attempts. Required for opaque transports used in capped runs. */
607
+ maximumAttempts?: number;
608
+ }
609
+ /**
610
+ * Sandbox-SDK transport. The caller supplies a canonical chat function for an
611
+ * already-configured Sandbox handle, so agent-eval does not import the SDK.
612
+ */
613
+ interface SandboxSdkTransportOpts extends BaseTransportOpts {
614
+ transport: 'sandbox-sdk';
615
+ chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
616
+ }
617
+ /** Caller-adapted SDK or transport returning the canonical ChatResponse shape. */
618
+ interface CustomTransportOpts extends BaseTransportOpts {
619
+ transport: 'custom';
620
+ chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
621
+ }
622
+ /**
623
+ * OpenAI-compatible HTTP transport: the caller names a `/v1` endpoint and
624
+ * hands over a bearer, and this package drives `POST {baseUrl}/chat/completions`.
625
+ *
626
+ * Use this instead of hand-rolling a `custom` transport around `fetch`. The
627
+ * result carries `servedModel` — the model id the provider echoed, verbatim —
628
+ * which is the only field `assertServedModel` and `assertCrossFamilyServed`
629
+ * can read to witness a gateway answering from a different model than the one
630
+ * requested.
631
+ *
632
+ * `baseUrl` ends at the `/v1` prefix; the `/chat/completions` path is this
633
+ * package's to append. A `baseUrl` that already carries the path is a
634
+ * construction error, not a request to a doubled URL.
635
+ *
636
+ * Exactly one credential form is required — `apiKey`, `bearer`, or
637
+ * `authHeader`. There is no environment fallback: agent-eval never goes
638
+ * looking for a key.
639
+ */
640
+ interface OpenAiCompatibleTransportOpts extends BaseTransportOpts, Pick<LlmClientOptions, 'assertServedModel' | 'customTokenPricing' | 'deadlineMs' | 'defaultTimeoutMs' | 'fetch' | 'jsonPayloadMode' | 'jsonSchemaTransport' | 'provider' | 'rawSink' | 'signal' | 'thinking'> {
641
+ transport: 'openai-compatible';
642
+ /** Endpoint ending at the `/v1` prefix. Required — there is no default endpoint. */
643
+ baseUrl: string;
644
+ /** Bearer credential. One of `apiKey`, `bearer`, or `authHeader` is required. */
645
+ apiKey?: string;
646
+ /** Bearer credential, alternate spelling. */
647
+ bearer?: string;
648
+ /** Non-bearer authorization header, for endpoints that want their own scheme. */
649
+ authHeader?: {
650
+ name: string;
651
+ value: string;
652
+ };
653
+ }
654
+ /**
655
+ * Mock transport for tests. The handler receives the request and returns
656
+ * whatever the test wants. No retries, no JSON-schema degrade.
657
+ */
658
+ interface MockTransportOpts extends BaseTransportOpts {
659
+ transport: 'mock';
660
+ handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
661
+ }
662
+ /**
663
+ * Build a ChatClient bound to a specific transport. The returned client
664
+ * is safe to share across analysts in a single registry run.
665
+ */
666
+ declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
667
+ //#endregion
668
+ //#region src/types.d.ts
669
+ interface Scenario {
670
+ id: string;
671
+ persona: string;
672
+ label: string;
673
+ thesis: string;
674
+ dimensions: string[];
675
+ turns: Turn[];
676
+ artifactChecks: ArtifactCheck[];
677
+ systemPromptAppend?: string;
678
+ }
679
+ interface Turn {
680
+ user: string;
681
+ expectedBehaviors: string[];
682
+ adversarial?: boolean;
683
+ feedbackType?: 'correction' | 'rejection' | 'vague' | 'contradictory' | 'escalation';
684
+ }
685
+ interface ArtifactCheck {
686
+ type: 'vault_file_exists' | 'vault_file_contains' | 'block_extracted' | 'code_valid' | 'generation_produced' | 'tool_created' | string;
687
+ target: string;
688
+ contains?: string;
689
+ minCount?: number;
690
+ description: string;
691
+ }
692
+ interface JudgeRubric {
693
+ name: string;
694
+ description: string;
695
+ dimensions: RubricDimension[];
696
+ }
697
+ interface RubricDimension {
698
+ name: string;
699
+ description: string;
700
+ anchor_low: string;
701
+ anchor_high: string;
702
+ weight: number;
703
+ }
704
+ interface TurnResult {
705
+ turnIndex: number;
706
+ userMessage: string;
707
+ agentResponse: string;
708
+ durationMs: number;
709
+ blocksExtracted: {
710
+ type: string;
711
+ title: string;
712
+ }[];
713
+ containsCode: boolean;
714
+ containsToolCall: boolean;
715
+ }
716
+ interface JudgeScore {
717
+ judgeName: string;
718
+ dimension: string;
719
+ score: number;
720
+ reasoning: string;
721
+ evidence?: string;
722
+ }
723
+ interface CollectedArtifacts {
724
+ vaultFiles: {
725
+ path: string;
726
+ content: string;
727
+ }[];
728
+ blocksExtracted: {
729
+ type: string;
730
+ fields: Record<string, string>;
731
+ }[];
732
+ codeBlocks: {
733
+ language: string;
734
+ code: string;
735
+ }[];
736
+ toolCalls: string[];
737
+ }
738
+ interface RouteMap {
739
+ signup?: string;
740
+ login?: string;
741
+ workspaces?: string;
742
+ threads?: string;
743
+ chat?: string;
744
+ tasks?: string;
745
+ events?: string;
746
+ approvals?: string;
747
+ vault?: string;
748
+ generations?: string;
749
+ [key: string]: string | undefined;
750
+ }
751
+ interface ProductClientConfig {
752
+ baseUrl: string;
753
+ routes: RouteMap;
754
+ /** Per-request timeout in ms before the request is aborted. Default 30s. */
755
+ timeoutMs?: number;
756
+ }
757
+ interface CompletionCriterion {
758
+ name: string;
759
+ check: (state: DriverState) => boolean;
760
+ progress?: (state: DriverState) => number;
761
+ }
762
+ /**
763
+ * How hard the simulated user pushes back. The driver LLM scales its tone
764
+ * and follow-up aggression to this:
765
+ * cooperative — forgiving early adopter; accepts reasonable answers.
766
+ * demanding — experienced professional; rejects vague or hedged answers.
767
+ * relentless — senior partner reviewing for a client who will litigate;
768
+ * interrogates every claim, accepts nothing undefended.
769
+ */
770
+ type PersonaRigor = 'cooperative' | 'demanding' | 'relentless';
771
+ interface PersonaConfig {
772
+ id: string;
773
+ role: string;
774
+ goal: string;
775
+ completionCriteria: CompletionCriterion[];
776
+ maxTurns: number;
777
+ /** How adversarial the simulated user is. Defaults to 'demanding'. */
778
+ rigor?: PersonaRigor;
779
+ /**
780
+ * Domain expertise the simulated user holds — quoted into the driver
781
+ * prompt so it challenges the agent with authority instead of vague
782
+ * dissatisfaction. e.g. "a 15-year M&A partner who knows GAAP
783
+ * working-capital mechanics cold".
784
+ */
785
+ expertise?: string;
786
+ /**
787
+ * Substantive issues a senior professional in this role would
788
+ * interrogate — traps the scenario hides, claims that must be defended.
789
+ * The driver probes these without revealing them verbatim; the agent
790
+ * must surface them on its own.
791
+ */
792
+ pressurePoints?: string[];
793
+ /**
794
+ * Curveballs the driver may inject once the agent is coasting — changed
795
+ * facts, a hostile counterparty position, a new constraint. Forces the
796
+ * agent to re-derive rather than recite.
797
+ */
798
+ curveballs?: string[];
799
+ }
800
+ interface DriverState {
801
+ tasks: number;
802
+ events: number;
803
+ proposals: {
804
+ pending: number;
805
+ approved: number;
806
+ rejected: number;
807
+ };
808
+ vaultFiles: string[];
809
+ codeBlocks: number;
810
+ generations: number;
811
+ }
812
+ interface JudgeInput {
813
+ scenario: Scenario;
814
+ turns: TurnResult[];
815
+ artifacts: CollectedArtifacts;
816
+ /** Shared ledger for paid built-in judges. Direct calls default to an uncapped ledger. */
817
+ costLedger?: CostLedgerHandle;
818
+ costPhase?: string;
819
+ costTags?: Record<string, string>;
820
+ signal?: AbortSignal;
821
+ }
822
+ type JudgeFn = (chat: ChatClient, input: JudgeInput) => Promise<JudgeScore[]>;
823
+ interface CheckResult {
824
+ name: string;
825
+ passed: boolean;
826
+ expected: string;
827
+ actual: string;
828
+ }
829
+ //#endregion
830
+ export { AssertCrossFamilyServedOptions as $, LlmThinkingMode as A, stripFencedJson as B, LlmCallError as C, LlmChargeBounds as D, LlmCallResult as E, LlmUsage as F, NoopRawProviderSink as G, FileSystemRawProviderSinkOptions as H, costReceiptFromLlm as I, RawProviderEvent as J, ProviderRedactor as K, costReceiptFromLlmError as L, LlmToolCall as M, LlmToolChoice as N, LlmMessage as O, LlmToolDefinition as P, providerFromBaseUrl as Q, isTransientLlmError as R, createChatClient as S, LlmCallRequest as T, InMemoryRawProviderSink as U, FileSystemRawProviderSink as V, InMemoryRawProviderSinkOptions as W, RawProviderSinkFilter as X, RawProviderSink as Y, defaultProviderRedactor as Z, CreateChatClientOpts as _, JudgeInput as a, ServedModelVerdict as at, OpenAiCompatibleTransportOpts as b, PersonaConfig as c, assertServedModels as ct, Scenario as d, AssertCrossFamilyOptions as dt, AssertServedModelOptions as et, ChatCallOpts as f, CrossFamilyError as ft, ChatTransport as g, ChatResponse as h, judgeFamily as ht, JudgeFn as i, ServedModelPolicy as it, LlmTokenLogprob as j, LlmResponseError as k, ProductClientConfig as l, checkServedModel as lt, ChatRequest as m, assertCrossFamily as mt, CompletionCriterion as n, ServedCrossFamilyError as nt, JudgeRubric as o, assertCrossFamilyServed as ot, ChatClient as p, JudgeFamily as pt, RawProviderDirection as q, DriverState as r, ServedModelCheck as rt, JudgeScore as s, assertServedModel as st, CheckResult as t, ModelSubstitutionError as tt, RouteMap as u, servedModelAcceptable as ut, CustomTransportOpts as v, LlmCallMetadata as w, SandboxSdkTransportOpts as x, MockTransportOpts as y, maximumChargeForLlmRequest as z };
831
+ //# sourceMappingURL=types-gvRsyJLh.d.ts.map