@tangle-network/agent-eval 0.136.0 → 0.137.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (211) hide show
  1. package/CHANGELOG.md +38 -1
  2. package/README.md +4 -2
  3. package/dist/{agent-profile-cell-OhuTee9n.js → agent-profile-cell-CbfBm2g6.js} +2 -2
  4. package/dist/{agent-profile-cell-OhuTee9n.js.map → agent-profile-cell-CbfBm2g6.js.map} +1 -1
  5. package/dist/{agent-profile-cell-CCm3l2v2.d.ts → agent-profile-cell-Cw0PVwDr.d.ts} +2 -2
  6. package/dist/{agent-profile-cell-CCm3l2v2.d.ts.map → agent-profile-cell-Cw0PVwDr.d.ts.map} +1 -1
  7. package/dist/analyst/index.d.ts +139 -17
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +606 -4
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{analyze-runs-jjCmF8pU.js → analyze-runs-BScZvqMV.js} +6 -6
  12. package/dist/{analyze-runs-jjCmF8pU.js.map → analyze-runs-BScZvqMV.js.map} +1 -1
  13. package/dist/{analyze-runs-Cda5Xkj1.d.ts → analyze-runs-PVtnfjvA.d.ts} +6 -6
  14. package/dist/{analyze-runs-Cda5Xkj1.d.ts.map → analyze-runs-PVtnfjvA.d.ts.map} +1 -1
  15. package/dist/{baseline-BUeFcgrn.js → baseline-C-GocmIW.js} +2 -2
  16. package/dist/{baseline-BUeFcgrn.js.map → baseline-C-GocmIW.js.map} +1 -1
  17. package/dist/benchmark-CHX4orG7.d.ts +184 -0
  18. package/dist/benchmark-CHX4orG7.d.ts.map +1 -0
  19. package/dist/benchmark-YDrpumqB.js +414 -0
  20. package/dist/benchmark-YDrpumqB.js.map +1 -0
  21. package/dist/benchmarks/index.d.ts +1 -1
  22. package/dist/benchmarks/index.js +1 -1
  23. package/dist/{benchmarks-Dfgm9ts5.js → benchmarks-DCLkQOmc.js} +3 -3
  24. package/dist/{benchmarks-Dfgm9ts5.js.map → benchmarks-DCLkQOmc.js.map} +1 -1
  25. package/dist/builder-eval/index.js +2 -2
  26. package/dist/campaign/index.d.ts +6 -6
  27. package/dist/campaign/index.js +3 -3
  28. package/dist/{campaign-Dz8uQnhC.js → campaign-lgObcHFC.js} +212 -78
  29. package/dist/campaign-lgObcHFC.js.map +1 -0
  30. package/dist/cli.js +1 -1
  31. package/dist/client-C8L6h6Wf.d.ts +202 -0
  32. package/dist/client-C8L6h6Wf.d.ts.map +1 -0
  33. package/dist/completion-verifier-DSyRNVzU.d.ts +240 -0
  34. package/dist/completion-verifier-DSyRNVzU.d.ts.map +1 -0
  35. package/dist/contract/index.d.ts +10 -9
  36. package/dist/contract/index.d.ts.map +1 -1
  37. package/dist/contract/index.js +10 -10
  38. package/dist/control.d.ts +2 -2
  39. package/dist/control.js +1 -1
  40. package/dist/{cost-ledger-DHAjwNj7.js → cost-ledger-D-5_-dhi.js} +2 -2
  41. package/dist/{cost-ledger-DHAjwNj7.js.map → cost-ledger-D-5_-dhi.js.map} +1 -1
  42. package/dist/{cost-ledger-fGS_u_O1.d.ts → cost-ledger-D2o6JOrL.d.ts} +2 -2
  43. package/dist/{cost-ledger-fGS_u_O1.d.ts.map → cost-ledger-D2o6JOrL.d.ts.map} +1 -1
  44. package/dist/{dataset-BvtnC8Dc.d.ts → dataset-v_Y5902-.d.ts} +2 -2
  45. package/dist/{dataset-BvtnC8Dc.d.ts.map → dataset-v_Y5902-.d.ts.map} +1 -1
  46. package/dist/{default-registry-CHmdy2An.js → default-registry-CLXbRt0f.js} +119 -38
  47. package/dist/default-registry-CLXbRt0f.js.map +1 -0
  48. package/dist/{default-registry-Brxr728w.d.ts → default-registry-Dc5D_Loc.d.ts} +63 -137
  49. package/dist/default-registry-Dc5D_Loc.d.ts.map +1 -0
  50. package/dist/{errors-8YnH8WlF.js → errors-D-LKuDhb.js} +8 -2
  51. package/dist/errors-D-LKuDhb.js.map +1 -0
  52. package/dist/{errors-CEk209JS.d.ts → errors-DkfjIDvD.d.ts} +9 -3
  53. package/dist/errors-DkfjIDvD.d.ts.map +1 -0
  54. package/dist/{eval-campaign-Cc8WZJ6b.js → eval-campaign-CHqfLnff.js} +6 -6
  55. package/dist/{eval-campaign-Cc8WZJ6b.js.map → eval-campaign-CHqfLnff.js.map} +1 -1
  56. package/dist/{extract-usage-DIQpN-ww.js → extract-usage-p-56bh8q.js} +3 -3
  57. package/dist/{extract-usage-DIQpN-ww.js.map → extract-usage-p-56bh8q.js.map} +1 -1
  58. package/dist/{feedback-trajectory-CVaeREXV.d.ts → feedback-trajectory-N_F0PwHz.d.ts} +90 -3
  59. package/dist/feedback-trajectory-N_F0PwHz.d.ts.map +1 -0
  60. package/dist/fuzz.d.ts +1 -1
  61. package/dist/fuzz.js +2 -2
  62. package/dist/hosted/index.d.ts +3 -2
  63. package/dist/hosted/index.d.ts.map +1 -1
  64. package/dist/index-BipJlj-C.d.ts +316 -0
  65. package/dist/index-BipJlj-C.d.ts.map +1 -0
  66. package/dist/{index-CQsJcqch.d.ts → index-BnP1QJUv.d.ts} +5 -5
  67. package/dist/{index-CQsJcqch.d.ts.map → index-BnP1QJUv.d.ts.map} +1 -1
  68. package/dist/{index-B4Fjfo5U.d.ts → index-C-Pr4OWg.d.ts} +88 -317
  69. package/dist/index-C-Pr4OWg.d.ts.map +1 -0
  70. package/dist/{index-DuhJaaiH.d.ts → index-DEb46kc6.d.ts} +2 -2
  71. package/dist/{index-DuhJaaiH.d.ts.map → index-DEb46kc6.d.ts.map} +1 -1
  72. package/dist/{index-C2fkZhv_.d.ts → index-DRNl6g_N.d.ts} +3 -3
  73. package/dist/{index-C2fkZhv_.d.ts.map → index-DRNl6g_N.d.ts.map} +1 -1
  74. package/dist/{index-AbhwHp0V.d.ts → index-U3RHOShi.d.ts} +2 -2
  75. package/dist/{index-AbhwHp0V.d.ts.map → index-U3RHOShi.d.ts.map} +1 -1
  76. package/dist/index.d.ts +29 -70
  77. package/dist/index.d.ts.map +1 -1
  78. package/dist/index.js +507 -33
  79. package/dist/index.js.map +1 -1
  80. package/dist/{client-DcvgkaZi.d.ts → insight-report-B9ooYH_g.d.ts} +5 -203
  81. package/dist/insight-report-B9ooYH_g.d.ts.map +1 -0
  82. package/dist/integrity-CCXTftiL.js +1360 -0
  83. package/dist/integrity-CCXTftiL.js.map +1 -0
  84. package/dist/{integrity-rmVhXWA7.d.ts → integrity-CKxosZ5Z.d.ts} +3 -3
  85. package/dist/{integrity-rmVhXWA7.d.ts.map → integrity-CKxosZ5Z.d.ts.map} +1 -1
  86. package/dist/{integrity-BzRbCHzi.js → integrity-fdt8XPAv.js} +2 -2
  87. package/dist/{integrity-BzRbCHzi.js.map → integrity-fdt8XPAv.js.map} +1 -1
  88. package/dist/ledger-core/index.d.ts +1 -1
  89. package/dist/ledger-core/index.js +1 -1
  90. package/dist/{ledger-core-DAKFKRzi.js → ledger-core-t6sItivm.js} +85 -85
  91. package/dist/{ledger-core-DAKFKRzi.js.map → ledger-core-t6sItivm.js.map} +1 -1
  92. package/dist/{llm-client-DHx8pzyJ.js → llm-client-DKB25jV8.js} +3 -3
  93. package/dist/{llm-client-DHx8pzyJ.js.map → llm-client-DKB25jV8.js.map} +1 -1
  94. package/dist/meta-eval/index.d.ts +2 -2
  95. package/dist/meta-eval/index.js +3 -3
  96. package/dist/{mint-DyRUc9k6.js → mint-Ctwk079K.js} +4 -4
  97. package/dist/{mint-DyRUc9k6.js.map → mint-Ctwk079K.js.map} +1 -1
  98. package/dist/multishot/index.d.ts +2 -2
  99. package/dist/openapi.json +1 -1
  100. package/dist/{paired-arms-BbFKrAU-.js → paired-arms-iZ08VFMN.js} +3 -3
  101. package/dist/{paired-arms-BbFKrAU-.js.map → paired-arms-iZ08VFMN.js.map} +1 -1
  102. package/dist/pipelines/index.js +2 -2
  103. package/dist/profile-cell.d.ts +1 -1
  104. package/dist/profile-cell.js +1 -1
  105. package/dist/{propose-review-control-SQ-n9-We.js → propose-review-control-DLXz4FCX.js} +2 -2
  106. package/dist/{propose-review-control-SQ-n9-We.js.map → propose-review-control-DLXz4FCX.js.map} +1 -1
  107. package/dist/registry-BdM7SuTr.d.ts +124 -0
  108. package/dist/registry-BdM7SuTr.d.ts.map +1 -0
  109. package/dist/{release-report-DooPguBc.js → release-report-B5XPBvAU.js} +4 -4
  110. package/dist/{release-report-DooPguBc.js.map → release-report-B5XPBvAU.js.map} +1 -1
  111. package/dist/{release-report-DpBxGGI1.d.ts → release-report-CofgVNZt.d.ts} +4 -4
  112. package/dist/{release-report-DpBxGGI1.d.ts.map → release-report-CofgVNZt.d.ts.map} +1 -1
  113. package/dist/{replay-C6wRg47C.js → replay-Bju0T8Ls.js} +248 -8
  114. package/dist/replay-Bju0T8Ls.js.map +1 -0
  115. package/dist/{replay-BRfMIs81.d.ts → replay-K8FaC0CB.d.ts} +227 -52
  116. package/dist/replay-K8FaC0CB.d.ts.map +1 -0
  117. package/dist/reporting.d.ts +4 -4
  118. package/dist/reporting.js +4 -4
  119. package/dist/{researcher-Doo95b50.d.ts → researcher-Da0Wj-bt.d.ts} +6 -7
  120. package/dist/researcher-Da0Wj-bt.d.ts.map +1 -0
  121. package/dist/{reward-hacking-D-QqXvg-.d.ts → reward-hacking-CQ3hTCO3.d.ts} +2 -2
  122. package/dist/{reward-hacking-D-QqXvg-.d.ts.map → reward-hacking-CQ3hTCO3.d.ts.map} +1 -1
  123. package/dist/{reward-hacking-a-kYs0-i.js → reward-hacking-GyN0kMd8.js} +3 -3
  124. package/dist/{reward-hacking-a-kYs0-i.js.map → reward-hacking-GyN0kMd8.js.map} +1 -1
  125. package/dist/rl.d.ts +6 -6
  126. package/dist/rl.js +9 -9
  127. package/dist/rollout/index.d.ts +1 -1
  128. package/dist/rollout/index.js +3 -3
  129. package/dist/{rollout-DLSUIWLu.js → rollout-DQFl0UXA.js} +2 -2
  130. package/dist/{rollout-DLSUIWLu.js.map → rollout-DQFl0UXA.js.map} +1 -1
  131. package/dist/{rubric-predictive-validity-BJf-8ejY.js → rubric-predictive-validity-BRR632r1.js} +2 -2
  132. package/dist/{rubric-predictive-validity-BJf-8ejY.js.map → rubric-predictive-validity-BRR632r1.js.map} +1 -1
  133. package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts → rubric-predictive-validity-C4sztLR3.d.ts} +2 -2
  134. package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts.map → rubric-predictive-validity-C4sztLR3.d.ts.map} +1 -1
  135. package/dist/{run-evidence-DokQtX0-.d.ts → run-evidence-BDIircdA.d.ts} +3 -3
  136. package/dist/{run-evidence-DokQtX0-.d.ts.map → run-evidence-BDIircdA.d.ts.map} +1 -1
  137. package/dist/{run-record-DcObtIGh.d.ts → run-record-BPCa2rQ8.d.ts} +4 -4
  138. package/dist/{run-record-DcObtIGh.d.ts.map → run-record-BPCa2rQ8.d.ts.map} +1 -1
  139. package/dist/{run-record-BIwU2wdV.js → run-record-vRgqWmJw.js} +3 -3
  140. package/dist/{run-record-BIwU2wdV.js.map → run-record-vRgqWmJw.js.map} +1 -1
  141. package/dist/{semantic-concept-judge-Btozx3Vc.js → semantic-concept-judge-Bz64IckK.js} +4 -4
  142. package/dist/{semantic-concept-judge-Btozx3Vc.js.map → semantic-concept-judge-Bz64IckK.js.map} +1 -1
  143. package/dist/{server-Bz3WQJs6.js → server-KjXZZUDX.js} +3 -3
  144. package/dist/{server-Bz3WQJs6.js.map → server-KjXZZUDX.js.map} +1 -1
  145. package/dist/{skill-usage-BDQVPIG1.d.ts → skill-usage-CFDLLlhF.d.ts} +25 -47
  146. package/dist/skill-usage-CFDLLlhF.d.ts.map +1 -0
  147. package/dist/{skillopt-optimization-method-CwSYkv35.d.ts → skillopt-optimization-method-BpbnlvAZ.d.ts} +11 -12
  148. package/dist/skillopt-optimization-method-BpbnlvAZ.d.ts.map +1 -0
  149. package/dist/{skillopt-optimization-method-0UmPD6aP.js → skillopt-optimization-method-f4o9sUT4.js} +7 -7
  150. package/dist/{skillopt-optimization-method-0UmPD6aP.js.map → skillopt-optimization-method-f4o9sUT4.js.map} +1 -1
  151. package/dist/{statistics-CnGCLLqc.js → statistics-ByxzSiOM.js} +2 -2
  152. package/dist/{statistics-CnGCLLqc.js.map → statistics-ByxzSiOM.js.map} +1 -1
  153. package/dist/{statistics-CKOqre5S.d.ts → statistics-_7P642CN.d.ts} +2 -2
  154. package/dist/{statistics-CKOqre5S.d.ts.map → statistics-_7P642CN.d.ts.map} +1 -1
  155. package/dist/{summary-report-BEk8OFLs.js → summary-report-9A5y7EsK.js} +4 -4
  156. package/dist/{summary-report-BEk8OFLs.js.map → summary-report-9A5y7EsK.js.map} +1 -1
  157. package/dist/{summary-report-CPMINBqs.d.ts → summary-report-DHipz9Kx.d.ts} +3 -3
  158. package/dist/{summary-report-CPMINBqs.d.ts.map → summary-report-DHipz9Kx.d.ts.map} +1 -1
  159. package/dist/supervisor-run/index.d.ts +3 -2
  160. package/dist/supervisor-run/index.js +3 -2
  161. package/dist/{supervisor-run-Dr5HnTup.js → supervisor-run-B2EWUmQY.js} +28 -454
  162. package/dist/supervisor-run-B2EWUmQY.js.map +1 -0
  163. package/dist/{test-graded-scenario-BsqWLmPt.js → test-graded-scenario-JHcKQNpq.js} +2 -2
  164. package/dist/{test-graded-scenario-BsqWLmPt.js.map → test-graded-scenario-JHcKQNpq.js.map} +1 -1
  165. package/dist/tools-DZk2Jn64.js +1876 -0
  166. package/dist/tools-DZk2Jn64.js.map +1 -0
  167. package/dist/traces.d.ts +6 -7
  168. package/dist/traces.js +5 -6
  169. package/dist/{types-DVjczBM9.d.ts → types-CKswbJGO.d.ts} +260 -6
  170. package/dist/types-CKswbJGO.d.ts.map +1 -0
  171. package/dist/{types-DiWLru6Z.d.ts → types-CTGbIm57.d.ts} +5 -5
  172. package/dist/{types-DiWLru6Z.d.ts.map → types-CTGbIm57.d.ts.map} +1 -1
  173. package/dist/types-CTvKfr5F.d.ts +804 -0
  174. package/dist/types-CTvKfr5F.d.ts.map +1 -0
  175. package/dist/{index-CyC1BTmn.d.ts → types-Dea6tiVI.d.ts} +16 -238
  176. package/dist/types-Dea6tiVI.d.ts.map +1 -0
  177. package/dist/wire/index.d.ts +3 -3
  178. package/dist/wire/index.js +1 -1
  179. package/docs/feedback-trajectories.md +100 -1
  180. package/docs/trace-analysis.md +374 -58
  181. package/package.json +5 -1
  182. package/dist/analyst-BkTS3C58.d.ts +0 -89
  183. package/dist/analyst-BkTS3C58.d.ts.map +0 -1
  184. package/dist/analyst-j5je5J7c.js +0 -152
  185. package/dist/analyst-j5je5J7c.js.map +0 -1
  186. package/dist/campaign-Dz8uQnhC.js.map +0 -1
  187. package/dist/client-DcvgkaZi.d.ts.map +0 -1
  188. package/dist/default-registry-Brxr728w.d.ts.map +0 -1
  189. package/dist/default-registry-CHmdy2An.js.map +0 -1
  190. package/dist/errors-8YnH8WlF.js.map +0 -1
  191. package/dist/errors-CEk209JS.d.ts.map +0 -1
  192. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +0 -1
  193. package/dist/index-B4Fjfo5U.d.ts.map +0 -1
  194. package/dist/index-CyC1BTmn.d.ts.map +0 -1
  195. package/dist/llm-client-BiK4HW0u.d.ts +0 -290
  196. package/dist/llm-client-BiK4HW0u.d.ts.map +0 -1
  197. package/dist/raw-provider-sink-BU29Sh8h.d.ts +0 -134
  198. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +0 -1
  199. package/dist/replay-BRfMIs81.d.ts.map +0 -1
  200. package/dist/replay-C6wRg47C.js.map +0 -1
  201. package/dist/researcher-Doo95b50.d.ts.map +0 -1
  202. package/dist/skill-usage-BDQVPIG1.d.ts.map +0 -1
  203. package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +0 -1
  204. package/dist/store-CxJry_cs.d.ts +0 -229
  205. package/dist/store-CxJry_cs.d.ts.map +0 -1
  206. package/dist/supervisor-run-Dr5HnTup.js.map +0 -1
  207. package/dist/tools-D8yTtNSN.js +0 -1190
  208. package/dist/tools-D8yTtNSN.js.map +0 -1
  209. package/dist/types-Cc3qbqzj.d.ts +0 -387
  210. package/dist/types-Cc3qbqzj.d.ts.map +0 -1
  211. package/dist/types-DVjczBM9.d.ts.map +0 -1
@@ -0,0 +1,804 @@
1
+ import { r as CaptureIntegrityError, t as AgentEvalError } from "./errors-DkfjIDvD.js";
2
+ import { b as CustomTokenPricing, c as CostLedgerHandle, f as CostLedgerSummary, g as CostReceiptInput, x as MaximumCharge } from "./cost-ledger-D2o6JOrL.js";
3
+ //#region src/trace/raw-provider-sink.d.ts
4
+ /**
5
+ * RawProviderSink — first-class persistence for the actual HTTP-level
6
+ * request/response bodies of every LLM provider call.
7
+ *
8
+ * Why this is a separate sink from the structured `LlmSpan`:
9
+ *
10
+ * - `LlmSpan` records the *intent* — model name, messages, output text,
11
+ * usage. It's what dashboards read; it's NOT enough for forensics.
12
+ * - When a downstream consumer reports "the verifier used the wrong route"
13
+ * or "tokens look right but reasoning was missing," the only way to
14
+ * answer is the raw HTTP body. Span fields can lie (a proxy can echo
15
+ * a different `model` value than what actually answered); the raw
16
+ * response is ground truth.
17
+ *
18
+ * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
19
+ * matrix runner / BuilderSession sets it up automatically) and every
20
+ * request, response, and error is recorded — including retries, with the
21
+ * attempt index attached so a flaky call's full event chain is recoverable.
22
+ *
23
+ * Redaction is enforced at sink time. The default redactor strips
24
+ * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
25
+ * payload field whose key matches `apiKey | api_key | bearer | password |
26
+ * secret | token` (case-insensitive). Override via the sink constructor or
27
+ * the per-call `redactor`. The `redactedFields` array on the persisted
28
+ * event lets a reviewer see what was stripped without exposing the values.
29
+ */
30
+ type RawProviderDirection = 'request' | 'response' | 'error';
31
+ interface RawProviderEvent {
32
+ /** Stable id. Generated by the sink if omitted. */
33
+ eventId: string;
34
+ /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
35
+ runId?: string;
36
+ spanId?: string;
37
+ /**
38
+ * Logical provider name. Free-form so callers can use whatever id matches
39
+ * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
40
+ * omitted, derived from `baseUrl` in `LlmClientOptions`.
41
+ */
42
+ provider: string;
43
+ model: string;
44
+ /** Endpoint path, e.g. `'/v1/chat/completions'`. */
45
+ endpoint: string;
46
+ /** Base URL used for the call (already-normalised — no trailing slash). */
47
+ baseUrl: string;
48
+ /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
49
+ attemptIndex: number;
50
+ direction: RawProviderDirection;
51
+ /** Unix ms. */
52
+ timestamp: number;
53
+ /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
54
+ durationMs?: number;
55
+ statusCode?: number;
56
+ requestHeaders?: Record<string, string>;
57
+ requestBody?: unknown;
58
+ responseHeaders?: Record<string, string>;
59
+ responseBody?: unknown;
60
+ /** Set on `direction: 'error'` events. */
61
+ errorMessage?: string;
62
+ /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
63
+ redactedFields: string[];
64
+ }
65
+ interface RawProviderSinkFilter {
66
+ runId?: string;
67
+ spanId?: string;
68
+ direction?: RawProviderDirection;
69
+ attemptIndex?: number;
70
+ }
71
+ interface RawProviderSink {
72
+ record(event: RawProviderEvent): Promise<void>;
73
+ /** Optional listing — implementations that durably persist (file, db) should support this. */
74
+ list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
75
+ /** Optional teardown for backed implementations. */
76
+ close?(): Promise<void>;
77
+ }
78
+ type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
79
+ /**
80
+ * Default redactor — strips well-known auth headers and any body field whose
81
+ * key matches the credential pattern. Records every redacted path on
82
+ * `event.redactedFields` so a downstream reviewer can see what was removed.
83
+ */
84
+ declare function defaultProviderRedactor(event: RawProviderEvent): RawProviderEvent;
85
+ interface InMemoryRawProviderSinkOptions {
86
+ redactor?: ProviderRedactor;
87
+ }
88
+ declare class InMemoryRawProviderSink implements RawProviderSink {
89
+ private events;
90
+ private redactor;
91
+ constructor(opts?: InMemoryRawProviderSinkOptions);
92
+ record(event: RawProviderEvent): Promise<void>;
93
+ list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
94
+ size(): number;
95
+ }
96
+ declare class NoopRawProviderSink implements RawProviderSink {
97
+ record(): Promise<void>;
98
+ /**
99
+ * Returns an empty array. Implemented so `assertRunCaptured` does not
100
+ * trip the `no_raw_sink` issue when a caller explicitly opts out of
101
+ * capture by passing this sink — opt-out is a deliberate choice, not a
102
+ * misconfiguration.
103
+ */
104
+ list(): Promise<RawProviderEvent[]>;
105
+ }
106
+ interface FileSystemRawProviderSinkOptions {
107
+ /** Directory the NDJSON file is written into. Created if missing. */
108
+ dir: string;
109
+ /** File name; default `'raw-provider-events.ndjson'`. */
110
+ fileName?: string;
111
+ /** Bytes after which the writer rolls over to a new file (default 32 MiB). */
112
+ rollAtBytes?: number;
113
+ redactor?: ProviderRedactor;
114
+ }
115
+ declare class FileSystemRawProviderSink implements RawProviderSink {
116
+ private dir;
117
+ private fileName;
118
+ private rollAtBytes;
119
+ private redactor;
120
+ private bytesWritten;
121
+ private rollIndex;
122
+ private initPromise;
123
+ constructor(opts: FileSystemRawProviderSinkOptions);
124
+ private ensureInit;
125
+ private currentPath;
126
+ record(event: RawProviderEvent): Promise<void>;
127
+ list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
128
+ }
129
+ /**
130
+ * Best-effort provider id from a base URL. Falls back to the URL host when
131
+ * none of the well-known patterns match.
132
+ */
133
+ declare function providerFromBaseUrl(baseUrl: string): string;
134
+ //#endregion
135
+ //#region src/llm-client.d.ts
136
+ interface LlmMessage {
137
+ role: 'system' | 'user' | 'assistant';
138
+ /**
139
+ * Either a plain text content string OR a multimodal content array
140
+ * (text + image_url parts) for vision-capable models.
141
+ */
142
+ content: string | Array<{
143
+ type: 'text';
144
+ text: string;
145
+ } | {
146
+ type: 'image_url';
147
+ image_url: {
148
+ url: string;
149
+ detail?: 'auto' | 'low' | 'high';
150
+ };
151
+ }>;
152
+ }
153
+ type LlmThinkingMode = 'enabled' | 'disabled';
154
+ interface LlmCallRequest {
155
+ model: string;
156
+ messages: LlmMessage[];
157
+ /** Optional JSON-mode response format (response_format: json_object). */
158
+ jsonMode?: boolean;
159
+ /** Optional structured output via JSON Schema. Falls back to json_object on 400. */
160
+ jsonSchema?: {
161
+ name: string;
162
+ schema: Record<string, unknown>;
163
+ };
164
+ temperature?: number;
165
+ maxTokens?: number;
166
+ /** OpenAI-compatible reasoning mode. Omitted when the provider default should apply. */
167
+ thinking?: LlmThinkingMode;
168
+ /** Per-call timeout, default 300s. */
169
+ timeoutMs?: number;
170
+ }
171
+ /** Conservative priced bound for the exact text request sent to a provider.
172
+ * Returns undefined when output or multimodal input is not bounded, causing a
173
+ * capped CostLedger to reject the call before execution. Pass
174
+ * `customTokenPricing` when package pricing does not cover the model or endpoint. */
175
+ declare function maximumChargeForLlmRequest(request: Pick<LlmCallRequest, 'model' | 'messages' | 'jsonSchema' | 'maxTokens' | 'thinking'>, options?: LlmClientOptions): MaximumCharge | undefined;
176
+ interface LlmUsage {
177
+ promptTokens: number;
178
+ completionTokens: number;
179
+ totalTokens: number;
180
+ /** False when the provider omitted or malformed prompt/completion usage. */
181
+ captured?: boolean;
182
+ /** Reasoning-token subset of completionTokens, when reported. */
183
+ reasoningTokens?: number;
184
+ /** Proxies populate this when prompt caching is on. */
185
+ cachedPromptTokens?: number;
186
+ }
187
+ interface LlmCallResult {
188
+ /** The text content of the first choice. Empty string if none. */
189
+ content: string;
190
+ usage: LlmUsage;
191
+ /**
192
+ * Cost in USD. Uses the provider's reported cost when present, otherwise
193
+ * caller-supplied token pricing. `null` when neither is available.
194
+ */
195
+ costUsd: number | null;
196
+ /** Model name actually used (echoed from response). */
197
+ model: string;
198
+ /** Wall-clock duration of the HTTP call (last attempt, if retried). */
199
+ durationMs: number;
200
+ /**
201
+ * `finish_reason` echoed from the first choice (`stop`, `length`,
202
+ * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
203
+ * Exposed so a free-form `callLlm` caller CAN detect a truncated answer
204
+ * (`length`) instead of treating a cut-off completion as complete. Note:
205
+ * `callLlm` does not itself reject on it — acting on this signal is the
206
+ * caller's responsibility (in-repo free-form drivers do not yet enforce it).
207
+ */
208
+ finishReason?: string | null;
209
+ /**
210
+ * True when `content.trim()` is empty. An empty completion is a silent zero
211
+ * for free-form `callLlm` callers; this flag is the signal a caller can
212
+ * inspect to fail loud rather than proceed on an empty string. `callLlm`
213
+ * surfaces it but does not throw on it.
214
+ */
215
+ contentEmpty?: boolean;
216
+ /** Raw response body. */
217
+ raw: Record<string, unknown>;
218
+ }
219
+ type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
220
+ /** Convert a provider result into the canonical paid-call receipt input. */
221
+ declare function costReceiptFromLlm(result: LlmCallResult, customTokenPricing?: CustomTokenPricing): CostReceiptInput;
222
+ /** Structured-response failures retain their completed provider receipt. */
223
+ declare function costReceiptFromLlmError(error: Error, customTokenPricing?: CustomTokenPricing): CostReceiptInput | undefined;
224
+ declare class LlmCallError extends AgentEvalError {
225
+ readonly status: number;
226
+ readonly body: string;
227
+ readonly model: string;
228
+ constructor(message: string, status: number, body: string, model: string);
229
+ }
230
+ /** A provider response completed and incurred measurable usage, but its content
231
+ * could not satisfy the caller's response contract. The response envelope is
232
+ * retained so accounting can commit the receipt before the error propagates. */
233
+ declare class LlmResponseError extends AgentEvalError {
234
+ readonly result: LlmCallResult;
235
+ constructor(message: string, result: LlmCallResult, options?: {
236
+ cause?: unknown;
237
+ });
238
+ }
239
+ interface LlmClientOptions {
240
+ /** Base URL (without trailing slash). Must end at the `/v1` prefix. */
241
+ baseUrl?: string;
242
+ /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
243
+ apiKey?: string;
244
+ bearer?: string;
245
+ /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
246
+ authHeader?: {
247
+ name: string;
248
+ value: string;
249
+ };
250
+ /** Stable provider idempotency key, reused across retries of this logical call. */
251
+ idempotencyKey?: string;
252
+ /** Default timeout in ms. Per-call can override. */
253
+ defaultTimeoutMs?: number;
254
+ /**
255
+ * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
256
+ * each attempt's per-attempt timeout controller, so aborting it cancels
257
+ * the in-flight fetch. A caller abort is FATAL: it is not retried even
258
+ * though an AbortError otherwise matches the transient patterns.
259
+ */
260
+ signal?: AbortSignal;
261
+ /**
262
+ * Cross-attempt wall-clock budget in ms, measured from the first attempt.
263
+ * Before launching each attempt the loop checks the remaining budget and
264
+ * stops retrying once it is exhausted, rather than waiting the full
265
+ * per-attempt timeout on every retry. Bounds total time independent of
266
+ * total attempts × `timeoutMs`.
267
+ */
268
+ deadlineMs?: number;
269
+ /** Total provider attempts. Default 3. */
270
+ maximumAttempts?: number;
271
+ /** Token rates used when the provider omits cost or package pricing does not cover the model. */
272
+ customTokenPricing?: CustomTokenPricing;
273
+ /**
274
+ * Transport for requests that declare `jsonSchema`. `native` sends
275
+ * `response_format: json_schema`; `json-object` sends the broadly supported
276
+ * JSON mode and relies on the caller to include the schema in model-visible
277
+ * instructions. Default: `native`.
278
+ */
279
+ jsonSchemaTransport?: 'native' | 'json-object';
280
+ /**
281
+ * JSON payload parsing policy. `extract` accepts fenced or prose-prefixed JSON.
282
+ * `exact` requires the complete response content to be one JSON value.
283
+ * Default: `extract`.
284
+ */
285
+ jsonPayloadMode?: 'extract' | 'exact';
286
+ /** Default provider reasoning mode. A per-call request value takes precedence. */
287
+ thinking?: LlmThinkingMode;
288
+ /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
289
+ fetch?: typeof fetch;
290
+ /**
291
+ * Optional raw HTTP capture sink. When provided, every request, response,
292
+ * and error (across all retry attempts) is recorded to the sink, with auth
293
+ * headers and credential-shaped body fields redacted by default. This is
294
+ * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
295
+ * raw events record what actually crossed the wire.
296
+ */
297
+ rawSink?: RawProviderSink;
298
+ /**
299
+ * Logical provider id attached to raw events. When omitted, derived from
300
+ * `baseUrl` via `providerFromBaseUrl`.
301
+ */
302
+ provider?: string;
303
+ /** Trace context attached to raw events; populated by emitter-aware callers. */
304
+ traceContext?: {
305
+ runId?: string;
306
+ spanId?: string;
307
+ };
308
+ /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
309
+ redactor?: ProviderRedactor;
310
+ }
311
+ /**
312
+ * True when an error is a transient transport/network fault worth retrying,
313
+ * as opposed to a deterministic failure (4xx schema reject, JSON parse) that
314
+ * a retry cannot fix. Inspects `LlmCallError.status`, then the error's
315
+ * name/message/code, then recurses into `error.cause` — undici nests the
316
+ * real socket fault one or more levels under `.cause`.
317
+ *
318
+ * This is the retry classifier for the package: `callLlm` and
319
+ * `withJudgeRetry` both route through it, so connection failures are treated
320
+ * consistently across transports.
321
+ */
322
+ declare function isTransientLlmError(err: unknown): boolean;
323
+ /** Exponential backoff: 500ms, 1s, 2s, 4s, ... capped at 16s. Attempt is 0-indexed. */
324
+ declare function backoffMs(attempt: number): number;
325
+ /**
326
+ * Strip a ```json / ``` code fence if the model emitted one.
327
+ * Idempotent for naked JSON. Some models (claude-code via router, certain
328
+ * deepseek models) wrap output even under json_object.
329
+ */
330
+ declare function stripFencedJson(raw: string): string;
331
+ /**
332
+ * Low-level call. Returns raw content + usage + cost. Retries on transient
333
+ * failures; does NOT degrade schema here — callers that want graceful
334
+ * degrade use `callLlmJson`.
335
+ */
336
+ declare function callLlm(req: LlmCallRequest, opts?: LlmClientOptions): Promise<LlmCallResult>;
337
+ /**
338
+ * Structured-output call. Returns parsed JSON plus the raw result envelope.
339
+ * Degrades `jsonSchema` → `jsonMode` on a 400 that names the schema param —
340
+ * critical for deepseek-v3/v4, kimi-k2.6, and other models that don't accept
341
+ * the `response_format.json_schema` shape but DO accept `json_object`.
342
+ */
343
+ declare function callLlmJson<T = unknown>(req: LlmCallRequest, opts?: LlmClientOptions): Promise<{
344
+ value: T;
345
+ result: LlmCallResult;
346
+ }>;
347
+ type LlmRouteAssertionReason = 'no_explicit_base_url' | 'base_url_blocked' | 'base_url_not_allowed' | 'no_auth' | 'wrong_provider';
348
+ declare class LlmRouteAssertionError extends CaptureIntegrityError {
349
+ readonly reason: LlmRouteAssertionReason;
350
+ readonly baseUrl: string;
351
+ constructor(message: string, reason: LlmRouteAssertionReason, baseUrl: string);
352
+ }
353
+ interface LlmRouteRequirements {
354
+ /**
355
+ * Throw if `opts.baseUrl` is undefined, i.e. the call would fall back to
356
+ * `DEFAULT_BASE_URL`. Set this for evaluation runs where silently using
357
+ * the public/free-tier router is a defect — the launch reviewer needs to
358
+ * know exactly which provider answered.
359
+ */
360
+ requireExplicitBaseUrl?: boolean;
361
+ /**
362
+ * Allowlist of acceptable base URLs. Strings match by prefix
363
+ * (case-insensitive); RegExps test against the full base URL.
364
+ */
365
+ allowedBaseUrls?: Array<string | RegExp>;
366
+ /** Blocklist that takes precedence over `allowedBaseUrls`. */
367
+ blockedBaseUrls?: Array<string | RegExp>;
368
+ /** Throw if no auth header / api key is configured. */
369
+ requireAuth?: boolean;
370
+ /**
371
+ * Logical provider id the configured `baseUrl` is expected to match (via
372
+ * `providerFromBaseUrl`). Mainly useful when paired with `requireExplicitBaseUrl`.
373
+ */
374
+ expectedProvider?: string;
375
+ }
376
+ /**
377
+ * Fail-loud assertion that the configured LLM client points at the route
378
+ * the caller intends. Designed for the matrix-runner preflight: invoke
379
+ * once before any LLM call to catch misconfiguration before a sweep burns
380
+ * dollars on the wrong provider.
381
+ *
382
+ * Throws `LlmRouteAssertionError`. Pure — no I/O — so it's safe to call
383
+ * from constructors and CI gates.
384
+ */
385
+ declare function assertLlmRoute(opts: LlmClientOptions, req?: LlmRouteRequirements): void;
386
+ /**
387
+ * Probe whether a model is reachable. Returns latency + null error on
388
+ * success; `ok=false` + error message on any failure (HTTP, timeout,
389
+ * network, parse). Designed for sweep preflights — fail loud at the
390
+ * boundary before burning a 30-leaf run on a misconfigured router.
391
+ *
392
+ * Sends a tiny `ping` message with `maxTokens=64`. Reasoning models
393
+ * (glm-5.1, deepseek-v4) can burn the entire budget on internal reasoning
394
+ * for short prompts, so don't tighten this further. We don't validate
395
+ * content; HTTP 200 means reachable.
396
+ */
397
+ declare function probeLlm(model: string, opts?: LlmClientOptions & {
398
+ timeoutMs?: number;
399
+ }): Promise<{
400
+ ok: boolean;
401
+ latencyMs: number;
402
+ error: string | null;
403
+ }>;
404
+ /**
405
+ * Stateful client — construct once with defaults, call many times.
406
+ * Thin wrapper around the free functions; exists for callers that want
407
+ * to inject a single configured instance into multiple primitives.
408
+ */
409
+ declare class LlmClient {
410
+ readonly maximumAttempts: number;
411
+ private readonly opts;
412
+ constructor(opts?: LlmClientOptions);
413
+ call(req: LlmCallRequest, per?: LlmClientOptions): Promise<LlmCallResult>;
414
+ callJson<T = unknown>(req: LlmCallRequest, per?: LlmClientOptions): Promise<{
415
+ value: T;
416
+ result: LlmCallResult;
417
+ }>;
418
+ }
419
+ //#endregion
420
+ //#region src/analyst/chat-client.d.ts
421
+ /**
422
+ * Unified chat interface using the package's canonical LLM request and result.
423
+ */
424
+ interface ChatClient {
425
+ /** Display name of the bound transport, included in telemetry. */
426
+ readonly transport: ChatTransport;
427
+ /** Default model when the caller omits one. */
428
+ readonly defaultModel?: string;
429
+ /** Total provider attempts this transport can make for one chat call. */
430
+ readonly maximumAttempts?: number;
431
+ /** Implementations must enforce `req.maxTokens` when it is present. */
432
+ chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
433
+ }
434
+ type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'custom' | 'mock';
435
+ interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
436
+ /** Optional — falls back to ChatClient.defaultModel. */
437
+ model?: string;
438
+ }
439
+ type ChatResponse = LlmCallResult;
440
+ interface ChatCallOpts {
441
+ /** Cancel the in-flight request. */
442
+ signal?: AbortSignal;
443
+ /** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
444
+ maxCostUsd?: number;
445
+ /** Correlation tag carried into request headers when the transport allows. */
446
+ correlationId?: string;
447
+ /** Stable provider idempotency key for retries/redrives of one paid call. */
448
+ idempotencyKey?: string;
449
+ }
450
+ type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | CustomTransportOpts | MockTransportOpts;
451
+ interface BaseTransportOpts {
452
+ defaultModel?: string;
453
+ /** Total provider attempts. Required for opaque transports used in capped runs. */
454
+ maximumAttempts?: number;
455
+ }
456
+ interface RouterTransportOpts extends BaseTransportOpts {
457
+ transport: 'router';
458
+ baseUrl?: string;
459
+ apiKey: string;
460
+ }
461
+ interface CliBridgeTransportOpts extends BaseTransportOpts {
462
+ transport: 'cli-bridge';
463
+ baseUrl?: string;
464
+ bearer?: string;
465
+ }
466
+ interface DirectProviderTransportOpts extends BaseTransportOpts {
467
+ transport: 'direct-provider';
468
+ baseUrl: string;
469
+ apiKey: string;
470
+ }
471
+ /**
472
+ * Sandbox-SDK transport. The caller supplies a canonical chat function for an
473
+ * already-configured Sandbox handle, so agent-eval does not import the SDK.
474
+ */
475
+ interface SandboxSdkTransportOpts extends BaseTransportOpts {
476
+ transport: 'sandbox-sdk';
477
+ chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
478
+ }
479
+ /** Caller-adapted SDK or transport returning the canonical ChatResponse shape. */
480
+ interface CustomTransportOpts extends BaseTransportOpts {
481
+ transport: 'custom';
482
+ chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
483
+ }
484
+ /**
485
+ * Mock transport for tests. The handler receives the request and returns
486
+ * whatever the test wants. No retries, no JSON-schema degrade.
487
+ */
488
+ interface MockTransportOpts extends BaseTransportOpts {
489
+ transport: 'mock';
490
+ handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
491
+ }
492
+ /**
493
+ * Build a ChatClient bound to a specific transport. The returned client
494
+ * is safe to share across analysts in a single registry run.
495
+ */
496
+ declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
497
+ //#endregion
498
+ //#region src/types.d.ts
499
+ interface Scenario {
500
+ id: string;
501
+ persona: string;
502
+ label: string;
503
+ thesis: string;
504
+ dimensions: string[];
505
+ turns: Turn[];
506
+ artifactChecks: ArtifactCheck[];
507
+ systemPromptAppend?: string;
508
+ }
509
+ interface Turn {
510
+ user: string;
511
+ expectedBehaviors: string[];
512
+ adversarial?: boolean;
513
+ feedbackType?: 'correction' | 'rejection' | 'vague' | 'contradictory' | 'escalation';
514
+ }
515
+ interface ArtifactCheck {
516
+ type: 'vault_file_exists' | 'vault_file_contains' | 'block_extracted' | 'code_valid' | 'generation_produced' | 'tool_created' | string;
517
+ target: string;
518
+ contains?: string;
519
+ minCount?: number;
520
+ description: string;
521
+ }
522
+ interface JudgeConfig {
523
+ model: string;
524
+ temperature: number;
525
+ rubric: JudgeRubric;
526
+ }
527
+ interface JudgeRubric {
528
+ name: string;
529
+ description: string;
530
+ dimensions: RubricDimension[];
531
+ }
532
+ interface RubricDimension {
533
+ name: string;
534
+ description: string;
535
+ anchor_low: string;
536
+ anchor_high: string;
537
+ weight: number;
538
+ }
539
+ interface ScenarioResult {
540
+ scenarioId: string;
541
+ persona: string;
542
+ turns: TurnResult[];
543
+ artifactResults: ArtifactResult[];
544
+ judgeScores: JudgeScore[];
545
+ judgeErrors: number;
546
+ overallScore: number;
547
+ totalDurationMs: number;
548
+ artifacts: CollectedArtifacts;
549
+ /** Agent and judge spend attributed to this scenario. */
550
+ cost?: CostLedgerSummary;
551
+ }
552
+ interface TurnResult {
553
+ turnIndex: number;
554
+ userMessage: string;
555
+ agentResponse: string;
556
+ durationMs: number;
557
+ blocksExtracted: {
558
+ type: string;
559
+ title: string;
560
+ }[];
561
+ containsCode: boolean;
562
+ containsToolCall: boolean;
563
+ }
564
+ interface ArtifactResult {
565
+ check: ArtifactCheck;
566
+ passed: boolean;
567
+ detail?: string;
568
+ }
569
+ interface JudgeScore {
570
+ judgeName: string;
571
+ dimension: string;
572
+ score: number;
573
+ reasoning: string;
574
+ evidence?: string;
575
+ }
576
+ interface CollectedArtifacts {
577
+ vaultFiles: {
578
+ path: string;
579
+ content: string;
580
+ }[];
581
+ blocksExtracted: {
582
+ type: string;
583
+ fields: Record<string, string>;
584
+ }[];
585
+ codeBlocks: {
586
+ language: string;
587
+ code: string;
588
+ }[];
589
+ toolCalls: string[];
590
+ }
591
+ interface BenchmarkReport {
592
+ timestamp: string;
593
+ generation: number;
594
+ promptVersion: string;
595
+ scenarioCount: number;
596
+ results: ScenarioResult[];
597
+ cost?: CostLedgerSummary;
598
+ summary: {
599
+ overallAvg: number;
600
+ byPersona: Record<string, {
601
+ avg: number;
602
+ passed: number;
603
+ total: number;
604
+ }>;
605
+ byDimension: Record<string, {
606
+ avg: number;
607
+ scores: number[];
608
+ }>;
609
+ weakest: {
610
+ scenario: string;
611
+ score: number;
612
+ reason: string;
613
+ }[];
614
+ strongest: {
615
+ scenario: string;
616
+ score: number;
617
+ reason: string;
618
+ }[];
619
+ };
620
+ }
621
+ interface RouteMap {
622
+ signup?: string;
623
+ login?: string;
624
+ workspaces?: string;
625
+ threads?: string;
626
+ chat?: string;
627
+ tasks?: string;
628
+ events?: string;
629
+ approvals?: string;
630
+ vault?: string;
631
+ generations?: string;
632
+ [key: string]: string | undefined;
633
+ }
634
+ interface ProductClientConfig {
635
+ baseUrl: string;
636
+ routes: RouteMap;
637
+ /** Per-request timeout in ms before the request is aborted. Default 30s. */
638
+ timeoutMs?: number;
639
+ }
640
+ interface ScenarioFile {
641
+ id: string;
642
+ category: string;
643
+ persona: string;
644
+ label: string;
645
+ thesis: string;
646
+ isControl?: boolean;
647
+ rubric?: {
648
+ dimensions: {
649
+ name: string;
650
+ description: string;
651
+ weight: number;
652
+ }[];
653
+ };
654
+ turns: Turn[];
655
+ artifactChecks: ArtifactCheck[];
656
+ }
657
+ interface CompletionCriterion {
658
+ name: string;
659
+ check: (state: DriverState) => boolean;
660
+ progress?: (state: DriverState) => number;
661
+ }
662
+ interface FeedbackPattern {
663
+ trigger: string;
664
+ response: string;
665
+ }
666
+ /**
667
+ * How hard the simulated user pushes back. The driver LLM scales its tone
668
+ * and follow-up aggression to this:
669
+ * cooperative — forgiving early adopter; accepts reasonable answers.
670
+ * demanding — experienced professional; rejects vague or hedged answers.
671
+ * relentless — senior partner reviewing for a client who will litigate;
672
+ * interrogates every claim, accepts nothing undefended.
673
+ */
674
+ type PersonaRigor = 'cooperative' | 'demanding' | 'relentless';
675
+ interface PersonaConfig {
676
+ id: string;
677
+ role: string;
678
+ goal: string;
679
+ completionCriteria: CompletionCriterion[];
680
+ feedbackPatterns?: FeedbackPattern[];
681
+ maxTurns: number;
682
+ driverModel?: string;
683
+ /** How adversarial the simulated user is. Defaults to 'demanding'. */
684
+ rigor?: PersonaRigor;
685
+ /**
686
+ * Domain expertise the simulated user holds — quoted into the driver
687
+ * prompt so it challenges the agent with authority instead of vague
688
+ * dissatisfaction. e.g. "a 15-year M&A partner who knows GAAP
689
+ * working-capital mechanics cold".
690
+ */
691
+ expertise?: string;
692
+ /**
693
+ * Substantive issues a senior professional in this role would
694
+ * interrogate — traps the scenario hides, claims that must be defended.
695
+ * The driver probes these without revealing them verbatim; the agent
696
+ * must surface them on its own.
697
+ */
698
+ pressurePoints?: string[];
699
+ /**
700
+ * Curveballs the driver may inject once the agent is coasting — changed
701
+ * facts, a hostile counterparty position, a new constraint. Forces the
702
+ * agent to re-derive rather than recite.
703
+ */
704
+ curveballs?: string[];
705
+ }
706
+ interface DriverState {
707
+ tasks: number;
708
+ events: number;
709
+ proposals: {
710
+ pending: number;
711
+ approved: number;
712
+ rejected: number;
713
+ };
714
+ vaultFiles: string[];
715
+ codeBlocks: number;
716
+ generations: number;
717
+ }
718
+ interface TurnMetrics {
719
+ turn: number;
720
+ timestamp: string;
721
+ tasks: number;
722
+ events: number;
723
+ proposals: {
724
+ pending: number;
725
+ approved: number;
726
+ rejected: number;
727
+ };
728
+ vaultFiles: number;
729
+ responseLatencyMs: number;
730
+ responseChars: number;
731
+ codeBlocksProduced: number;
732
+ blocksExtracted: number;
733
+ qualityScore?: number;
734
+ inputTokens: number;
735
+ outputTokens: number;
736
+ estimatedCostUsd: number;
737
+ totalCostUsd: number;
738
+ completionPercent: number;
739
+ }
740
+ interface DriverResult {
741
+ personaId: string;
742
+ /** True when the simulated user professionally signed off (driver said DONE). */
743
+ completed: boolean;
744
+ /** Turn at which the simulated user signed off, or null if it never did. */
745
+ turnsToCompletion: number | null;
746
+ /**
747
+ * Turn at which nominal completionCriteria were first all met, or null.
748
+ * Distinct from turnsToCompletion: criteria can be met while the
749
+ * simulated professional is still unsatisfied with the work's rigor.
750
+ */
751
+ criteriaMetAtTurn: number | null;
752
+ totalTurns: number;
753
+ metrics: TurnMetrics[];
754
+ finalState: DriverState;
755
+ convergenceCurve: number[];
756
+ totalCostUsd: number;
757
+ finalQualityScore: number | null;
758
+ }
759
+ interface BenchmarkRunnerConfig {
760
+ scenarios: Scenario[];
761
+ judges: JudgeFn[];
762
+ systemPrompt: string;
763
+ model?: string;
764
+ judgeModel?: string;
765
+ passThreshold?: number;
766
+ generation?: number;
767
+ promptVersion?: string;
768
+ /** Shared ledger for agent and judge calls made by the benchmark. */
769
+ costLedger?: CostLedgerHandle;
770
+ }
771
+ interface JudgeInput {
772
+ scenario: Scenario;
773
+ turns: TurnResult[];
774
+ artifacts: CollectedArtifacts;
775
+ /** Shared ledger for paid built-in judges. Direct calls default to an uncapped ledger. */
776
+ costLedger?: CostLedgerHandle;
777
+ costPhase?: string;
778
+ costTags?: Record<string, string>;
779
+ signal?: AbortSignal;
780
+ }
781
+ type JudgeFn = (chat: ChatClient, input: JudgeInput) => Promise<JudgeScore[]>;
782
+ interface TestResult {
783
+ name: string;
784
+ passed: boolean;
785
+ duration: number;
786
+ detail?: string;
787
+ checks: CheckResult[];
788
+ }
789
+ interface CheckResult {
790
+ name: string;
791
+ passed: boolean;
792
+ expected: string;
793
+ actual: string;
794
+ }
795
+ interface EvalResult {
796
+ scenario: string;
797
+ status: 'pass' | 'fail' | 'skip';
798
+ duration: number;
799
+ detail?: string;
800
+ artifact?: string;
801
+ }
802
+ //#endregion
803
+ export { assertLlmRoute as $, ChatClient as A, SandboxSdkTransportOpts as B, ScenarioFile as C, TurnMetrics as D, Turn as E, CreateChatClientOpts as F, LlmCallResult as G, LlmCallError as H, CustomTransportOpts as I, LlmMessage as J, LlmClient as K, DirectProviderTransportOpts as L, ChatResponse as M, ChatTransport as N, TurnResult as O, CliBridgeTransportOpts as P, LlmUsage as Q, MockTransportOpts as R, Scenario as S, TestResult as T, LlmCallMetadata as U, createChatClient as V, LlmCallRequest as W, LlmRouteAssertionError as X, LlmResponseError as Y, LlmRouteRequirements as Z, PersonaConfig as _, RawProviderSink as _t, CheckResult as a, isTransientLlmError as at, RouteMap as b, providerFromBaseUrl as bt, DriverResult as c, stripFencedJson as ct, FeedbackPattern as d, InMemoryRawProviderSink as dt, backoffMs as et, JudgeConfig as f, InMemoryRawProviderSinkOptions as ft, JudgeScore as g, RawProviderEvent as gt, JudgeRubric as h, RawProviderDirection as ht, BenchmarkRunnerConfig as i, costReceiptFromLlmError as it, ChatRequest as j, ChatCallOpts as k, DriverState as l, FileSystemRawProviderSink as lt, JudgeInput as m, ProviderRedactor as mt, ArtifactResult as n, callLlmJson as nt, CollectedArtifacts as o, maximumChargeForLlmRequest as ot, JudgeFn as p, NoopRawProviderSink as pt, LlmClientOptions as q, BenchmarkReport as r, costReceiptFromLlm as rt, CompletionCriterion as s, probeLlm as st, ArtifactCheck as t, callLlm as tt, EvalResult as u, FileSystemRawProviderSinkOptions as ut, PersonaRigor as v, RawProviderSinkFilter as vt, ScenarioResult as w, RubricDimension as x, ProductClientConfig as y, defaultProviderRedactor as yt, RouterTransportOpts as z };
804
+ //# sourceMappingURL=types-CTvKfr5F.d.ts.map