@tangle-network/agent-eval 0.116.0 → 0.117.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/CHANGELOG.md +31 -0
  2. package/dist/analyst/index.d.ts +18 -11
  3. package/dist/analyst/index.js +10 -7
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
  6. package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
  7. package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
  8. package/dist/belief-state/index.d.ts +6 -6
  9. package/dist/belief-state/index.js +1 -1
  10. package/dist/benchmarks/index.d.ts +11 -8
  11. package/dist/benchmarks/index.js +11 -10
  12. package/dist/builder-eval/index.d.ts +4 -4
  13. package/dist/builder-eval/index.js +1 -1
  14. package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
  15. package/dist/campaign/index.d.ts +53 -29
  16. package/dist/campaign/index.js +18 -13
  17. package/dist/chunk-3YYRZDON.js +45 -0
  18. package/dist/chunk-3YYRZDON.js.map +1 -0
  19. package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
  20. package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
  21. package/dist/chunk-CCZIVI3F.js.map +1 -0
  22. package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
  23. package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
  24. package/dist/chunk-HHWE3POT.js +94 -0
  25. package/dist/chunk-HHWE3POT.js.map +1 -0
  26. package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
  27. package/dist/chunk-HQPHZGL6.js.map +1 -0
  28. package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
  29. package/dist/chunk-IDZTTFRR.js.map +1 -0
  30. package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
  31. package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
  32. package/dist/chunk-LQUTGLOZ.js.map +1 -0
  33. package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
  34. package/dist/chunk-LTVG32KX.js.map +1 -0
  35. package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
  36. package/dist/chunk-MGEHEHSN.js.map +1 -0
  37. package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
  38. package/dist/chunk-NJC7U437.js.map +1 -0
  39. package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
  40. package/dist/chunk-ODVOOEWQ.js.map +1 -0
  41. package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
  42. package/dist/chunk-S2F4J57L.js.map +1 -0
  43. package/dist/chunk-VCTY3W6J.js +798 -0
  44. package/dist/chunk-VCTY3W6J.js.map +1 -0
  45. package/dist/chunk-VF3XSYTI.js +545 -0
  46. package/dist/chunk-VF3XSYTI.js.map +1 -0
  47. package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
  48. package/dist/chunk-YZPO4UHR.js.map +1 -0
  49. package/dist/{chunk-GSW3OBHK.js → chunk-ZUXV7UWZ.js} +350 -724
  50. package/dist/chunk-ZUXV7UWZ.js.map +1 -0
  51. package/dist/cli.js +4 -2
  52. package/dist/cli.js.map +1 -1
  53. package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
  54. package/dist/contract/index.d.ts +43 -29
  55. package/dist/contract/index.js +56 -19
  56. package/dist/contract/index.js.map +1 -1
  57. package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
  58. package/dist/control.d.ts +6 -6
  59. package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
  60. package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
  61. package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
  62. package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
  63. package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
  64. package/dist/fuzz.d.ts +8 -16
  65. package/dist/fuzz.js +72 -42
  66. package/dist/fuzz.js.map +1 -1
  67. package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
  68. package/dist/hosted/index.d.ts +13 -10
  69. package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
  70. package/dist/index.d.ts +102 -57
  71. package/dist/index.js +328 -235
  72. package/dist/index.js.map +1 -1
  73. package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
  74. package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
  75. package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
  76. package/dist/llm-client-qoDd18Qz.d.ts +289 -0
  77. package/dist/meta-eval/index.d.ts +8 -7
  78. package/dist/meta-eval/index.js +1 -1
  79. package/dist/multishot/index.d.ts +9 -6
  80. package/dist/openapi.json +1 -1
  81. package/dist/pipelines/index.d.ts +16 -6
  82. package/dist/pipelines/index.js +119 -23
  83. package/dist/pipelines/index.js.map +1 -1
  84. package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
  85. package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
  86. package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
  87. package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
  88. package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
  89. package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
  90. package/dist/reporting.d.ts +10 -9
  91. package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
  92. package/dist/rl.d.ts +18 -15
  93. package/dist/rl.js +2 -2
  94. package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
  95. package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
  96. package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
  97. package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
  98. package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
  99. package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
  100. package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
  101. package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
  102. package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
  103. package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
  104. package/dist/storyboard/index.d.ts +1 -1
  105. package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
  106. package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
  107. package/dist/traces.d.ts +25 -14
  108. package/dist/traces.js +16 -4
  109. package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
  110. package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
  111. package/dist/wire/index.d.ts +28 -19
  112. package/dist/wire/index.js +4 -2
  113. package/docs/distributed-driver.md +1 -1
  114. package/package.json +3 -3
  115. package/dist/chunk-3274WNK7.js.map +0 -1
  116. package/dist/chunk-4D5RVB3W.js.map +0 -1
  117. package/dist/chunk-7GKEAIAD.js +0 -205
  118. package/dist/chunk-7GKEAIAD.js.map +0 -1
  119. package/dist/chunk-CIUOICJT.js.map +0 -1
  120. package/dist/chunk-FAOEFFRT.js.map +0 -1
  121. package/dist/chunk-GSW3OBHK.js.map +0 -1
  122. package/dist/chunk-GY4SYVPJ.js.map +0 -1
  123. package/dist/chunk-LNQEP766.js.map +0 -1
  124. package/dist/chunk-MHNQWM4I.js.map +0 -1
  125. package/dist/chunk-MPHTT5HE.js +0 -74
  126. package/dist/chunk-MPHTT5HE.js.map +0 -1
  127. package/dist/chunk-NBSS5NDZ.js.map +0 -1
  128. package/dist/chunk-NYFUT3B3.js.map +0 -1
  129. package/dist/chunk-TLDB7WRY.js.map +0 -1
  130. package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
  131. /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
  132. /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
  133. /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
  134. /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
  135. /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
@@ -1,4 +1,4 @@
1
- import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-DTNgQycC.js';
1
+ import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-C5bKFfm-.js';
2
2
  import { C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
3
3
 
4
4
  /**
@@ -1,6 +1,6 @@
1
1
  import { C as CaptureIntegrityError } from './errors-oeQrLqXC.js';
2
- import { R as RawProviderSink } from './store-9cAScOcb.js';
3
- import { T as TraceStore } from './store-BsVi7ncX.js';
2
+ import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
3
+ import { T as TraceStore } from './store-DGqD0Pyo.js';
4
4
 
5
5
  /**
6
6
  * Run-completion integrity check — at end of run, verify the expected event
@@ -1,7 +1,7 @@
1
1
  import { AxAIService, AxFunction } from '@ax-llm/ax';
2
- import { T as TraceAnalysisStore } from './store-9cAScOcb.js';
2
+ import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
3
3
  import { z } from 'zod';
4
- import { h as AnalystCost, b as AnalystContext, A as Analyst } from './policy-edit-Clb2v6Oa.js';
4
+ import { g as AnalystCost, a as AnalystContext, A as Analyst } from './policy-edit-wG9uFEFm.js';
5
5
 
6
6
  /**
7
7
  * Typed Ax output for analyst findings.
@@ -0,0 +1,289 @@
1
+ import { c as CostReceiptInput, M as MaximumCharge } from './cost-ledger-DWy3XdJc.js';
2
+ import { A as AgentEvalError, C as CaptureIntegrityError } from './errors-oeQrLqXC.js';
3
+ import { R as RawProviderSink, P as ProviderRedactor } from './raw-provider-sink-C46HDghv.js';
4
+
5
+ /**
6
+ * LLM client with graceful degrade.
7
+ *
8
+ * OpenAI-compatible `/v1/chat/completions` client with:
9
+ * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
10
+ * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
11
+ * - Graceful json_schema → json_object degrade on 400 with schema-reject body.
12
+ * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
13
+ * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
14
+ * directly, cli-bridge subscriptions, and any router that speaks the spec.
15
+ *
16
+ * Usage:
17
+ * const { value, result } = await callLlmJson<MyType>(
18
+ * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
19
+ * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
20
+ * )
21
+ *
22
+ * This is THE llm-calling seam for agent-eval primitives that need structured
23
+ * output (semantic concept judge, reviewer directives, critic scores). Primitives
24
+ * that need free-form text use `callLlm` and parse output themselves.
25
+ */
26
+
27
+ interface LlmMessage {
28
+ role: 'system' | 'user' | 'assistant';
29
+ /**
30
+ * Either a plain text content string OR a multimodal content array
31
+ * (text + image_url parts) for vision-capable models.
32
+ */
33
+ content: string | Array<{
34
+ type: 'text';
35
+ text: string;
36
+ } | {
37
+ type: 'image_url';
38
+ image_url: {
39
+ url: string;
40
+ detail?: 'auto' | 'low' | 'high';
41
+ };
42
+ }>;
43
+ }
44
+ interface LlmCallRequest {
45
+ model: string;
46
+ messages: LlmMessage[];
47
+ /** Optional JSON-mode response format (response_format: json_object). */
48
+ jsonMode?: boolean;
49
+ /** Optional structured output via JSON Schema. Falls back to json_object on 400. */
50
+ jsonSchema?: {
51
+ name: string;
52
+ schema: Record<string, unknown>;
53
+ };
54
+ temperature?: number;
55
+ maxTokens?: number;
56
+ /** Per-call timeout, default 300s. */
57
+ timeoutMs?: number;
58
+ }
59
+ /** Conservative priced bound for the exact text request sent to a provider.
60
+ * Returns undefined when output or multimodal input is not bounded, causing a
61
+ * capped CostLedger to reject the call before execution. */
62
+ declare function maximumChargeForLlmRequest(request: Pick<LlmCallRequest, 'model' | 'messages' | 'jsonSchema' | 'maxTokens'>, options?: LlmClientOptions): MaximumCharge | undefined;
63
+ interface LlmUsage {
64
+ promptTokens: number;
65
+ completionTokens: number;
66
+ totalTokens: number;
67
+ /** False when the provider omitted or malformed prompt/completion usage. */
68
+ captured?: boolean;
69
+ /** Proxies populate this when prompt caching is on. */
70
+ cachedPromptTokens?: number;
71
+ }
72
+ interface LlmCallResult {
73
+ /** The text content of the first choice. Empty string if none. */
74
+ content: string;
75
+ usage: LlmUsage;
76
+ /**
77
+ * Cost in USD. Pulled from proxy's `_response_cost` field when present;
78
+ * `null` when neither the proxy nor the caller can derive it.
79
+ */
80
+ costUsd: number | null;
81
+ /** Model name actually used (echoed from response). */
82
+ model: string;
83
+ /** Wall-clock duration of the HTTP call (last attempt, if retried). */
84
+ durationMs: number;
85
+ /**
86
+ * `finish_reason` echoed from the first choice (`stop`, `length`,
87
+ * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
88
+ * Exposed so a free-form `callLlm` caller CAN detect a truncated answer
89
+ * (`length`) instead of treating a cut-off completion as complete. Note:
90
+ * `callLlm` does not itself reject on it — acting on this signal is the
91
+ * caller's responsibility (in-repo free-form drivers do not yet enforce it).
92
+ */
93
+ finishReason?: string | null;
94
+ /**
95
+ * True when `content.trim()` is empty. An empty completion is a silent zero
96
+ * for free-form `callLlm` callers; this flag is the signal a caller can
97
+ * inspect to fail loud rather than proceed on an empty string. `callLlm`
98
+ * surfaces it but does not throw on it.
99
+ */
100
+ contentEmpty?: boolean;
101
+ /** Raw response body. */
102
+ raw: Record<string, unknown>;
103
+ }
104
+ type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
105
+ /** Convert a provider result into the canonical paid-call receipt input. */
106
+ declare function costReceiptFromLlm(result: LlmCallResult): CostReceiptInput;
107
+ /** Structured-response failures retain their completed provider receipt. */
108
+ declare function costReceiptFromLlmError(error: Error): CostReceiptInput | undefined;
109
+ declare class LlmCallError extends AgentEvalError {
110
+ readonly status: number;
111
+ readonly body: string;
112
+ readonly model: string;
113
+ constructor(message: string, status: number, body: string, model: string);
114
+ }
115
+ /** A provider response completed and incurred measurable usage, but its content
116
+ * could not satisfy the caller's response contract. The response envelope is
117
+ * retained so accounting can commit the receipt before the error propagates. */
118
+ declare class LlmResponseError extends AgentEvalError {
119
+ readonly result: LlmCallResult;
120
+ constructor(message: string, result: LlmCallResult, options?: {
121
+ cause?: unknown;
122
+ });
123
+ }
124
+ interface LlmClientOptions {
125
+ /** Base URL (without trailing slash). Must end at the `/v1` prefix. */
126
+ baseUrl?: string;
127
+ /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
128
+ apiKey?: string;
129
+ bearer?: string;
130
+ /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
131
+ authHeader?: {
132
+ name: string;
133
+ value: string;
134
+ };
135
+ /** Stable provider idempotency key, reused across retries of this logical call. */
136
+ idempotencyKey?: string;
137
+ /** Default timeout in ms. Per-call can override. */
138
+ defaultTimeoutMs?: number;
139
+ /**
140
+ * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
141
+ * each attempt's per-attempt timeout controller, so aborting it cancels
142
+ * the in-flight fetch. A caller abort is FATAL: it is not retried even
143
+ * though an AbortError otherwise matches the transient patterns.
144
+ */
145
+ signal?: AbortSignal;
146
+ /**
147
+ * Cross-attempt wall-clock budget in ms, measured from the first attempt.
148
+ * Before launching each attempt the loop checks the remaining budget and
149
+ * stops retrying once it is exhausted, rather than waiting the full
150
+ * per-attempt timeout on every retry. Bounds total time independent of
151
+ * total attempts × `timeoutMs`.
152
+ */
153
+ deadlineMs?: number;
154
+ /** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
155
+ maxRetries?: number;
156
+ /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
157
+ fetch?: typeof fetch;
158
+ /**
159
+ * Optional raw HTTP capture sink. When provided, every request, response,
160
+ * and error (across all retry attempts) is recorded to the sink, with auth
161
+ * headers and credential-shaped body fields redacted by default. This is
162
+ * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
163
+ * raw events record what actually crossed the wire.
164
+ */
165
+ rawSink?: RawProviderSink;
166
+ /**
167
+ * Logical provider id attached to raw events. When omitted, derived from
168
+ * `baseUrl` via `providerFromBaseUrl`.
169
+ */
170
+ provider?: string;
171
+ /** Trace context attached to raw events; populated by emitter-aware callers. */
172
+ traceContext?: {
173
+ runId?: string;
174
+ spanId?: string;
175
+ };
176
+ /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
177
+ redactor?: ProviderRedactor;
178
+ }
179
+ /**
180
+ * True when an error is a transient transport/network fault worth retrying,
181
+ * as opposed to a deterministic failure (4xx schema reject, JSON parse) that
182
+ * a retry cannot fix. Inspects `LlmCallError.status`, then the error's
183
+ * name/message/code, then recurses into `error.cause` — undici nests the
184
+ * real socket fault one or more levels under `.cause`.
185
+ *
186
+ * This is THE retry classifier for the package: `callLlm` and
187
+ * `withJudgeRetry` both route through it, so a connection-class error is
188
+ * treated identically whether it surfaces in the HTTP client or a
189
+ * TCloud-backed judge.
190
+ */
191
+ declare function isTransientLlmError(err: unknown): boolean;
192
+ /** Exponential backoff: 500ms, 1s, 2s, 4s, ... capped at 16s. Attempt is 0-indexed. */
193
+ declare function backoffMs(attempt: number): number;
194
+ /**
195
+ * Strip a ```json / ``` code fence if the model emitted one.
196
+ * Idempotent for naked JSON. Some models (claude-code via router, certain
197
+ * deepseek models) wrap output even under json_object.
198
+ */
199
+ declare function stripFencedJson(raw: string): string;
200
+ /**
201
+ * Low-level call. Returns raw content + usage + cost. Retries on transient
202
+ * failures; does NOT degrade schema here — callers that want graceful
203
+ * degrade use `callLlmJson`.
204
+ */
205
+ declare function callLlm(req: LlmCallRequest, opts?: LlmClientOptions): Promise<LlmCallResult>;
206
+ /**
207
+ * Structured-output call. Returns parsed JSON plus the raw result envelope.
208
+ * Degrades `jsonSchema` → `jsonMode` on a 400 that names the schema param —
209
+ * critical for deepseek-v3/v4, kimi-k2.6, and other models that don't accept
210
+ * the `response_format.json_schema` shape but DO accept `json_object`.
211
+ */
212
+ declare function callLlmJson<T = unknown>(req: LlmCallRequest, opts?: LlmClientOptions): Promise<{
213
+ value: T;
214
+ result: LlmCallResult;
215
+ }>;
216
+ type LlmRouteAssertionReason = 'no_explicit_base_url' | 'base_url_blocked' | 'base_url_not_allowed' | 'no_auth' | 'wrong_provider';
217
+ declare class LlmRouteAssertionError extends CaptureIntegrityError {
218
+ readonly reason: LlmRouteAssertionReason;
219
+ readonly baseUrl: string;
220
+ constructor(message: string, reason: LlmRouteAssertionReason, baseUrl: string);
221
+ }
222
+ interface LlmRouteRequirements {
223
+ /**
224
+ * Throw if `opts.baseUrl` is undefined, i.e. the call would fall back to
225
+ * `DEFAULT_BASE_URL`. Set this for evaluation runs where silently using
226
+ * the public/free-tier router is a defect — the launch reviewer needs to
227
+ * know exactly which provider answered.
228
+ */
229
+ requireExplicitBaseUrl?: boolean;
230
+ /**
231
+ * Allowlist of acceptable base URLs. Strings match by prefix
232
+ * (case-insensitive); RegExps test against the full base URL.
233
+ */
234
+ allowedBaseUrls?: Array<string | RegExp>;
235
+ /** Blocklist that takes precedence over `allowedBaseUrls`. */
236
+ blockedBaseUrls?: Array<string | RegExp>;
237
+ /** Throw if no auth header / api key is configured. */
238
+ requireAuth?: boolean;
239
+ /**
240
+ * Logical provider id the configured `baseUrl` is expected to match (via
241
+ * `providerFromBaseUrl`). Mainly useful when paired with `requireExplicitBaseUrl`.
242
+ */
243
+ expectedProvider?: string;
244
+ }
245
+ /**
246
+ * Fail-loud assertion that the configured LLM client points at the route
247
+ * the caller intends. Designed for the matrix-runner preflight: invoke
248
+ * once before any LLM call to catch misconfiguration before a sweep burns
249
+ * dollars on the wrong provider.
250
+ *
251
+ * Throws `LlmRouteAssertionError`. Pure — no I/O — so it's safe to call
252
+ * from constructors and CI gates.
253
+ */
254
+ declare function assertLlmRoute(opts: LlmClientOptions, req?: LlmRouteRequirements): void;
255
+ /**
256
+ * Probe whether a model is reachable. Returns latency + null error on
257
+ * success; `ok=false` + error message on any failure (HTTP, timeout,
258
+ * network, parse). Designed for sweep preflights — fail loud at the
259
+ * boundary before burning a 30-leaf run on a misconfigured router.
260
+ *
261
+ * Sends a tiny `ping` message with `maxTokens=64`. Reasoning models
262
+ * (glm-5.1, deepseek-v4) can burn the entire budget on internal reasoning
263
+ * for short prompts, so don't tighten this further. We don't validate
264
+ * content; HTTP 200 means reachable.
265
+ */
266
+ declare function probeLlm(model: string, opts?: LlmClientOptions & {
267
+ timeoutMs?: number;
268
+ }): Promise<{
269
+ ok: boolean;
270
+ latencyMs: number;
271
+ error: string | null;
272
+ }>;
273
+ /**
274
+ * Stateful client — construct once with defaults, call many times.
275
+ * Thin wrapper around the free functions; exists for callers that want
276
+ * to inject a single configured instance into multiple primitives.
277
+ */
278
+ declare class LlmClient {
279
+ readonly maximumAttempts: number;
280
+ private readonly opts;
281
+ constructor(opts?: LlmClientOptions);
282
+ call(req: LlmCallRequest, per?: LlmClientOptions): Promise<LlmCallResult>;
283
+ callJson<T = unknown>(req: LlmCallRequest, per?: LlmClientOptions): Promise<{
284
+ value: T;
285
+ result: LlmCallResult;
286
+ }>;
287
+ }
288
+
289
+ export { type LlmCallMetadata as L, type LlmClientOptions as a, type LlmRouteRequirements as b, type LlmCallRequest as c, type LlmCallResult as d, type LlmUsage as e, LlmCallError as f, LlmClient as g, type LlmMessage as h, LlmResponseError as i, LlmRouteAssertionError as j, assertLlmRoute as k, backoffMs as l, callLlm as m, callLlmJson as n, costReceiptFromLlm as o, costReceiptFromLlmError as p, isTransientLlmError as q, maximumChargeForLlmRequest as r, probeLlm as s, stripFencedJson as t };
@@ -1,15 +1,16 @@
1
- export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-Dz8TQV4y.js';
1
+ export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-C8MTS7cw.js';
2
2
  export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, O as OutcomeFilter, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
3
- export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-BIdf9h4R.js';
3
+ export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-p49lLVrE.js';
4
4
  import { C as ContinuousAgreement, a as CalibrationResult, b as ContinuousCalibrationResult, c as CandidateScore, G as GoldenItem } from '../judge-calibration-7C-IDmKr.js';
5
5
  import { S as SeriesConvergenceOptions, a as SeriesConvergenceResult } from '../series-convergence-D5OWMBg6.js';
6
- import { C as CorpusAgreementReport } from '../statistics-oUbOJe-S.js';
7
- import '../store-BsVi7ncX.js';
8
- import '../schema-SGWcK9wa.js';
9
- import '../run-record-CZmcpWPo.js';
6
+ import { C as CorpusAgreementReport } from '../statistics-KUnG73jH.js';
7
+ import '../store-DGqD0Pyo.js';
8
+ import '../schema-B3Q3l9Z_.js';
9
+ import '../run-record-BDH49H2E.js';
10
10
  import '@tangle-network/agent-interface';
11
11
  import '../errors-oeQrLqXC.js';
12
- import '../types-C7DGg5ex.js';
12
+ import '../types-BkfcQnxV.js';
13
+ import '../cost-ledger-DWy3XdJc.js';
13
14
  import '@tangle-network/tcloud';
14
15
 
15
16
  /**
@@ -19,7 +19,7 @@ import {
19
19
  import {
20
20
  aggregateLlm,
21
21
  llmSpans
22
- } from "../chunk-MHNQWM4I.js";
22
+ } from "../chunk-LQUTGLOZ.js";
23
23
  import {
24
24
  ValidationError
25
25
  } from "../chunk-ONWEPEDO.js";
@@ -1,13 +1,16 @@
1
- import { J as JudgeScore } from '../types-Ca_63YSD.js';
1
+ import { J as JudgeScore } from '../types-BSw1rOUB.js';
2
2
  import { AgentProfile } from '@tangle-network/agent-interface';
3
3
  import { M as MatrixResult } from '../types-BUxNaJ8c.js';
4
- import '../policy-edit-Clb2v6Oa.js';
5
- import '../run-record-CZmcpWPo.js';
4
+ import '../policy-edit-wG9uFEFm.js';
5
+ import '../run-record-BDH49H2E.js';
6
6
  import '../errors-oeQrLqXC.js';
7
- import '../schema-SGWcK9wa.js';
8
- import '../store-9cAScOcb.js';
9
- import '../types-C7DGg5ex.js';
7
+ import '../schema-B3Q3l9Z_.js';
8
+ import '../store-C1YxJDEK.js';
9
+ import '../types-BkfcQnxV.js';
10
+ import '../cost-ledger-DWy3XdJc.js';
10
11
  import '@tangle-network/tcloud';
12
+ import '../llm-client-qoDd18Qz.js';
13
+ import '../raw-provider-sink-C46HDghv.js';
11
14
  import '../verdict-C9MlYujm.js';
12
15
 
13
16
  interface MultishotMessage {
package/dist/openapi.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "openapi": "3.1.0",
3
3
  "info": {
4
4
  "title": "@tangle-network/agent-eval — wire protocol",
5
- "version": "0.116.0",
5
+ "version": "0.117.0",
6
6
  "description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
7
7
  "contact": {
8
8
  "name": "Tangle Network",
@@ -1,9 +1,9 @@
1
- import { h as BudgetSpec, R as Run, T as ToolSpan } from '../schema-SGWcK9wa.js';
2
- import { T as TraceStore, R as RunFilter } from '../store-BsVi7ncX.js';
3
- export { a as FailureCluster, F as FailureClusterReport, f as failureClusterView } from '../failure-cluster-C48PiReX.js';
4
- import { T as TrajectoryStep, B as BaselineOptions, a as BaselineReport } from '../baseline-DsNteOgR.js';
5
- export { c as computeToolUseMetrics } from '../baseline-DsNteOgR.js';
6
- import { l as llmSpans } from '../query-Ck190MOd.js';
1
+ import { h as BudgetSpec, R as Run, T as ToolSpan } from '../schema-B3Q3l9Z_.js';
2
+ import { T as TraceStore, R as RunFilter } from '../store-DGqD0Pyo.js';
3
+ export { a as FailureCluster, F as FailureClusterReport, f as failureClusterView } from '../failure-cluster-DOAcSJ87.js';
4
+ import { T as TrajectoryStep, B as BaselineOptions, a as BaselineReport } from '../baseline-DKq3gJpP.js';
5
+ export { c as computeToolUseMetrics } from '../baseline-DKq3gJpP.js';
6
+ import { l as llmSpans } from '../query-CF7PG61p.js';
7
7
 
8
8
  /**
9
9
  * BudgetBreachView — aggregates breach events across the corpus.
@@ -121,8 +121,11 @@ interface StuckLoopFinding {
121
121
  runId: string;
122
122
  toolName: string;
123
123
  argHash: string;
124
+ /** Calls in this episode's densest qualifying interval, not the whole-run total. */
124
125
  occurrences: number;
125
126
  spanIds: string[];
127
+ /** Nearest agent ancestor, or the direct parent when ancestry is incomplete. */
128
+ scopeSpanId?: string;
126
129
  /** Milliseconds between first and last call in the loop. */
127
130
  windowMs: number;
128
131
  }
@@ -134,6 +137,13 @@ interface StuckLoopReport {
134
137
  interface StuckLoopOptions {
135
138
  /** Minimum call count to flag a loop (default 3). */
136
139
  minOccurrences?: number;
140
+ /** Maximum time between the first and last repeated call (default 60 seconds). */
141
+ maxWindowMs?: number;
142
+ /**
143
+ * Maximum other tool calls allowed between adjacent repeats (default 0).
144
+ * Set to 1 to detect alternating patterns such as A,B,A,B,A.
145
+ */
146
+ maxInterveningToolCalls?: number;
137
147
  /** Filter to a specific runId; omit to scan the entire corpus. */
138
148
  runId?: string;
139
149
  }
@@ -4,7 +4,10 @@ import {
4
4
  classifyFailure,
5
5
  compareToBaseline,
6
6
  computeToolUseMetrics
7
- } from "../chunk-NYFUT3B3.js";
7
+ } from "../chunk-ODVOOEWQ.js";
8
+ import {
9
+ executionTrackByLane
10
+ } from "../chunk-HHWE3POT.js";
8
11
  import {
9
12
  interRaterReliability,
10
13
  pearsonR
@@ -12,10 +15,12 @@ import {
12
15
  import {
13
16
  aggregateLlm,
14
17
  argHash,
18
+ hasCapturedToolArgs,
19
+ isToolSpan,
15
20
  llmSpans,
16
21
  runFailureClass,
17
22
  toolSpans
18
- } from "../chunk-MHNQWM4I.js";
23
+ } from "../chunk-LQUTGLOZ.js";
19
24
  import "../chunk-ONWEPEDO.js";
20
25
  import "../chunk-PZ5AY32C.js";
21
26
 
@@ -80,7 +85,7 @@ async function failureClusterView(store, options = {}) {
80
85
  const trig = spans.find((s) => s.spanId === cls.triggerSpanId);
81
86
  if (trig?.kind === "tool") {
82
87
  toolName = trig.toolName;
83
- argPrefix = argHash(trig.args).slice(0, 16);
88
+ if (hasCapturedToolArgs(trig)) argPrefix = argHash(trig.args).slice(0, 16);
84
89
  } else if (trig?.kind === "judge") {
85
90
  dimension = trig.dimension;
86
91
  }
@@ -90,7 +95,7 @@ async function failureClusterView(store, options = {}) {
90
95
  const errored = ts.filter((t) => t.status === "error").pop();
91
96
  if (errored) {
92
97
  toolName = errored.toolName;
93
- argPrefix = argHash(errored.args).slice(0, 16);
98
+ if (hasCapturedToolArgs(errored)) argPrefix = argHash(errored.args).slice(0, 16);
94
99
  }
95
100
  }
96
101
  if (!dimension) {
@@ -296,33 +301,97 @@ function defaultExtract(metric) {
296
301
  }
297
302
 
298
303
  // src/pipelines/stuck-loop.ts
304
+ var DEFAULT_MAX_WINDOW_MS = 6e4;
305
+ var DEFAULT_MAX_INTERVENING_TOOL_CALLS = 0;
299
306
  async function stuckLoopView(store, options = {}) {
300
307
  const minOccurrences = options.minOccurrences ?? 3;
308
+ const maxWindowMs = options.maxWindowMs ?? DEFAULT_MAX_WINDOW_MS;
309
+ const maxInterveningToolCalls = options.maxInterveningToolCalls ?? DEFAULT_MAX_INTERVENING_TOOL_CALLS;
310
+ if (!Number.isInteger(minOccurrences) || minOccurrences < 1) {
311
+ throw new RangeError("minOccurrences must be a positive integer");
312
+ }
313
+ if (!Number.isFinite(maxWindowMs) || maxWindowMs < 0) {
314
+ throw new RangeError("maxWindowMs must be a finite non-negative number");
315
+ }
316
+ if (!Number.isInteger(maxInterveningToolCalls) || maxInterveningToolCalls < 0) {
317
+ throw new RangeError("maxInterveningToolCalls must be a non-negative integer");
318
+ }
301
319
  const runs = options.runId ? [{ runId: options.runId }] : (await store.listRuns()).map((r) => ({ runId: r.runId }));
302
320
  const findings = [];
303
321
  for (const { runId } of runs) {
304
- const tools = await toolSpans(store, runId);
322
+ const spans = await store.spans({ runId });
323
+ const spansById = new Map(spans.map((span) => [span.spanId, span]));
324
+ const scopedTools = spans.filter(isToolSpan).map((span, sourceIndex) => ({ span, sourceIndex })).sort((a, b) => a.span.startedAt - b.span.startedAt || a.sourceIndex - b.sourceIndex).map(({ span }) => ({ span, ...executionScope(span, spansById) }));
325
+ const trackByLane = executionTrackByLane(
326
+ scopedTools.map((call) => {
327
+ const direct = call.laneSpanId === null || call.laneSpanId === call.scopeSpanId;
328
+ const timed = direct ? call.span : call.laneSpanId ? spansById.get(call.laneSpanId) : void 0;
329
+ return {
330
+ key: executionKey(call),
331
+ scopeKey: JSON.stringify(call.scopeSpanId),
332
+ start: timed?.startedAt ?? null,
333
+ end: timed?.endedAt ?? null
334
+ };
335
+ })
336
+ );
337
+ const nextToolIndexByTrack = /* @__PURE__ */ new Map();
338
+ const orderedTools = scopedTools.map((call) => {
339
+ const trackId = trackByLane.get(executionKey(call));
340
+ const toolCallIndex = nextToolIndexByTrack.get(trackId) ?? 0;
341
+ nextToolIndexByTrack.set(trackId, toolCallIndex + 1);
342
+ return { ...call, toolCallIndex, trackId };
343
+ });
305
344
  const byKey = /* @__PURE__ */ new Map();
306
- for (const t of tools) {
307
- const h = argHash(t.args);
308
- const key = `${t.toolName}\0${h}`;
309
- const bucket = byKey.get(key) ?? { spans: [], argHash: h, toolName: t.toolName };
310
- bucket.spans.push(t);
345
+ for (const call of orderedTools) {
346
+ if (!hasCapturedToolArgs(call.span)) continue;
347
+ const h = argHash(call.span.args);
348
+ const key = JSON.stringify([call.trackId, call.span.toolName, h]);
349
+ const bucket = byKey.get(key) ?? {
350
+ calls: [],
351
+ argHash: h,
352
+ toolName: call.span.toolName,
353
+ scopeSpanId: call.scopeSpanId
354
+ };
355
+ bucket.calls.push(call);
311
356
  byKey.set(key, bucket);
312
357
  }
313
- for (const { spans, argHash: h, toolName } of byKey.values()) {
314
- if (spans.length < minOccurrences) continue;
315
- const sorted = [...spans].sort((a, b) => a.startedAt - b.startedAt);
316
- const first = sorted[0].startedAt;
317
- const last = sorted[sorted.length - 1].startedAt;
318
- findings.push({
319
- runId,
320
- toolName,
321
- argHash: h,
322
- occurrences: sorted.length,
323
- spanIds: sorted.map((s) => s.spanId),
324
- windowMs: last - first
325
- });
358
+ for (const { calls, argHash: h, toolName, scopeSpanId } of byKey.values()) {
359
+ if (calls.length < minOccurrences) continue;
360
+ let episodeStart = 0;
361
+ for (let episodeEnd = 1; episodeEnd <= calls.length; episodeEnd += 1) {
362
+ const previous = calls[episodeEnd - 1];
363
+ const next = calls[episodeEnd];
364
+ const episodeEnded = next === void 0 || next.span.startedAt - previous.span.startedAt > maxWindowMs || next.toolCallIndex - previous.toolCallIndex - 1 > maxInterveningToolCalls || !callsAreSerial(previous, next);
365
+ if (!episodeEnded) continue;
366
+ const episode = calls.slice(episodeStart, episodeEnd);
367
+ let left = 0;
368
+ let bestStart = 0;
369
+ let bestEnd = -1;
370
+ for (let right = 0; right < episode.length; right += 1) {
371
+ while (episode[right].span.startedAt - episode[left].span.startedAt > maxWindowMs) {
372
+ left += 1;
373
+ }
374
+ if (right - left > bestEnd - bestStart) {
375
+ bestStart = left;
376
+ bestEnd = right;
377
+ }
378
+ }
379
+ if (bestEnd - bestStart + 1 >= minOccurrences) {
380
+ const loop = episode.slice(bestStart, bestEnd + 1);
381
+ const first = loop[0].span.startedAt;
382
+ const last = loop[loop.length - 1].span.startedAt;
383
+ findings.push({
384
+ runId,
385
+ toolName,
386
+ argHash: h,
387
+ occurrences: loop.length,
388
+ spanIds: loop.map((call) => call.span.spanId),
389
+ ...scopeSpanId ? { scopeSpanId } : {},
390
+ windowMs: last - first
391
+ });
392
+ }
393
+ episodeStart = episodeEnd;
394
+ }
326
395
  }
327
396
  }
328
397
  const affectedRuns = new Set(findings.map((f) => f.runId));
@@ -332,6 +401,33 @@ async function stuckLoopView(store, options = {}) {
332
401
  totalRuns: runs.length
333
402
  };
334
403
  }
404
+ function laneKey(call) {
405
+ return JSON.stringify([call.scopeSpanId, call.laneSpanId]);
406
+ }
407
+ function executionKey(call) {
408
+ return call.laneSpanId === null || call.laneSpanId === call.scopeSpanId ? JSON.stringify([call.scopeSpanId, call.span.spanId]) : laneKey(call);
409
+ }
410
+ function executionScope(span, spansById) {
411
+ const directParent = span.parentSpanId;
412
+ if (!directParent) return { scopeSpanId: null, laneSpanId: null };
413
+ let currentId = directParent;
414
+ let laneSpanId = null;
415
+ const seen = /* @__PURE__ */ new Set();
416
+ while (currentId && !seen.has(currentId)) {
417
+ seen.add(currentId);
418
+ const current = spansById.get(currentId);
419
+ if (!current) return { scopeSpanId: directParent, laneSpanId: directParent };
420
+ if (current.kind === "agent") {
421
+ return { scopeSpanId: current.spanId, laneSpanId: laneSpanId ?? current.spanId };
422
+ }
423
+ laneSpanId = current.spanId;
424
+ currentId = current.parentSpanId;
425
+ }
426
+ return { scopeSpanId: directParent, laneSpanId: directParent };
427
+ }
428
+ function callsAreSerial(previous, next) {
429
+ return previous.span.endedAt !== void 0 && previous.span.endedAt <= next.span.startedAt;
430
+ }
335
431
 
336
432
  // src/pipelines/tool-waste.ts
337
433
  async function toolWasteView(store, options = {}) {