@tangle-network/agent-eval 0.115.2 → 0.116.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/CHANGELOG.md +34 -0
  2. package/dist/analyst/index.d.ts +8 -10
  3. package/dist/analyst/index.js +28 -23
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyst-C8HHvfJp.d.ts → analyst-CFBc14Wc.d.ts} +1 -1
  6. package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs-0rz_m29H.d.ts} +3 -3
  7. package/dist/belief-state/index.d.ts +3 -3
  8. package/dist/benchmarks/index.d.ts +7 -3
  9. package/dist/benchmarks/index.js +5 -5
  10. package/dist/campaign/index.d.ts +212 -23
  11. package/dist/campaign/index.js +20 -5
  12. package/dist/{chunk-N6MTC3GK.js → chunk-3274WNK7.js} +428 -94
  13. package/dist/chunk-3274WNK7.js.map +1 -0
  14. package/dist/{chunk-DRPIZQIT.js → chunk-4D5RVB3W.js} +2 -2
  15. package/dist/{chunk-LVTGFSHF.js → chunk-7GKEAIAD.js} +2 -2
  16. package/dist/{chunk-DWLIGZBX.js → chunk-CIUOICJT.js} +748 -3
  17. package/dist/chunk-CIUOICJT.js.map +1 -0
  18. package/dist/{chunk-5NVBGKPH.js → chunk-GSW3OBHK.js} +1284 -182
  19. package/dist/chunk-GSW3OBHK.js.map +1 -0
  20. package/dist/{chunk-FUCQVFMU.js → chunk-GY4SYVPJ.js} +12 -3
  21. package/dist/chunk-GY4SYVPJ.js.map +1 -0
  22. package/dist/chunk-MPHTT5HE.js +74 -0
  23. package/dist/chunk-MPHTT5HE.js.map +1 -0
  24. package/dist/{chunk-I2HNIE6N.js → chunk-NBSS5NDZ.js} +4 -4
  25. package/dist/{chunk-QG5F6463.js → chunk-ONM6PEAE.js} +2 -2
  26. package/dist/cli.js +2 -2
  27. package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CdxteG0y.d.ts} +1 -1
  28. package/dist/contract/index.d.ts +19 -19
  29. package/dist/contract/index.js +6 -4
  30. package/dist/contract/index.js.map +1 -1
  31. package/dist/{control-CcBiAEnn.d.ts → control-DbcDxouY.d.ts} +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DDfv22MQ.d.ts} +2 -1
  34. package/dist/{gepa-dne9JDPL.d.ts → gepa-CQelRtuC.d.ts} +10 -8
  35. package/dist/hosted/index.d.ts +8 -4
  36. package/dist/{index-BTEpx9He.d.ts → index-DbCXJfZ1.d.ts} +2 -2
  37. package/dist/index.d.ts +27 -30
  38. package/dist/index.js +28 -22
  39. package/dist/index.js.map +1 -1
  40. package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-oMVxDTxl.d.ts} +1 -1
  41. package/dist/{integrity-qemeBAyx.d.ts → integrity-C6PZ73iC.d.ts} +1 -1
  42. package/dist/kind-factory-DWOvXjR_.d.ts +171 -0
  43. package/dist/meta-eval/index.d.ts +2 -2
  44. package/dist/multishot/index.d.ts +6 -2
  45. package/dist/openapi.json +1 -1
  46. package/dist/policy-edit-Clb2v6Oa.d.ts +708 -0
  47. package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration--vU0mMtD.d.ts} +4 -4
  48. package/dist/{provenance-Bibyg1U9.d.ts → provenance-BbVagC68.d.ts} +26 -14
  49. package/dist/{release-report-CCtzajxP.d.ts → release-report-CamNDe90.d.ts} +2 -2
  50. package/dist/reporting.d.ts +4 -4
  51. package/dist/{researcher-Dq-EtpbE.d.ts → researcher-Dwbo_Fxx.d.ts} +5 -5
  52. package/dist/rl.d.ts +11 -9
  53. package/dist/rl.js +2 -2
  54. package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-BIdf9h4R.d.ts} +1 -1
  55. package/dist/{run-record-B7RTi_ix.d.ts → run-record-CZmcpWPo.d.ts} +1 -1
  56. package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-CC0jx9ql.d.ts} +1 -1
  57. package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CKjePUMh.d.ts} +3 -3
  58. package/dist/{store-C1YxJDEK.d.ts → store-9cAScOcb.d.ts} +132 -1
  59. package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-DTNgQycC.d.ts} +1 -1
  60. package/dist/traces.d.ts +6 -8
  61. package/dist/{types-C5gJrOVT.d.ts → types-Ca_63YSD.d.ts} +59 -2
  62. package/dist/wire/index.js +2 -2
  63. package/docs/design/loop-taxonomy.md +1 -2
  64. package/package.json +1 -1
  65. package/dist/chunk-5NVBGKPH.js.map +0 -1
  66. package/dist/chunk-AN5UYSVD.js +0 -761
  67. package/dist/chunk-AN5UYSVD.js.map +0 -1
  68. package/dist/chunk-DWLIGZBX.js.map +0 -1
  69. package/dist/chunk-FUCQVFMU.js.map +0 -1
  70. package/dist/chunk-N6MTC3GK.js.map +0 -1
  71. package/dist/kind-factory-DcNg13sZ.d.ts +0 -508
  72. package/dist/llm-client-DyqEH4jH.d.ts +0 -265
  73. package/dist/policy-edit-RLn8GWof.d.ts +0 -103
  74. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  75. /package/dist/{chunk-DRPIZQIT.js.map → chunk-4D5RVB3W.js.map} +0 -0
  76. /package/dist/{chunk-LVTGFSHF.js.map → chunk-7GKEAIAD.js.map} +0 -0
  77. /package/dist/{chunk-I2HNIE6N.js.map → chunk-NBSS5NDZ.js.map} +0 -0
  78. /package/dist/{chunk-QG5F6463.js.map → chunk-ONM6PEAE.js.map} +0 -0
@@ -0,0 +1,708 @@
1
+ import { R as RunRecord, A as AgentProfileCell, f as AgentProfileJson } from './run-record-CZmcpWPo.js';
2
+ import { A as AgentEvalError, C as CaptureIntegrityError, V as ValidationError } from './errors-oeQrLqXC.js';
3
+ import { R as RawProviderSink, P as ProviderRedactor, T as TraceAnalysisStore } from './store-9cAScOcb.js';
4
+ import { a as JudgeInput } from './types-C7DGg5ex.js';
5
+
6
+ /**
7
+ * LLM client with graceful degrade.
8
+ *
9
+ * OpenAI-compatible `/v1/chat/completions` client with:
10
+ * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
11
+ * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
12
+ * - Graceful json_schema → json_object degrade on 400 with schema-reject body.
13
+ * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
14
+ * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
15
+ * directly, cli-bridge subscriptions, and any router that speaks the spec.
16
+ *
17
+ * Usage:
18
+ * const { value, result } = await callLlmJson<MyType>(
19
+ * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
20
+ * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
21
+ * )
22
+ *
23
+ * This is THE llm-calling seam for agent-eval primitives that need structured
24
+ * output (semantic concept judge, reviewer directives, critic scores). Primitives
25
+ * that need free-form text use `callLlm` and parse output themselves.
26
+ */
27
+
28
+ interface LlmMessage {
29
+ role: 'system' | 'user' | 'assistant';
30
+ /**
31
+ * Either a plain text content string OR a multimodal content array
32
+ * (text + image_url parts) for vision-capable models.
33
+ */
34
+ content: string | Array<{
35
+ type: 'text';
36
+ text: string;
37
+ } | {
38
+ type: 'image_url';
39
+ image_url: {
40
+ url: string;
41
+ detail?: 'auto' | 'low' | 'high';
42
+ };
43
+ }>;
44
+ }
45
+ interface LlmCallRequest {
46
+ model: string;
47
+ messages: LlmMessage[];
48
+ /** Optional JSON-mode response format (response_format: json_object). */
49
+ jsonMode?: boolean;
50
+ /** Optional structured output via JSON Schema. Falls back to json_object on 400. */
51
+ jsonSchema?: {
52
+ name: string;
53
+ schema: Record<string, unknown>;
54
+ };
55
+ temperature?: number;
56
+ maxTokens?: number;
57
+ /** Per-call timeout, default 300s. */
58
+ timeoutMs?: number;
59
+ }
60
+ interface LlmUsage {
61
+ promptTokens: number;
62
+ completionTokens: number;
63
+ totalTokens: number;
64
+ /** Proxies populate this when prompt caching is on. */
65
+ cachedPromptTokens?: number;
66
+ }
67
+ interface LlmCallResult {
68
+ /** The text content of the first choice. Empty string if none. */
69
+ content: string;
70
+ usage: LlmUsage;
71
+ /**
72
+ * Cost in USD. Pulled from proxy's `_response_cost` field when present;
73
+ * `null` when neither the proxy nor the caller can derive it.
74
+ */
75
+ costUsd: number | null;
76
+ /** Model name actually used (echoed from response). */
77
+ model: string;
78
+ /** Wall-clock duration of the HTTP call (last attempt, if retried). */
79
+ durationMs: number;
80
+ /**
81
+ * `finish_reason` echoed from the first choice (`stop`, `length`,
82
+ * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
83
+ * Exposed so a free-form `callLlm` caller CAN detect a truncated answer
84
+ * (`length`) instead of treating a cut-off completion as complete. Note:
85
+ * `callLlm` does not itself reject on it — acting on this signal is the
86
+ * caller's responsibility (in-repo free-form drivers do not yet enforce it).
87
+ */
88
+ finishReason?: string | null;
89
+ /**
90
+ * True when `content.trim()` is empty. An empty completion is a silent zero
91
+ * for free-form `callLlm` callers; this flag is the signal a caller can
92
+ * inspect to fail loud rather than proceed on an empty string. `callLlm`
93
+ * surfaces it but does not throw on it.
94
+ */
95
+ contentEmpty?: boolean;
96
+ /** Raw response body. */
97
+ raw: Record<string, unknown>;
98
+ }
99
+ declare class LlmCallError extends AgentEvalError {
100
+ readonly status: number;
101
+ readonly body: string;
102
+ readonly model: string;
103
+ constructor(message: string, status: number, body: string, model: string);
104
+ }
105
+ interface LlmClientOptions {
106
+ /** Base URL (without trailing slash). Must end at the `/v1` prefix. */
107
+ baseUrl?: string;
108
+ /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
109
+ apiKey?: string;
110
+ bearer?: string;
111
+ /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
112
+ authHeader?: {
113
+ name: string;
114
+ value: string;
115
+ };
116
+ /** Default timeout in ms. Per-call can override. */
117
+ defaultTimeoutMs?: number;
118
+ /**
119
+ * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
120
+ * each attempt's per-attempt timeout controller, so aborting it cancels
121
+ * the in-flight fetch. A caller abort is FATAL: it is not retried even
122
+ * though an AbortError otherwise matches the transient patterns.
123
+ */
124
+ signal?: AbortSignal;
125
+ /**
126
+ * Cross-attempt wall-clock budget in ms, measured from the first attempt.
127
+ * Before launching each attempt the loop checks the remaining budget and
128
+ * stops retrying once it is exhausted, rather than waiting the full
129
+ * per-attempt timeout on every retry. Bounds total time independent of
130
+ * `maxRetries` × `timeoutMs`.
131
+ */
132
+ deadlineMs?: number;
133
+ /** Max retry attempts on retriable errors. Default 3 (1 initial + 2 retries). */
134
+ maxRetries?: number;
135
+ /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
136
+ fetch?: typeof fetch;
137
+ /**
138
+ * Optional raw HTTP capture sink. When provided, every request, response,
139
+ * and error (across all retry attempts) is recorded to the sink, with auth
140
+ * headers and credential-shaped body fields redacted by default. This is
141
+ * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
142
+ * raw events record what actually crossed the wire.
143
+ */
144
+ rawSink?: RawProviderSink;
145
+ /**
146
+ * Logical provider id attached to raw events. When omitted, derived from
147
+ * `baseUrl` via `providerFromBaseUrl`.
148
+ */
149
+ provider?: string;
150
+ /** Trace context attached to raw events; populated by emitter-aware callers. */
151
+ traceContext?: {
152
+ runId?: string;
153
+ spanId?: string;
154
+ };
155
+ /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
156
+ redactor?: ProviderRedactor;
157
+ }
158
+ /**
159
+ * True when an error is a transient transport/network fault worth retrying,
160
+ * as opposed to a deterministic failure (4xx schema reject, JSON parse) that
161
+ * a retry cannot fix. Inspects `LlmCallError.status`, then the error's
162
+ * name/message/code, then recurses into `error.cause` — undici nests the
163
+ * real socket fault one or more levels under `.cause`.
164
+ *
165
+ * This is THE retry classifier for the package: `callLlm` and
166
+ * `withJudgeRetry` both route through it, so a connection-class error is
167
+ * treated identically whether it surfaces in the HTTP client or a
168
+ * TCloud-backed judge.
169
+ */
170
+ declare function isTransientLlmError(err: unknown): boolean;
171
+ /** Exponential backoff: 500ms, 1s, 2s, 4s, ... capped at 16s. Attempt is 0-indexed. */
172
+ declare function backoffMs(attempt: number): number;
173
+ /**
174
+ * Strip a ```json / ``` code fence if the model emitted one.
175
+ * Idempotent for naked JSON. Some models (claude-code via router, certain
176
+ * deepseek models) wrap output even under json_object.
177
+ */
178
+ declare function stripFencedJson(raw: string): string;
179
+ /**
180
+ * Low-level call. Returns raw content + usage + cost. Retries on transient
181
+ * failures; does NOT degrade schema here — callers that want graceful
182
+ * degrade use `callLlmJson`.
183
+ */
184
+ declare function callLlm(req: LlmCallRequest, opts?: LlmClientOptions): Promise<LlmCallResult>;
185
+ /**
186
+ * Structured-output call. Returns parsed JSON plus the raw result envelope.
187
+ * Degrades `jsonSchema` → `jsonMode` on a 400 that names the schema param —
188
+ * critical for deepseek-v3/v4, kimi-k2.6, and other models that don't accept
189
+ * the `response_format.json_schema` shape but DO accept `json_object`.
190
+ */
191
+ declare function callLlmJson<T = unknown>(req: LlmCallRequest, opts?: LlmClientOptions): Promise<{
192
+ value: T;
193
+ result: LlmCallResult;
194
+ }>;
195
+ type LlmRouteAssertionReason = 'no_explicit_base_url' | 'base_url_blocked' | 'base_url_not_allowed' | 'no_auth' | 'wrong_provider';
196
+ declare class LlmRouteAssertionError extends CaptureIntegrityError {
197
+ readonly reason: LlmRouteAssertionReason;
198
+ readonly baseUrl: string;
199
+ constructor(message: string, reason: LlmRouteAssertionReason, baseUrl: string);
200
+ }
201
+ interface LlmRouteRequirements {
202
+ /**
203
+ * Throw if `opts.baseUrl` is undefined, i.e. the call would fall back to
204
+ * `DEFAULT_BASE_URL`. Set this for evaluation runs where silently using
205
+ * the public/free-tier router is a defect — the launch reviewer needs to
206
+ * know exactly which provider answered.
207
+ */
208
+ requireExplicitBaseUrl?: boolean;
209
+ /**
210
+ * Allowlist of acceptable base URLs. Strings match by prefix
211
+ * (case-insensitive); RegExps test against the full base URL.
212
+ */
213
+ allowedBaseUrls?: Array<string | RegExp>;
214
+ /** Blocklist that takes precedence over `allowedBaseUrls`. */
215
+ blockedBaseUrls?: Array<string | RegExp>;
216
+ /** Throw if no auth header / api key is configured. */
217
+ requireAuth?: boolean;
218
+ /**
219
+ * Logical provider id the configured `baseUrl` is expected to match (via
220
+ * `providerFromBaseUrl`). Mainly useful when paired with `requireExplicitBaseUrl`.
221
+ */
222
+ expectedProvider?: string;
223
+ }
224
+ /**
225
+ * Fail-loud assertion that the configured LLM client points at the route
226
+ * the caller intends. Designed for the matrix-runner preflight: invoke
227
+ * once before any LLM call to catch misconfiguration before a sweep burns
228
+ * dollars on the wrong provider.
229
+ *
230
+ * Throws `LlmRouteAssertionError`. Pure — no I/O — so it's safe to call
231
+ * from constructors and CI gates.
232
+ */
233
+ declare function assertLlmRoute(opts: LlmClientOptions, req?: LlmRouteRequirements): void;
234
+ /**
235
+ * Probe whether a model is reachable. Returns latency + null error on
236
+ * success; `ok=false` + error message on any failure (HTTP, timeout,
237
+ * network, parse). Designed for sweep preflights — fail loud at the
238
+ * boundary before burning a 30-leaf run on a misconfigured router.
239
+ *
240
+ * Sends a tiny `ping` message with `maxTokens=64`. Reasoning models
241
+ * (glm-5.1, deepseek-v4) can burn the entire budget on internal reasoning
242
+ * for short prompts, so don't tighten this further. We don't validate
243
+ * content; HTTP 200 means reachable.
244
+ */
245
+ declare function probeLlm(model: string, opts?: LlmClientOptions & {
246
+ timeoutMs?: number;
247
+ }): Promise<{
248
+ ok: boolean;
249
+ latencyMs: number;
250
+ error: string | null;
251
+ }>;
252
+ /**
253
+ * Stateful client — construct once with defaults, call many times.
254
+ * Thin wrapper around the free functions; exists for callers that want
255
+ * to inject a single configured instance into multiple primitives.
256
+ */
257
+ declare class LlmClient {
258
+ private readonly opts;
259
+ constructor(opts?: LlmClientOptions);
260
+ call(req: LlmCallRequest, per?: LlmClientOptions): Promise<LlmCallResult>;
261
+ callJson<T = unknown>(req: LlmCallRequest, per?: LlmClientOptions): Promise<{
262
+ value: T;
263
+ result: LlmCallResult;
264
+ }>;
265
+ }
266
+
267
+ /**
268
+ * ChatClient — the single LLM abstraction analysts call.
269
+ *
270
+ * agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
271
+ * graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
272
+ * mixed patterns force every analyst author to pick a transport, which
273
+ * couples analyst code to runtime concerns (cli-bridge vs router vs
274
+ * sandbox-sdk) it shouldn't know about.
275
+ *
276
+ * `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
277
+ * The operator decides at the registry boundary which transport binds
278
+ * to it. Analyst code stays transport-agnostic; swapping production
279
+ * (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
280
+ * line factory call.
281
+ *
282
+ * Designed to coexist: existing `LlmClient` callers and existing
283
+ * `TCloud`-based judges keep working untouched. New analyst code uses
284
+ * `ChatClient`. When old call sites migrate, they pick up budgeting,
285
+ * cancellation, and unified telemetry for free.
286
+ */
287
+
288
+ /**
289
+ * Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
290
+ * compatible mental model stays. Two methods: a one-shot `chat()` and
291
+ * an `streamChat()` for future agentic loops (not yet exposed).
292
+ */
293
+ interface ChatClient {
294
+ /** Display name of the bound transport — included in telemetry. */
295
+ readonly transport: ChatTransport;
296
+ /** Default model when caller omits — operators bind this per environment. */
297
+ readonly defaultModel?: string;
298
+ chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
299
+ }
300
+ type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
301
+ interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
302
+ /** Optional — falls back to ChatClient.defaultModel. */
303
+ model?: string;
304
+ }
305
+ type ChatResponse = LlmCallResult;
306
+ interface ChatCallOpts {
307
+ /** Cancel the in-flight request. */
308
+ signal?: AbortSignal;
309
+ /** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
310
+ maxCostUsd?: number;
311
+ /** Correlation tag carried into request headers when the transport allows. */
312
+ correlationId?: string;
313
+ }
314
+ type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
315
+ interface BaseTransportOpts {
316
+ defaultModel?: string;
317
+ }
318
+ interface RouterTransportOpts extends BaseTransportOpts {
319
+ transport: 'router';
320
+ baseUrl?: string;
321
+ apiKey: string;
322
+ }
323
+ interface CliBridgeTransportOpts extends BaseTransportOpts {
324
+ transport: 'cli-bridge';
325
+ baseUrl?: string;
326
+ bearer?: string;
327
+ }
328
+ interface DirectProviderTransportOpts extends BaseTransportOpts {
329
+ transport: 'direct-provider';
330
+ baseUrl: string;
331
+ apiKey: string;
332
+ }
333
+ /**
334
+ * Sandbox-SDK transport. Provided as a thin pass-through: the caller
335
+ * supplies a callable that mimics LlmClient.chat() against an already-
336
+ * configured Sandbox handle. We don't import the SDK here to keep
337
+ * agent-eval dep-free of @tangle-network/sandbox.
338
+ */
339
+ interface SandboxSdkTransportOpts extends BaseTransportOpts {
340
+ transport: 'sandbox-sdk';
341
+ chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
342
+ }
343
+ /**
344
+ * Mock transport for tests. The handler receives the request and returns
345
+ * whatever the test wants. No retries, no JSON-schema degrade.
346
+ */
347
+ interface MockTransportOpts extends BaseTransportOpts {
348
+ transport: 'mock';
349
+ handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
350
+ }
351
+ /**
352
+ * Build a ChatClient bound to a specific transport. The returned client
353
+ * is safe to share across analysts in a single registry run.
354
+ */
355
+ declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
356
+
357
+ /**
358
+ * Analyst contract — the missing orchestration layer over agent-eval's
359
+ * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
360
+ * SemanticConceptJudge, JudgeFn, ...).
361
+ *
362
+ * Each existing primitive returns its own output shape. The Analyst
363
+ * contract is the single envelope every primitive lifts into, so a
364
+ * registry can run N analysts against a run and a single renderer can
365
+ * compose findings without knowing which analyzer produced them.
366
+ *
367
+ * The contract is intentionally domain-agnostic: nothing here knows
368
+ * about code, voice, RAG, or any particular agent stack. Analysts
369
+ * declare what INPUT KIND they need (a trace store, an artifact dir,
370
+ * a RunRecord, a JudgeInput, or `custom`), and the registry routes
371
+ * the matching input from `AnalystRunInputs`.
372
+ */
373
+
374
+ /**
375
+ * Unified envelope every analyst emits. Schema-versioned so renderers
376
+ * and time-series diffs survive future field additions.
377
+ */
378
+ interface AnalystFinding {
379
+ schema_version: '1.0.0';
380
+ /**
381
+ * Stable hash over identity-defining fields (analyst_id + canonical
382
+ * claim + area + optional subject). Two findings from two runs that
383
+ * "are the same finding" share this id — that's what `diffFindings`
384
+ * uses to compute appeared/disappeared sets across runs.
385
+ */
386
+ finding_id: string;
387
+ analyst_id: string;
388
+ produced_at: string;
389
+ severity: AnalystSeverity;
390
+ /**
391
+ * Coarse classification. Renderers group by this. Free-form so
392
+ * domain-specific analysts can introduce categories without a
393
+ * schema change ('agent-reasoning', 'verification', 'cost',
394
+ * 'tool-use', 'safety', 'latency', 'data-quality', ...).
395
+ */
396
+ area: string;
397
+ claim: string;
398
+ rationale?: string;
399
+ evidence_refs: EvidenceRef[];
400
+ recommended_action?: string;
401
+ validation_plan?: string;
402
+ /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
403
+ confidence: number;
404
+ /**
405
+ * Optional subject the finding is about — leaf id, agent id, request
406
+ * id. Included in finding_id when present so per-subject findings
407
+ * diff cleanly across runs.
408
+ */
409
+ subject?: string;
410
+ /** FIREWALL provenance (docs/learning-flywheel.md): true iff this finding was
411
+ * lifted from a JUDGE verdict (an acceptance score), not OBSERVED from the
412
+ * agent's behavior. A judge-derived finding must NEVER be admitted as a
413
+ * steering input — that is the held-out judge leaking into the loop. Set at
414
+ * the lift site (createJudgeAdapter); checked by `assertNoJudgeVerdict`.
415
+ * Provenance, not evidence presence, is the correct discriminator: an
416
+ * evidence-less trace-analyst observation legitimately steers, while a judge
417
+ * verdict that happens to cite an artifact must not. */
418
+ derived_from_judge?: boolean;
419
+ /** Analyst-private extras; renderers ignore unless they know the analyst. */
420
+ metadata?: Record<string, unknown>;
421
+ }
422
+ type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
423
+ interface EvidenceRef {
424
+ /**
425
+ * Where the evidence lives. `span` and `event` refer to OTLP trace
426
+ * elements; `artifact` to a file inside the run's artifact tree;
427
+ * `finding` to another AnalystFinding (cross-analyst chaining);
428
+ * `metric` to a named scalar reading the renderer knows how to read.
429
+ */
430
+ kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
431
+ uri: string;
432
+ excerpt?: string;
433
+ }
434
+ /**
435
+ * The discriminator the registry uses to pass the right input.
436
+ * `custom` is the escape hatch — analysts that need something else
437
+ * (e.g. an embedding cache, a partner SDK handle) read it from
438
+ * `AnalystRunInputs.custom[<analyst id>]`.
439
+ */
440
+ type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
441
+ interface AnalystCost {
442
+ /** `deterministic` analysts MUST NOT call the LLM. */
443
+ kind: 'deterministic' | 'llm';
444
+ /** Optional declared upper bound; the registry can enforce a budget. */
445
+ est_usd_per_run?: number;
446
+ /** Models the analyst expects to use (informational). */
447
+ models?: string[];
448
+ }
449
+ interface AnalystRequirements {
450
+ /** Min number of shots / samples the analyst needs to produce signal. */
451
+ min_shots?: number;
452
+ /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
453
+ capabilities?: string[];
454
+ }
455
+ /**
456
+ * What's passed to every analyst call. The registry resolves which
457
+ * field the analyst's `inputKind` selects and asserts it's present.
458
+ */
459
+ interface AnalystRunInputs {
460
+ traceStore?: TraceAnalysisStore;
461
+ artifactDir?: string;
462
+ runRecord?: RunRecord;
463
+ judgeInput?: JudgeInput;
464
+ /** Keyed by analyst id; populated by callers that registered custom analysts. */
465
+ custom?: Record<string, unknown>;
466
+ }
467
+ interface AnalystContext {
468
+ runId: string;
469
+ /** Stable correlation id so logs from a single registry.run() share a tag. */
470
+ correlationId: string;
471
+ /** Wall-clock deadline (epoch ms). Analysts SHOULD honor for graceful cancel. */
472
+ deadlineMs?: number;
473
+ /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
474
+ budgetUsd?: number;
475
+ /**
476
+ * Shared chat client. Analysts that call an LLM go through this so
477
+ * the operator picks transport (sandbox-sdk | router | cli-bridge |
478
+ * direct-provider | mock) at the registry boundary without touching
479
+ * analyst code.
480
+ */
481
+ chat?: ChatClient;
482
+ /**
483
+ * Findings from a prior run the operator wants the analyst to see as
484
+ * retrieval context. Kinds that take advantage of cross-run memory
485
+ * (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
486
+ * page I asked for is still missing") render these into the actor's
487
+ * working set. Filtering is the operator's job: pass the slice that
488
+ * matches the analyst's id, or pass everything and let the kind
489
+ * filter. Empty / absent means no cross-run context.
490
+ */
491
+ priorFindings?: ReadonlyArray<AnalystFinding>;
492
+ /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
493
+ tags?: Record<string, string>;
494
+ /** Logger callback — analysts SHOULD prefer this over console.* for testability. */
495
+ log?: (msg: string, fields?: Record<string, unknown>) => void;
496
+ /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
497
+ signal?: AbortSignal;
498
+ }
499
+ /**
500
+ * The minimal contract. Concrete analysts can refine `TInput` so
501
+ * implementations stay type-safe (e.g. a trace analyst's `TInput` is
502
+ * `TraceAnalysisStore`); the registry passes the right field from
503
+ * `AnalystRunInputs` based on `inputKind`.
504
+ */
505
+ interface Analyst<TInput = unknown> {
506
+ /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
507
+ readonly id: string;
508
+ /** Human-readable. One sentence. */
509
+ readonly description: string;
510
+ readonly inputKind: AnalystInputKind;
511
+ readonly cost: AnalystCost;
512
+ readonly requires?: AnalystRequirements;
513
+ /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
514
+ readonly version: string;
515
+ analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
516
+ }
517
+ /**
518
+ * Compute the stable finding_id from the identity-defining fields.
519
+ * Default implementation hashes {analyst_id, area, subject, normalized claim}.
520
+ * Analysts that emit findings whose claim text varies per run (timestamps,
521
+ * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
522
+ * or (b) move the variable part into `rationale`/`metadata` and keep the
523
+ * `claim` static.
524
+ */
525
+ declare function computeFindingId(input: {
526
+ analyst_id: string;
527
+ area: string;
528
+ subject?: string;
529
+ claim: string;
530
+ /** Override the claim for hashing — use when the displayed claim has run-specific bits. */
531
+ id_basis?: string;
532
+ }): string;
533
+ /**
534
+ * Convenience factory: produce a fully-formed AnalystFinding with the
535
+ * id computed automatically. Analyst code stays terse.
536
+ */
537
+ declare function makeFinding(init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {
538
+ id_basis?: string;
539
+ produced_at?: string;
540
+ }): AnalystFinding;
541
+ interface AnalystRunSummary {
542
+ analyst_id: string;
543
+ status: 'ok' | 'skipped' | 'failed';
544
+ /** Why skipped — missing input, budget exceeded, capability unmet. */
545
+ reason?: string;
546
+ findings_count: number;
547
+ latency_ms: number;
548
+ cost_usd: number;
549
+ /** When `status='failed'`: the error class + message, never the full stack. */
550
+ error?: {
551
+ class: string;
552
+ message: string;
553
+ };
554
+ }
555
+ interface AnalystRunResult {
556
+ run_id: string;
557
+ correlation_id: string;
558
+ started_at: string;
559
+ ended_at: string;
560
+ findings: AnalystFinding[];
561
+ per_analyst: AnalystRunSummary[];
562
+ /** Total LLM cost in USD across all analysts in this registry.run(). */
563
+ total_cost_usd: number;
564
+ }
565
+ /**
566
+ * Events emitted by `AnalystRegistry.runStream(...)` in real time as
567
+ * the registry executes. UIs subscribe via `for await (const ev of
568
+ * registry.runStream(...))`; `registry.run(...)` is a thin collector
569
+ * over the same stream, so the two surfaces share their invariants.
570
+ *
571
+ * Per-finding events are intentionally omitted — analyzers are batch
572
+ * operations (an Ax actor returns the full `findings:json[]` at the
573
+ * end of the responder), so streaming inside one analyst would only
574
+ * emit partial JSON consumers can't render. The kind-completion event
575
+ * is the right granularity; subscribers wanting per-finding rendering
576
+ * iterate `event.findings` themselves.
577
+ */
578
+ type AnalystRunEvent = {
579
+ type: 'run-started';
580
+ run_id: string;
581
+ correlation_id: string;
582
+ started_at: string;
583
+ /** The ordered list of analyst ids the registry will run. */
584
+ analyst_ids: ReadonlyArray<string>;
585
+ } | {
586
+ type: 'analyst-skipped';
587
+ summary: AnalystRunSummary;
588
+ } | {
589
+ type: 'analyst-started';
590
+ analyst_id: string;
591
+ started_at: string;
592
+ } | {
593
+ type: 'analyst-completed';
594
+ /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
595
+ summary: AnalystRunSummary;
596
+ findings: ReadonlyArray<AnalystFinding>;
597
+ } | {
598
+ type: 'run-completed';
599
+ result: AnalystRunResult;
600
+ };
601
+
602
+ type PolicyEditSchemaVersion = 'policy-edit/v1';
603
+ declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
604
+ type PolicyEditAxis = (typeof POLICY_EDIT_AXES)[number];
605
+ declare const POLICY_EDIT_TARGET_SURFACES: readonly ["prompt", "tool-contract", "runtime-config", "memory", "agent-profile", "code", "deployment"];
606
+ type PolicyEditTargetSurface = (typeof POLICY_EDIT_TARGET_SURFACES)[number];
607
+ type PolicyEditRisk = 'low' | 'medium' | 'high' | 'unknown';
608
+ type PolicyEditGainDirection = 'increase' | 'decrease';
609
+ type PolicyEditGainUnit = 'absolute' | 'relative' | 'percent' | 'score';
610
+ interface PolicyEditTarget {
611
+ surface: PolicyEditTargetSurface;
612
+ /** Stable path inside the target surface, for example `system-prompt:tools`
613
+ * or `budget.maxTurns`. */
614
+ path?: string;
615
+ /** Optional canonical deployment identity. Store the existing cell, not a
616
+ * local profile shape. */
617
+ agentProfileCell?: AgentProfileCell;
618
+ /** Human label when the path is not enough for a readable audit trail. */
619
+ label?: string;
620
+ }
621
+ type PolicyEditChange = {
622
+ kind: 'text';
623
+ mode: 'append' | 'prepend' | 'replace';
624
+ value: string;
625
+ /** Required when `mode === 'replace'`; exact match only. */
626
+ find?: string;
627
+ } | {
628
+ kind: 'json';
629
+ mode: 'set' | 'merge' | 'remove';
630
+ path: string;
631
+ value?: AgentProfileJson;
632
+ };
633
+ interface PolicyEditExpectedGain {
634
+ /** Metric this edit is expected to move, e.g. `holdout.composite`. */
635
+ metric: string;
636
+ direction: PolicyEditGainDirection;
637
+ /** Positive magnitude in the metric's native units. */
638
+ amount: number;
639
+ unit?: PolicyEditGainUnit;
640
+ rationale?: string;
641
+ }
642
+ interface PolicyEditSource {
643
+ findingIds: string[];
644
+ analystIds: string[];
645
+ evidenceRefs: EvidenceRef[];
646
+ /** Mirrors `AnalystFinding.derived_from_judge`; admission rejects it. */
647
+ derivedFromJudge?: boolean;
648
+ }
649
+ interface PolicyEdit {
650
+ schemaVersion: PolicyEditSchemaVersion;
651
+ editId: string;
652
+ axis: PolicyEditAxis;
653
+ target: PolicyEditTarget;
654
+ change: PolicyEditChange;
655
+ claim: string;
656
+ expectedGain: PolicyEditExpectedGain;
657
+ confidence: number;
658
+ risk: PolicyEditRisk;
659
+ source: PolicyEditSource;
660
+ rationale?: string;
661
+ validationPlan?: string;
662
+ metadata?: Record<string, unknown>;
663
+ }
664
+ declare const POLICY_EDIT_CANDIDATE_RECORD_SCHEMA: "tangle.policy-edit-candidate.v1";
665
+ /** JSON-safe attribution carried with a measured candidate and its scores. */
666
+ interface PolicyEditCandidateRecord {
667
+ schema: typeof POLICY_EDIT_CANDIDATE_RECORD_SCHEMA;
668
+ policyEdit: PolicyEdit;
669
+ }
670
+ type PolicyEditInit = Omit<PolicyEdit, 'schemaVersion' | 'editId'> & {
671
+ schemaVersion?: PolicyEditSchemaVersion;
672
+ editId?: string;
673
+ };
674
+ declare class PolicyEditValidationError extends ValidationError {
675
+ readonly path: string;
676
+ constructor(message: string, path?: string);
677
+ }
678
+ interface FindingToPolicyEditOptions {
679
+ expectedGain?: PolicyEditExpectedGain | ((finding: AnalystFinding) => PolicyEditExpectedGain | null | undefined);
680
+ risk?: PolicyEditRisk | ((finding: AnalystFinding) => PolicyEditRisk);
681
+ defaultAxis?: PolicyEditAxis;
682
+ defaultTargetSurface?: PolicyEditTargetSurface;
683
+ }
684
+ interface PolicyEditAdmissionOptions {
685
+ minScore?: number;
686
+ minExpectedGain?: number;
687
+ allowHighRisk?: boolean;
688
+ requireEvidence?: boolean;
689
+ }
690
+ interface PolicyEditAdmission {
691
+ edit: PolicyEdit;
692
+ decision: 'admit' | 'reject';
693
+ score: number;
694
+ reasons: string[];
695
+ }
696
+ declare function makePolicyEdit(init: PolicyEditInit): PolicyEdit;
697
+ declare function computePolicyEditId(edit: Omit<PolicyEdit, 'editId'> | PolicyEdit): string;
698
+ declare function validatePolicyEdit(input: unknown): PolicyEdit;
699
+ declare function makePolicyEditCandidateRecord(edit: PolicyEdit): PolicyEditCandidateRecord;
700
+ declare function validatePolicyEditCandidateRecord(input: unknown): PolicyEditCandidateRecord;
701
+ declare function isPolicyEdit(input: unknown): input is PolicyEdit;
702
+ declare function policyEditsFromFindings(findings: ReadonlyArray<AnalystFinding>, opts?: FindingToPolicyEditOptions): PolicyEdit[];
703
+ declare function policyEditFromFinding(finding: AnalystFinding, opts?: FindingToPolicyEditOptions): PolicyEdit | null;
704
+ declare function scorePolicyEditReadiness(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): number;
705
+ declare function admitPolicyEdit(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): PolicyEditAdmission;
706
+ declare function applyPolicyEditToSurface(surface: unknown, edit: PolicyEdit): unknown;
707
+
708
+ export { makePolicyEditCandidateRecord as $, type Analyst as A, type PolicyEditGainDirection as B, type ChatClient as C, type DirectProviderTransportOpts as D, type EvidenceRef as E, type FindingToPolicyEditOptions as F, type PolicyEditGainUnit as G, type PolicyEditInit as H, type PolicyEditRisk as I, type PolicyEditSchemaVersion as J, type PolicyEditSource as K, type LlmClientOptions as L, type MockTransportOpts as M, type PolicyEditTarget as N, type PolicyEditTargetSurface as O, type PolicyEditCandidateRecord as P, PolicyEditValidationError as Q, type RouterTransportOpts as R, type SandboxSdkTransportOpts as S, admitPolicyEdit as T, applyPolicyEditToSurface as U, computeFindingId as V, computePolicyEditId as W, createChatClient as X, isPolicyEdit as Y, makeFinding as Z, makePolicyEdit as _, type LlmRouteRequirements as a, policyEditFromFinding as a0, policyEditsFromFindings as a1, scorePolicyEditReadiness as a2, validatePolicyEdit as a3, validatePolicyEditCandidateRecord as a4, LlmCallError as a5, type LlmCallRequest as a6, type LlmCallResult as a7, LlmClient as a8, type LlmMessage as a9, LlmRouteAssertionError as aa, type LlmUsage as ab, assertLlmRoute as ac, backoffMs as ad, callLlm as ae, callLlmJson as af, isTransientLlmError as ag, probeLlm as ah, stripFencedJson as ai, type AnalystContext as b, type AnalystRunSummary as c, type AnalystFinding as d, type AnalystRunResult as e, type AnalystRunInputs as f, type AnalystRunEvent as g, type AnalystCost as h, type AnalystSeverity as i, type AnalystInputKind as j, type AnalystRequirements as k, type ChatCallOpts as l, type ChatRequest as m, type ChatResponse as n, type ChatTransport as o, type CliBridgeTransportOpts as p, type CreateChatClientOpts as q, POLICY_EDIT_AXES as r, POLICY_EDIT_CANDIDATE_RECORD_SCHEMA as s, POLICY_EDIT_TARGET_SURFACES as t, type PolicyEdit as u, type PolicyEditAdmission as v, type PolicyEditAdmissionOptions as w, type PolicyEditAxis as x, type PolicyEditChange as y, type PolicyEditExpectedGain as z };