@tangle-network/agent-eval 0.115.2 → 0.116.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/CHANGELOG.md +34 -0
  2. package/dist/analyst/index.d.ts +8 -10
  3. package/dist/analyst/index.js +28 -23
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyst-C8HHvfJp.d.ts → analyst-CFBc14Wc.d.ts} +1 -1
  6. package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs-0rz_m29H.d.ts} +3 -3
  7. package/dist/belief-state/index.d.ts +3 -3
  8. package/dist/benchmarks/index.d.ts +7 -3
  9. package/dist/benchmarks/index.js +5 -5
  10. package/dist/campaign/index.d.ts +212 -23
  11. package/dist/campaign/index.js +20 -5
  12. package/dist/{chunk-N6MTC3GK.js → chunk-3274WNK7.js} +428 -94
  13. package/dist/chunk-3274WNK7.js.map +1 -0
  14. package/dist/{chunk-DRPIZQIT.js → chunk-4D5RVB3W.js} +2 -2
  15. package/dist/{chunk-LVTGFSHF.js → chunk-7GKEAIAD.js} +2 -2
  16. package/dist/{chunk-DWLIGZBX.js → chunk-CIUOICJT.js} +748 -3
  17. package/dist/chunk-CIUOICJT.js.map +1 -0
  18. package/dist/{chunk-5NVBGKPH.js → chunk-GSW3OBHK.js} +1284 -182
  19. package/dist/chunk-GSW3OBHK.js.map +1 -0
  20. package/dist/{chunk-FUCQVFMU.js → chunk-GY4SYVPJ.js} +12 -3
  21. package/dist/chunk-GY4SYVPJ.js.map +1 -0
  22. package/dist/chunk-MPHTT5HE.js +74 -0
  23. package/dist/chunk-MPHTT5HE.js.map +1 -0
  24. package/dist/{chunk-I2HNIE6N.js → chunk-NBSS5NDZ.js} +4 -4
  25. package/dist/{chunk-QG5F6463.js → chunk-ONM6PEAE.js} +2 -2
  26. package/dist/cli.js +2 -2
  27. package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CdxteG0y.d.ts} +1 -1
  28. package/dist/contract/index.d.ts +19 -19
  29. package/dist/contract/index.js +6 -4
  30. package/dist/contract/index.js.map +1 -1
  31. package/dist/{control-CcBiAEnn.d.ts → control-DbcDxouY.d.ts} +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DDfv22MQ.d.ts} +2 -1
  34. package/dist/{gepa-dne9JDPL.d.ts → gepa-CQelRtuC.d.ts} +10 -8
  35. package/dist/hosted/index.d.ts +8 -4
  36. package/dist/{index-BTEpx9He.d.ts → index-DbCXJfZ1.d.ts} +2 -2
  37. package/dist/index.d.ts +27 -30
  38. package/dist/index.js +28 -22
  39. package/dist/index.js.map +1 -1
  40. package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-oMVxDTxl.d.ts} +1 -1
  41. package/dist/{integrity-qemeBAyx.d.ts → integrity-C6PZ73iC.d.ts} +1 -1
  42. package/dist/kind-factory-DWOvXjR_.d.ts +171 -0
  43. package/dist/meta-eval/index.d.ts +2 -2
  44. package/dist/multishot/index.d.ts +6 -2
  45. package/dist/openapi.json +1 -1
  46. package/dist/policy-edit-Clb2v6Oa.d.ts +708 -0
  47. package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration--vU0mMtD.d.ts} +4 -4
  48. package/dist/{provenance-Bibyg1U9.d.ts → provenance-BbVagC68.d.ts} +26 -14
  49. package/dist/{release-report-CCtzajxP.d.ts → release-report-CamNDe90.d.ts} +2 -2
  50. package/dist/reporting.d.ts +4 -4
  51. package/dist/{researcher-Dq-EtpbE.d.ts → researcher-Dwbo_Fxx.d.ts} +5 -5
  52. package/dist/rl.d.ts +11 -9
  53. package/dist/rl.js +2 -2
  54. package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-BIdf9h4R.d.ts} +1 -1
  55. package/dist/{run-record-B7RTi_ix.d.ts → run-record-CZmcpWPo.d.ts} +1 -1
  56. package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-CC0jx9ql.d.ts} +1 -1
  57. package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CKjePUMh.d.ts} +3 -3
  58. package/dist/{store-C1YxJDEK.d.ts → store-9cAScOcb.d.ts} +132 -1
  59. package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-DTNgQycC.d.ts} +1 -1
  60. package/dist/traces.d.ts +6 -8
  61. package/dist/{types-C5gJrOVT.d.ts → types-Ca_63YSD.d.ts} +59 -2
  62. package/dist/wire/index.js +2 -2
  63. package/docs/design/loop-taxonomy.md +1 -2
  64. package/package.json +1 -1
  65. package/dist/chunk-5NVBGKPH.js.map +0 -1
  66. package/dist/chunk-AN5UYSVD.js +0 -761
  67. package/dist/chunk-AN5UYSVD.js.map +0 -1
  68. package/dist/chunk-DWLIGZBX.js.map +0 -1
  69. package/dist/chunk-FUCQVFMU.js.map +0 -1
  70. package/dist/chunk-N6MTC3GK.js.map +0 -1
  71. package/dist/kind-factory-DcNg13sZ.d.ts +0 -508
  72. package/dist/llm-client-DyqEH4jH.d.ts +0 -265
  73. package/dist/policy-edit-RLn8GWof.d.ts +0 -103
  74. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  75. /package/dist/{chunk-DRPIZQIT.js.map → chunk-4D5RVB3W.js.map} +0 -0
  76. /package/dist/{chunk-LVTGFSHF.js.map → chunk-7GKEAIAD.js.map} +0 -0
  77. /package/dist/{chunk-I2HNIE6N.js.map → chunk-NBSS5NDZ.js.map} +0 -0
  78. /package/dist/{chunk-QG5F6463.js.map → chunk-ONM6PEAE.js.map} +0 -0
@@ -1,508 +0,0 @@
1
- import { AxAIService, AxFunction } from '@ax-llm/ax';
2
- import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
3
- import { z } from 'zod';
4
- import { R as RunRecord } from './run-record-B7RTi_ix.js';
5
- import { a as JudgeInput } from './types-C7DGg5ex.js';
6
- import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-DyqEH4jH.js';
7
-
8
- /**
9
- * ChatClient — the single LLM abstraction analysts call.
10
- *
11
- * agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
12
- * graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
13
- * mixed patterns force every analyst author to pick a transport, which
14
- * couples analyst code to runtime concerns (cli-bridge vs router vs
15
- * sandbox-sdk) it shouldn't know about.
16
- *
17
- * `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
18
- * The operator decides at the registry boundary which transport binds
19
- * to it. Analyst code stays transport-agnostic; swapping production
20
- * (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
21
- * line factory call.
22
- *
23
- * Designed to coexist: existing `LlmClient` callers and existing
24
- * `TCloud`-based judges keep working untouched. New analyst code uses
25
- * `ChatClient`. When old call sites migrate, they pick up budgeting,
26
- * cancellation, and unified telemetry for free.
27
- */
28
-
29
- /**
30
- * Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
31
- * compatible mental model stays. Two methods: a one-shot `chat()` and
32
- * an `streamChat()` for future agentic loops (not yet exposed).
33
- */
34
- interface ChatClient {
35
- /** Display name of the bound transport — included in telemetry. */
36
- readonly transport: ChatTransport;
37
- /** Default model when caller omits — operators bind this per environment. */
38
- readonly defaultModel?: string;
39
- chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
40
- }
41
- type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
42
- interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
43
- /** Optional — falls back to ChatClient.defaultModel. */
44
- model?: string;
45
- }
46
- type ChatResponse = LlmCallResult;
47
- interface ChatCallOpts {
48
- /** Cancel the in-flight request. */
49
- signal?: AbortSignal;
50
- /** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
51
- maxCostUsd?: number;
52
- /** Correlation tag carried into request headers when the transport allows. */
53
- correlationId?: string;
54
- }
55
- type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
56
- interface BaseTransportOpts {
57
- defaultModel?: string;
58
- }
59
- interface RouterTransportOpts extends BaseTransportOpts {
60
- transport: 'router';
61
- baseUrl?: string;
62
- apiKey: string;
63
- }
64
- interface CliBridgeTransportOpts extends BaseTransportOpts {
65
- transport: 'cli-bridge';
66
- baseUrl?: string;
67
- bearer?: string;
68
- }
69
- interface DirectProviderTransportOpts extends BaseTransportOpts {
70
- transport: 'direct-provider';
71
- baseUrl: string;
72
- apiKey: string;
73
- }
74
- /**
75
- * Sandbox-SDK transport. Provided as a thin pass-through: the caller
76
- * supplies a callable that mimics LlmClient.chat() against an already-
77
- * configured Sandbox handle. We don't import the SDK here to keep
78
- * agent-eval dep-free of @tangle-network/sandbox.
79
- */
80
- interface SandboxSdkTransportOpts extends BaseTransportOpts {
81
- transport: 'sandbox-sdk';
82
- chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
83
- }
84
- /**
85
- * Mock transport for tests. The handler receives the request and returns
86
- * whatever the test wants. No retries, no JSON-schema degrade.
87
- */
88
- interface MockTransportOpts extends BaseTransportOpts {
89
- transport: 'mock';
90
- handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
91
- }
92
- /**
93
- * Build a ChatClient bound to a specific transport. The returned client
94
- * is safe to share across analysts in a single registry run.
95
- */
96
- declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
97
-
98
- /**
99
- * Typed Ax output for analyst findings.
100
- *
101
- * Replaces the legacy `findings:string[]` pattern (where every bullet
102
- * became a flat-severity `AnalystFinding`) with a structured object
103
- * array. Ax binds the field as `findings:json[]` so the provider emits
104
- * native structured output; at the kind-factory boundary we Zod-validate
105
- * each emitted finding so malformed rows fail loud instead of being
106
- * silently lifted with default severity.
107
- *
108
- * Why not `f.object().array()` directly in the signature? The Ax
109
- * signature string `question:string -> findings:json[]` already lets
110
- * the provider emit JSON arrays. A Zod boundary is required either
111
- * way (the provider can return any JSON), and Zod gives us a single
112
- * validation surface independent of which Ax version is installed.
113
- */
114
-
115
- declare const ANALYST_SEVERITIES: readonly ["critical", "high", "medium", "low", "info"];
116
- declare const RawAnalystFindingSchema: z.ZodObject<{
117
- severity: z.ZodEnum<{
118
- info: "info";
119
- critical: "critical";
120
- medium: "medium";
121
- low: "low";
122
- high: "high";
123
- }>;
124
- claim: z.ZodString;
125
- subject: z.ZodOptional<z.ZodString>;
126
- evidence_uri: z.ZodString;
127
- evidence_excerpt: z.ZodOptional<z.ZodString>;
128
- confidence: z.ZodNumber;
129
- rationale: z.ZodOptional<z.ZodString>;
130
- recommended_action: z.ZodOptional<z.ZodString>;
131
- }, z.core.$strict>;
132
- type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
133
- /**
134
- * Description embedded into the actor prompt so the LLM knows what
135
- * shape to emit. Kept here so kinds share one source of truth rather
136
- * than restating the schema in every prompt.
137
- */
138
- declare const RAW_FINDING_SCHEMA_PROMPT = "Each finding MUST be a JSON object with these fields:\n - severity: one of \"critical\" | \"high\" | \"medium\" | \"low\" | \"info\"\n - claim: one-sentence statement (max 2000 chars)\n - subject?: the routing locus this finding is about. It MUST be one of the exact subject forms listed in this kind's instructions above (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`, `tool-doc:<tool>`). A free phrase, a bare noun, or any form not in that list is REJECTED at parse time and the finding is discarded \u2014 omit subject entirely rather than guess a form.\n - evidence_uri: REQUIRED, never blank. Exactly one of \"span://<trace_id>/<span_id>\" (trace evidence), \"artifact://<relative-path>\" (files), \"metric://<name>\" (named scalars) \u2014 ALWAYS cite a real id surfaced by the tools. If you have no citable id, do not emit the finding.\n - evidence_excerpt?: short quote (<=2000 chars) from the cited span/artifact\n - confidence: number 0..1 \u2014 0.9+ when backed by exact quotes, 0.6-0.8 for inferred patterns, <0.5 for speculative\n - rationale?: one or two sentences explaining the reasoning\n - recommended_action?: concrete change phrased as an imperative (\"Add ...\", \"Replace ...\", \"Stop ...\") \u2014 omit when the finding is purely descriptive\n\nEmit an empty array when the question has no findings to report. Do not fabricate evidence.";
139
- /**
140
- * Validate one row emitted by the LLM. Returns the typed finding on
141
- * success; returns `null` and logs the reason on failure so the kind
142
- * factory can skip-and-count rather than abort the whole analyst run.
143
- */
144
- declare function parseRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): RawAnalystFinding | null;
145
-
146
- /**
147
- * Analyst contract — the missing orchestration layer over agent-eval's
148
- * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
149
- * SemanticConceptJudge, JudgeFn, ...).
150
- *
151
- * Each existing primitive returns its own output shape. The Analyst
152
- * contract is the single envelope every primitive lifts into, so a
153
- * registry can run N analysts against a run and a single renderer can
154
- * compose findings without knowing which analyzer produced them.
155
- *
156
- * The contract is intentionally domain-agnostic: nothing here knows
157
- * about code, voice, RAG, or any particular agent stack. Analysts
158
- * declare what INPUT KIND they need (a trace store, an artifact dir,
159
- * a RunRecord, a JudgeInput, or `custom`), and the registry routes
160
- * the matching input from `AnalystRunInputs`.
161
- */
162
-
163
- /**
164
- * Unified envelope every analyst emits. Schema-versioned so renderers
165
- * and time-series diffs survive future field additions.
166
- */
167
- interface AnalystFinding {
168
- schema_version: '1.0.0';
169
- /**
170
- * Stable hash over identity-defining fields (analyst_id + canonical
171
- * claim + area + optional subject). Two findings from two runs that
172
- * "are the same finding" share this id — that's what `diffFindings`
173
- * uses to compute appeared/disappeared sets across runs.
174
- */
175
- finding_id: string;
176
- analyst_id: string;
177
- produced_at: string;
178
- severity: AnalystSeverity;
179
- /**
180
- * Coarse classification. Renderers group by this. Free-form so
181
- * domain-specific analysts can introduce categories without a
182
- * schema change ('agent-reasoning', 'verification', 'cost',
183
- * 'tool-use', 'safety', 'latency', 'data-quality', ...).
184
- */
185
- area: string;
186
- claim: string;
187
- rationale?: string;
188
- evidence_refs: EvidenceRef[];
189
- recommended_action?: string;
190
- validation_plan?: string;
191
- /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
192
- confidence: number;
193
- /**
194
- * Optional subject the finding is about — leaf id, agent id, request
195
- * id. Included in finding_id when present so per-subject findings
196
- * diff cleanly across runs.
197
- */
198
- subject?: string;
199
- /** FIREWALL provenance (docs/learning-flywheel.md): true iff this finding was
200
- * lifted from a JUDGE verdict (an acceptance score), not OBSERVED from the
201
- * agent's behavior. A judge-derived finding must NEVER be admitted as a
202
- * steering input — that is the held-out judge leaking into the loop. Set at
203
- * the lift site (createJudgeAdapter); checked by `assertNoJudgeVerdict`.
204
- * Provenance, not evidence presence, is the correct discriminator: an
205
- * evidence-less trace-analyst observation legitimately steers, while a judge
206
- * verdict that happens to cite an artifact must not. */
207
- derived_from_judge?: boolean;
208
- /** Analyst-private extras; renderers ignore unless they know the analyst. */
209
- metadata?: Record<string, unknown>;
210
- }
211
- type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
212
- interface EvidenceRef {
213
- /**
214
- * Where the evidence lives. `span` and `event` refer to OTLP trace
215
- * elements; `artifact` to a file inside the run's artifact tree;
216
- * `finding` to another AnalystFinding (cross-analyst chaining);
217
- * `metric` to a named scalar reading the renderer knows how to read.
218
- */
219
- kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
220
- uri: string;
221
- excerpt?: string;
222
- }
223
- /**
224
- * The discriminator the registry uses to pass the right input.
225
- * `custom` is the escape hatch — analysts that need something else
226
- * (e.g. an embedding cache, a partner SDK handle) read it from
227
- * `AnalystRunInputs.custom[<analyst id>]`.
228
- */
229
- type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
230
- interface AnalystCost {
231
- /** `deterministic` analysts MUST NOT call the LLM. */
232
- kind: 'deterministic' | 'llm';
233
- /** Optional declared upper bound; the registry can enforce a budget. */
234
- est_usd_per_run?: number;
235
- /** Models the analyst expects to use (informational). */
236
- models?: string[];
237
- }
238
- interface AnalystRequirements {
239
- /** Min number of shots / samples the analyst needs to produce signal. */
240
- min_shots?: number;
241
- /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
242
- capabilities?: string[];
243
- }
244
- /**
245
- * What's passed to every analyst call. The registry resolves which
246
- * field the analyst's `inputKind` selects and asserts it's present.
247
- */
248
- interface AnalystRunInputs {
249
- traceStore?: TraceAnalysisStore;
250
- artifactDir?: string;
251
- runRecord?: RunRecord;
252
- judgeInput?: JudgeInput;
253
- /** Keyed by analyst id; populated by callers that registered custom analysts. */
254
- custom?: Record<string, unknown>;
255
- }
256
- interface AnalystContext {
257
- runId: string;
258
- /** Stable correlation id so logs from a single registry.run() share a tag. */
259
- correlationId: string;
260
- /** Wall-clock deadline (epoch ms). Analysts SHOULD honor for graceful cancel. */
261
- deadlineMs?: number;
262
- /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
263
- budgetUsd?: number;
264
- /**
265
- * Shared chat client. Analysts that call an LLM go through this so
266
- * the operator picks transport (sandbox-sdk | router | cli-bridge |
267
- * direct-provider | mock) at the registry boundary without touching
268
- * analyst code.
269
- */
270
- chat?: ChatClient;
271
- /**
272
- * Findings from a prior run the operator wants the analyst to see as
273
- * retrieval context. Kinds that take advantage of cross-run memory
274
- * (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
275
- * page I asked for is still missing") render these into the actor's
276
- * working set. Filtering is the operator's job: pass the slice that
277
- * matches the analyst's id, or pass everything and let the kind
278
- * filter. Empty / absent means no cross-run context.
279
- */
280
- priorFindings?: ReadonlyArray<AnalystFinding>;
281
- /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
282
- tags?: Record<string, string>;
283
- /** Logger callback — analysts SHOULD prefer this over console.* for testability. */
284
- log?: (msg: string, fields?: Record<string, unknown>) => void;
285
- /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
286
- signal?: AbortSignal;
287
- }
288
- /**
289
- * The minimal contract. Concrete analysts can refine `TInput` so
290
- * implementations stay type-safe (e.g. a trace analyst's `TInput` is
291
- * `TraceAnalysisStore`); the registry passes the right field from
292
- * `AnalystRunInputs` based on `inputKind`.
293
- */
294
- interface Analyst<TInput = unknown> {
295
- /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
296
- readonly id: string;
297
- /** Human-readable. One sentence. */
298
- readonly description: string;
299
- readonly inputKind: AnalystInputKind;
300
- readonly cost: AnalystCost;
301
- readonly requires?: AnalystRequirements;
302
- /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
303
- readonly version: string;
304
- analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
305
- }
306
- /**
307
- * Compute the stable finding_id from the identity-defining fields.
308
- * Default implementation hashes {analyst_id, area, subject, normalized claim}.
309
- * Analysts that emit findings whose claim text varies per run (timestamps,
310
- * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
311
- * or (b) move the variable part into `rationale`/`metadata` and keep the
312
- * `claim` static.
313
- */
314
- declare function computeFindingId(input: {
315
- analyst_id: string;
316
- area: string;
317
- subject?: string;
318
- claim: string;
319
- /** Override the claim for hashing — use when the displayed claim has run-specific bits. */
320
- id_basis?: string;
321
- }): string;
322
- /**
323
- * Convenience factory: produce a fully-formed AnalystFinding with the
324
- * id computed automatically. Analyst code stays terse.
325
- */
326
- declare function makeFinding(init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {
327
- id_basis?: string;
328
- produced_at?: string;
329
- }): AnalystFinding;
330
- interface AnalystRunSummary {
331
- analyst_id: string;
332
- status: 'ok' | 'skipped' | 'failed';
333
- /** Why skipped — missing input, budget exceeded, capability unmet. */
334
- reason?: string;
335
- findings_count: number;
336
- latency_ms: number;
337
- cost_usd: number;
338
- /** When `status='failed'`: the error class + message, never the full stack. */
339
- error?: {
340
- class: string;
341
- message: string;
342
- };
343
- }
344
- interface AnalystRunResult {
345
- run_id: string;
346
- correlation_id: string;
347
- started_at: string;
348
- ended_at: string;
349
- findings: AnalystFinding[];
350
- per_analyst: AnalystRunSummary[];
351
- /** Total LLM cost in USD across all analysts in this registry.run(). */
352
- total_cost_usd: number;
353
- }
354
- /**
355
- * Events emitted by `AnalystRegistry.runStream(...)` in real time as
356
- * the registry executes. UIs subscribe via `for await (const ev of
357
- * registry.runStream(...))`; `registry.run(...)` is a thin collector
358
- * over the same stream, so the two surfaces share their invariants.
359
- *
360
- * Per-finding events are intentionally omitted — analyzers are batch
361
- * operations (an Ax actor returns the full `findings:json[]` at the
362
- * end of the responder), so streaming inside one analyst would only
363
- * emit partial JSON consumers can't render. The kind-completion event
364
- * is the right granularity; subscribers wanting per-finding rendering
365
- * iterate `event.findings` themselves.
366
- */
367
- type AnalystRunEvent = {
368
- type: 'run-started';
369
- run_id: string;
370
- correlation_id: string;
371
- started_at: string;
372
- /** The ordered list of analyst ids the registry will run. */
373
- analyst_ids: ReadonlyArray<string>;
374
- } | {
375
- type: 'analyst-skipped';
376
- summary: AnalystRunSummary;
377
- } | {
378
- type: 'analyst-started';
379
- analyst_id: string;
380
- started_at: string;
381
- } | {
382
- type: 'analyst-completed';
383
- /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
384
- summary: AnalystRunSummary;
385
- findings: ReadonlyArray<AnalystFinding>;
386
- } | {
387
- type: 'run-completed';
388
- result: AnalystRunResult;
389
- };
390
-
391
- /**
392
- * Analyst-kind factory — the typed way to define trace analysts.
393
- *
394
- * A "kind" is a specialized analyst whose actor prompt, tool subset,
395
- * and Ax recursion config target one failure-mode lens (failure-mode
396
- * classification, knowledge gap discovery, knowledge poisoning, recursive
397
- * self-improvement, ...). Kinds emit findings in the typed `RawAnalystFinding`
398
- * shape via a JSON-array Ax output; the factory validates each row with
399
- * Zod and lifts it into `AnalystFinding[]` with no shape guessing.
400
- *
401
- * Composition rules:
402
- * - Each kind owns its actor description. No generic "answer this
403
- * question" prompt — the prompt names the failure lens.
404
- * - Each kind picks a narrow tool subset from `ANALYST_TOOL_GROUPS`.
405
- * A kind that never needs full-trace dumps can drop `viewTrace` /
406
- * `viewSpans` and stay cheap.
407
- * - Each kind declares its recursion + parallelism budget. Discovery-
408
- * heavy kinds (failure-mode) get higher `maxDepth`; lens kinds
409
- * (poisoning) usually stay at 0 since they have a tighter brief.
410
- *
411
- * Optimizer hook: kinds may declare `goldens` — labeled examples used
412
- * by `AxMiPRO` / `AxBootstrapFewShot` / `AxGEPA` to fit the actor
413
- * description programmatically. Stored on the kind, not the registry,
414
- * because the right metric is kind-specific.
415
- */
416
-
417
- /**
418
- * Per-kind specification. The factory turns this into a regular
419
- * `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
420
- */
421
- interface TraceAnalystKindSpec {
422
- /** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
423
- id: string;
424
- /** One-sentence description shown in `registry.list()`. */
425
- description: string;
426
- /** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
427
- area: string;
428
- /** Bump on any breaking change to the actor prompt or output schema. */
429
- version: string;
430
- /** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
431
- actorDescription: string;
432
- /** Responder system prompt; falls back to a minimal "format the findings" instruction. */
433
- responderDescription?: string;
434
- /** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
435
- buildTools: (store: TraceAnalysisStore) => AxFunction[];
436
- /** Recursion budget. `maxDepth: 0` disables subagents. */
437
- recursion?: {
438
- maxDepth: number;
439
- maxParallelSubagents?: number;
440
- };
441
- /** Actor turn cap. Default 12. */
442
- maxTurns?: number;
443
- /** Runtime char cap. Default 6000. */
444
- maxRuntimeChars?: number;
445
- /** Cost classification surfaced in `registry.list()` and budget enforcement. */
446
- cost: AnalystCost;
447
- /** Per-finding-row hook — kinds may reject / rewrite before lifting. */
448
- postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
449
- /** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
450
- goldens?: TraceAnalystGolden[];
451
- }
452
- /**
453
- * One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
454
- * Each input is the same `{question}` an analyst would receive; `expected`
455
- * is the ground-truth finding set a fitted prompt should produce on this
456
- * input. Metric: kind-specific (default: F1 on `finding_id` overlap).
457
- */
458
- interface TraceAnalystGolden {
459
- question: string;
460
- expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
461
- }
462
- interface CreateTraceAnalystKindOpts {
463
- /** AxAIService bound at registration time. */
464
- ai: AxAIService;
465
- /** Optional model override; falls back to the AI service's default. */
466
- model?: string;
467
- /** Override the spec's `version` (e.g. when an optimizer has fitted a new prompt). */
468
- versionSuffix?: string;
469
- /**
470
- * Optional two-phase recovery: when the agentic harvest is empty but the
471
- * actor produced a substantive free-form `report`, extract findings from that
472
- * prose via a tolerant chat-completions pass (`structureFindings`) — no
473
- * strict-emission contract, so it works on weak models. Omit to leave the
474
- * actor's harvest as-is (the report is still surfaced fail-loud either way).
475
- */
476
- recovery?: {
477
- baseUrl: string;
478
- apiKey?: string;
479
- model?: string;
480
- fetchImpl?: typeof fetch;
481
- };
482
- }
483
- /**
484
- * Build an `Analyst<TraceAnalysisStore>` from a kind spec.
485
- *
486
- * Lifts the Ax pipeline once at registration time so the registry
487
- * gets a stateless analyst. The Ax agent is freshly constructed per
488
- * `analyze()` call (the agent carries chat-log + usage state we don't
489
- * want shared across analyst runs).
490
- */
491
- declare function createTraceAnalystKind(spec: TraceAnalystKindSpec, opts: CreateTraceAnalystKindOpts): Analyst<TraceAnalysisStore>;
492
- /**
493
- * Render a compact prior-findings block the actor reads alongside its
494
- * brief. Each row is one line so the actor can scan dozens cheaply.
495
- * The kind's prompt instructs the actor to (a) check whether a new
496
- * cluster matches a prior `finding_id` (carry the id forward via
497
- * `id_basis` to keep diffs stable) and (b) raise severity / confidence
498
- * when a prior finding has reappeared without remediation.
499
- *
500
- * Returns the empty string when there are no prior findings — most
501
- * runs are "first-of-its-kind" and the prompt stays unchanged.
502
- *
503
- * Exported for tests + for consumers that build their own actor
504
- * prompts (e.g. specialized analysts living outside the default kinds).
505
- */
506
- declare function renderPriorFindings(prior: AnalystContext['priorFindings']): string;
507
-
508
- export { type AnalystFinding as A, parseRawFinding as B, type ChatClient as C, type DirectProviderTransportOpts as D, type EvidenceRef as E, renderPriorFindings as F, type MockTransportOpts as M, RAW_FINDING_SCHEMA_PROMPT as R, type SandboxSdkTransportOpts as S, type TraceAnalystKindSpec as T, type Analyst as a, type AnalystContext as b, type AnalystRunSummary as c, type AnalystRunResult as d, type AnalystRunInputs as e, type AnalystRunEvent as f, type AnalystSeverity as g, ANALYST_SEVERITIES as h, type AnalystCost as i, type AnalystInputKind as j, type AnalystRequirements as k, type ChatCallOpts as l, type ChatRequest as m, type ChatResponse as n, type ChatTransport as o, type CliBridgeTransportOpts as p, type CreateChatClientOpts as q, type CreateTraceAnalystKindOpts as r, type RawAnalystFinding as s, RawAnalystFindingSchema as t, type RouterTransportOpts as u, type TraceAnalystGolden as v, computeFindingId as w, createChatClient as x, createTraceAnalystKind as y, makeFinding as z };