@tangle-network/agent-eval 0.115.2 → 0.116.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +34 -0
- package/dist/analyst/index.d.ts +8 -10
- package/dist/analyst/index.js +28 -23
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyst-C8HHvfJp.d.ts → analyst-CFBc14Wc.d.ts} +1 -1
- package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs-0rz_m29H.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +3 -3
- package/dist/benchmarks/index.d.ts +7 -3
- package/dist/benchmarks/index.js +5 -5
- package/dist/campaign/index.d.ts +212 -23
- package/dist/campaign/index.js +20 -5
- package/dist/{chunk-N6MTC3GK.js → chunk-3274WNK7.js} +428 -94
- package/dist/chunk-3274WNK7.js.map +1 -0
- package/dist/{chunk-DRPIZQIT.js → chunk-4D5RVB3W.js} +2 -2
- package/dist/{chunk-LVTGFSHF.js → chunk-7GKEAIAD.js} +2 -2
- package/dist/{chunk-DWLIGZBX.js → chunk-CIUOICJT.js} +748 -3
- package/dist/chunk-CIUOICJT.js.map +1 -0
- package/dist/{chunk-5NVBGKPH.js → chunk-GSW3OBHK.js} +1284 -182
- package/dist/chunk-GSW3OBHK.js.map +1 -0
- package/dist/{chunk-FUCQVFMU.js → chunk-GY4SYVPJ.js} +12 -3
- package/dist/chunk-GY4SYVPJ.js.map +1 -0
- package/dist/chunk-MPHTT5HE.js +74 -0
- package/dist/chunk-MPHTT5HE.js.map +1 -0
- package/dist/{chunk-I2HNIE6N.js → chunk-NBSS5NDZ.js} +4 -4
- package/dist/{chunk-QG5F6463.js → chunk-ONM6PEAE.js} +2 -2
- package/dist/cli.js +2 -2
- package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CdxteG0y.d.ts} +1 -1
- package/dist/contract/index.d.ts +19 -19
- package/dist/contract/index.js +6 -4
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CcBiAEnn.d.ts → control-DbcDxouY.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DDfv22MQ.d.ts} +2 -1
- package/dist/{gepa-dne9JDPL.d.ts → gepa-CQelRtuC.d.ts} +10 -8
- package/dist/hosted/index.d.ts +8 -4
- package/dist/{index-BTEpx9He.d.ts → index-DbCXJfZ1.d.ts} +2 -2
- package/dist/index.d.ts +27 -30
- package/dist/index.js +28 -22
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-oMVxDTxl.d.ts} +1 -1
- package/dist/{integrity-qemeBAyx.d.ts → integrity-C6PZ73iC.d.ts} +1 -1
- package/dist/kind-factory-DWOvXjR_.d.ts +171 -0
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +6 -2
- package/dist/openapi.json +1 -1
- package/dist/policy-edit-Clb2v6Oa.d.ts +708 -0
- package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration--vU0mMtD.d.ts} +4 -4
- package/dist/{provenance-Bibyg1U9.d.ts → provenance-BbVagC68.d.ts} +26 -14
- package/dist/{release-report-CCtzajxP.d.ts → release-report-CamNDe90.d.ts} +2 -2
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-Dq-EtpbE.d.ts → researcher-Dwbo_Fxx.d.ts} +5 -5
- package/dist/rl.d.ts +11 -9
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-BIdf9h4R.d.ts} +1 -1
- package/dist/{run-record-B7RTi_ix.d.ts → run-record-CZmcpWPo.d.ts} +1 -1
- package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-CC0jx9ql.d.ts} +1 -1
- package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CKjePUMh.d.ts} +3 -3
- package/dist/{store-C1YxJDEK.d.ts → store-9cAScOcb.d.ts} +132 -1
- package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-DTNgQycC.d.ts} +1 -1
- package/dist/traces.d.ts +6 -8
- package/dist/{types-C5gJrOVT.d.ts → types-Ca_63YSD.d.ts} +59 -2
- package/dist/wire/index.js +2 -2
- package/docs/design/loop-taxonomy.md +1 -2
- package/package.json +1 -1
- package/dist/chunk-5NVBGKPH.js.map +0 -1
- package/dist/chunk-AN5UYSVD.js +0 -761
- package/dist/chunk-AN5UYSVD.js.map +0 -1
- package/dist/chunk-DWLIGZBX.js.map +0 -1
- package/dist/chunk-FUCQVFMU.js.map +0 -1
- package/dist/chunk-N6MTC3GK.js.map +0 -1
- package/dist/kind-factory-DcNg13sZ.d.ts +0 -508
- package/dist/llm-client-DyqEH4jH.d.ts +0 -265
- package/dist/policy-edit-RLn8GWof.d.ts +0 -103
- package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
- /package/dist/{chunk-DRPIZQIT.js.map → chunk-4D5RVB3W.js.map} +0 -0
- /package/dist/{chunk-LVTGFSHF.js.map → chunk-7GKEAIAD.js.map} +0 -0
- /package/dist/{chunk-I2HNIE6N.js.map → chunk-NBSS5NDZ.js.map} +0 -0
- /package/dist/{chunk-QG5F6463.js.map → chunk-ONM6PEAE.js.map} +0 -0
|
@@ -0,0 +1,708 @@
|
|
|
1
|
+
import { R as RunRecord, A as AgentProfileCell, f as AgentProfileJson } from './run-record-CZmcpWPo.js';
|
|
2
|
+
import { A as AgentEvalError, C as CaptureIntegrityError, V as ValidationError } from './errors-oeQrLqXC.js';
|
|
3
|
+
import { R as RawProviderSink, P as ProviderRedactor, T as TraceAnalysisStore } from './store-9cAScOcb.js';
|
|
4
|
+
import { a as JudgeInput } from './types-C7DGg5ex.js';
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* LLM client with graceful degrade.
|
|
8
|
+
*
|
|
9
|
+
* OpenAI-compatible `/v1/chat/completions` client with:
|
|
10
|
+
* - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
|
|
11
|
+
* - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
|
|
12
|
+
* - Graceful json_schema → json_object degrade on 400 with schema-reject body.
|
|
13
|
+
* - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
|
|
14
|
+
* - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
|
|
15
|
+
* directly, cli-bridge subscriptions, and any router that speaks the spec.
|
|
16
|
+
*
|
|
17
|
+
* Usage:
|
|
18
|
+
* const { value, result } = await callLlmJson<MyType>(
|
|
19
|
+
* { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
|
|
20
|
+
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
21
|
+
* )
|
|
22
|
+
*
|
|
23
|
+
* This is THE llm-calling seam for agent-eval primitives that need structured
|
|
24
|
+
* output (semantic concept judge, reviewer directives, critic scores). Primitives
|
|
25
|
+
* that need free-form text use `callLlm` and parse output themselves.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
interface LlmMessage {
|
|
29
|
+
role: 'system' | 'user' | 'assistant';
|
|
30
|
+
/**
|
|
31
|
+
* Either a plain text content string OR a multimodal content array
|
|
32
|
+
* (text + image_url parts) for vision-capable models.
|
|
33
|
+
*/
|
|
34
|
+
content: string | Array<{
|
|
35
|
+
type: 'text';
|
|
36
|
+
text: string;
|
|
37
|
+
} | {
|
|
38
|
+
type: 'image_url';
|
|
39
|
+
image_url: {
|
|
40
|
+
url: string;
|
|
41
|
+
detail?: 'auto' | 'low' | 'high';
|
|
42
|
+
};
|
|
43
|
+
}>;
|
|
44
|
+
}
|
|
45
|
+
interface LlmCallRequest {
|
|
46
|
+
model: string;
|
|
47
|
+
messages: LlmMessage[];
|
|
48
|
+
/** Optional JSON-mode response format (response_format: json_object). */
|
|
49
|
+
jsonMode?: boolean;
|
|
50
|
+
/** Optional structured output via JSON Schema. Falls back to json_object on 400. */
|
|
51
|
+
jsonSchema?: {
|
|
52
|
+
name: string;
|
|
53
|
+
schema: Record<string, unknown>;
|
|
54
|
+
};
|
|
55
|
+
temperature?: number;
|
|
56
|
+
maxTokens?: number;
|
|
57
|
+
/** Per-call timeout, default 300s. */
|
|
58
|
+
timeoutMs?: number;
|
|
59
|
+
}
|
|
60
|
+
interface LlmUsage {
|
|
61
|
+
promptTokens: number;
|
|
62
|
+
completionTokens: number;
|
|
63
|
+
totalTokens: number;
|
|
64
|
+
/** Proxies populate this when prompt caching is on. */
|
|
65
|
+
cachedPromptTokens?: number;
|
|
66
|
+
}
|
|
67
|
+
interface LlmCallResult {
|
|
68
|
+
/** The text content of the first choice. Empty string if none. */
|
|
69
|
+
content: string;
|
|
70
|
+
usage: LlmUsage;
|
|
71
|
+
/**
|
|
72
|
+
* Cost in USD. Pulled from proxy's `_response_cost` field when present;
|
|
73
|
+
* `null` when neither the proxy nor the caller can derive it.
|
|
74
|
+
*/
|
|
75
|
+
costUsd: number | null;
|
|
76
|
+
/** Model name actually used (echoed from response). */
|
|
77
|
+
model: string;
|
|
78
|
+
/** Wall-clock duration of the HTTP call (last attempt, if retried). */
|
|
79
|
+
durationMs: number;
|
|
80
|
+
/**
|
|
81
|
+
* `finish_reason` echoed from the first choice (`stop`, `length`,
|
|
82
|
+
* `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
|
|
83
|
+
* Exposed so a free-form `callLlm` caller CAN detect a truncated answer
|
|
84
|
+
* (`length`) instead of treating a cut-off completion as complete. Note:
|
|
85
|
+
* `callLlm` does not itself reject on it — acting on this signal is the
|
|
86
|
+
* caller's responsibility (in-repo free-form drivers do not yet enforce it).
|
|
87
|
+
*/
|
|
88
|
+
finishReason?: string | null;
|
|
89
|
+
/**
|
|
90
|
+
* True when `content.trim()` is empty. An empty completion is a silent zero
|
|
91
|
+
* for free-form `callLlm` callers; this flag is the signal a caller can
|
|
92
|
+
* inspect to fail loud rather than proceed on an empty string. `callLlm`
|
|
93
|
+
* surfaces it but does not throw on it.
|
|
94
|
+
*/
|
|
95
|
+
contentEmpty?: boolean;
|
|
96
|
+
/** Raw response body. */
|
|
97
|
+
raw: Record<string, unknown>;
|
|
98
|
+
}
|
|
99
|
+
declare class LlmCallError extends AgentEvalError {
|
|
100
|
+
readonly status: number;
|
|
101
|
+
readonly body: string;
|
|
102
|
+
readonly model: string;
|
|
103
|
+
constructor(message: string, status: number, body: string, model: string);
|
|
104
|
+
}
|
|
105
|
+
interface LlmClientOptions {
|
|
106
|
+
/** Base URL (without trailing slash). Must end at the `/v1` prefix. */
|
|
107
|
+
baseUrl?: string;
|
|
108
|
+
/** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
|
|
109
|
+
apiKey?: string;
|
|
110
|
+
bearer?: string;
|
|
111
|
+
/** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
|
|
112
|
+
authHeader?: {
|
|
113
|
+
name: string;
|
|
114
|
+
value: string;
|
|
115
|
+
};
|
|
116
|
+
/** Default timeout in ms. Per-call can override. */
|
|
117
|
+
defaultTimeoutMs?: number;
|
|
118
|
+
/**
|
|
119
|
+
* Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
|
|
120
|
+
* each attempt's per-attempt timeout controller, so aborting it cancels
|
|
121
|
+
* the in-flight fetch. A caller abort is FATAL: it is not retried even
|
|
122
|
+
* though an AbortError otherwise matches the transient patterns.
|
|
123
|
+
*/
|
|
124
|
+
signal?: AbortSignal;
|
|
125
|
+
/**
|
|
126
|
+
* Cross-attempt wall-clock budget in ms, measured from the first attempt.
|
|
127
|
+
* Before launching each attempt the loop checks the remaining budget and
|
|
128
|
+
* stops retrying once it is exhausted, rather than waiting the full
|
|
129
|
+
* per-attempt timeout on every retry. Bounds total time independent of
|
|
130
|
+
* `maxRetries` × `timeoutMs`.
|
|
131
|
+
*/
|
|
132
|
+
deadlineMs?: number;
|
|
133
|
+
/** Max retry attempts on retriable errors. Default 3 (1 initial + 2 retries). */
|
|
134
|
+
maxRetries?: number;
|
|
135
|
+
/** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
|
|
136
|
+
fetch?: typeof fetch;
|
|
137
|
+
/**
|
|
138
|
+
* Optional raw HTTP capture sink. When provided, every request, response,
|
|
139
|
+
* and error (across all retry attempts) is recorded to the sink, with auth
|
|
140
|
+
* headers and credential-shaped body fields redacted by default. This is
|
|
141
|
+
* the layer-1 forensics primitive: structured `LlmSpan`s record intent,
|
|
142
|
+
* raw events record what actually crossed the wire.
|
|
143
|
+
*/
|
|
144
|
+
rawSink?: RawProviderSink;
|
|
145
|
+
/**
|
|
146
|
+
* Logical provider id attached to raw events. When omitted, derived from
|
|
147
|
+
* `baseUrl` via `providerFromBaseUrl`.
|
|
148
|
+
*/
|
|
149
|
+
provider?: string;
|
|
150
|
+
/** Trace context attached to raw events; populated by emitter-aware callers. */
|
|
151
|
+
traceContext?: {
|
|
152
|
+
runId?: string;
|
|
153
|
+
spanId?: string;
|
|
154
|
+
};
|
|
155
|
+
/** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
|
|
156
|
+
redactor?: ProviderRedactor;
|
|
157
|
+
}
|
|
158
|
+
/**
|
|
159
|
+
* True when an error is a transient transport/network fault worth retrying,
|
|
160
|
+
* as opposed to a deterministic failure (4xx schema reject, JSON parse) that
|
|
161
|
+
* a retry cannot fix. Inspects `LlmCallError.status`, then the error's
|
|
162
|
+
* name/message/code, then recurses into `error.cause` — undici nests the
|
|
163
|
+
* real socket fault one or more levels under `.cause`.
|
|
164
|
+
*
|
|
165
|
+
* This is THE retry classifier for the package: `callLlm` and
|
|
166
|
+
* `withJudgeRetry` both route through it, so a connection-class error is
|
|
167
|
+
* treated identically whether it surfaces in the HTTP client or a
|
|
168
|
+
* TCloud-backed judge.
|
|
169
|
+
*/
|
|
170
|
+
declare function isTransientLlmError(err: unknown): boolean;
|
|
171
|
+
/** Exponential backoff: 500ms, 1s, 2s, 4s, ... capped at 16s. Attempt is 0-indexed. */
|
|
172
|
+
declare function backoffMs(attempt: number): number;
|
|
173
|
+
/**
|
|
174
|
+
* Strip a ```json / ``` code fence if the model emitted one.
|
|
175
|
+
* Idempotent for naked JSON. Some models (claude-code via router, certain
|
|
176
|
+
* deepseek models) wrap output even under json_object.
|
|
177
|
+
*/
|
|
178
|
+
declare function stripFencedJson(raw: string): string;
|
|
179
|
+
/**
|
|
180
|
+
* Low-level call. Returns raw content + usage + cost. Retries on transient
|
|
181
|
+
* failures; does NOT degrade schema here — callers that want graceful
|
|
182
|
+
* degrade use `callLlmJson`.
|
|
183
|
+
*/
|
|
184
|
+
declare function callLlm(req: LlmCallRequest, opts?: LlmClientOptions): Promise<LlmCallResult>;
|
|
185
|
+
/**
|
|
186
|
+
* Structured-output call. Returns parsed JSON plus the raw result envelope.
|
|
187
|
+
* Degrades `jsonSchema` → `jsonMode` on a 400 that names the schema param —
|
|
188
|
+
* critical for deepseek-v3/v4, kimi-k2.6, and other models that don't accept
|
|
189
|
+
* the `response_format.json_schema` shape but DO accept `json_object`.
|
|
190
|
+
*/
|
|
191
|
+
declare function callLlmJson<T = unknown>(req: LlmCallRequest, opts?: LlmClientOptions): Promise<{
|
|
192
|
+
value: T;
|
|
193
|
+
result: LlmCallResult;
|
|
194
|
+
}>;
|
|
195
|
+
type LlmRouteAssertionReason = 'no_explicit_base_url' | 'base_url_blocked' | 'base_url_not_allowed' | 'no_auth' | 'wrong_provider';
|
|
196
|
+
declare class LlmRouteAssertionError extends CaptureIntegrityError {
|
|
197
|
+
readonly reason: LlmRouteAssertionReason;
|
|
198
|
+
readonly baseUrl: string;
|
|
199
|
+
constructor(message: string, reason: LlmRouteAssertionReason, baseUrl: string);
|
|
200
|
+
}
|
|
201
|
+
interface LlmRouteRequirements {
|
|
202
|
+
/**
|
|
203
|
+
* Throw if `opts.baseUrl` is undefined, i.e. the call would fall back to
|
|
204
|
+
* `DEFAULT_BASE_URL`. Set this for evaluation runs where silently using
|
|
205
|
+
* the public/free-tier router is a defect — the launch reviewer needs to
|
|
206
|
+
* know exactly which provider answered.
|
|
207
|
+
*/
|
|
208
|
+
requireExplicitBaseUrl?: boolean;
|
|
209
|
+
/**
|
|
210
|
+
* Allowlist of acceptable base URLs. Strings match by prefix
|
|
211
|
+
* (case-insensitive); RegExps test against the full base URL.
|
|
212
|
+
*/
|
|
213
|
+
allowedBaseUrls?: Array<string | RegExp>;
|
|
214
|
+
/** Blocklist that takes precedence over `allowedBaseUrls`. */
|
|
215
|
+
blockedBaseUrls?: Array<string | RegExp>;
|
|
216
|
+
/** Throw if no auth header / api key is configured. */
|
|
217
|
+
requireAuth?: boolean;
|
|
218
|
+
/**
|
|
219
|
+
* Logical provider id the configured `baseUrl` is expected to match (via
|
|
220
|
+
* `providerFromBaseUrl`). Mainly useful when paired with `requireExplicitBaseUrl`.
|
|
221
|
+
*/
|
|
222
|
+
expectedProvider?: string;
|
|
223
|
+
}
|
|
224
|
+
/**
|
|
225
|
+
* Fail-loud assertion that the configured LLM client points at the route
|
|
226
|
+
* the caller intends. Designed for the matrix-runner preflight: invoke
|
|
227
|
+
* once before any LLM call to catch misconfiguration before a sweep burns
|
|
228
|
+
* dollars on the wrong provider.
|
|
229
|
+
*
|
|
230
|
+
* Throws `LlmRouteAssertionError`. Pure — no I/O — so it's safe to call
|
|
231
|
+
* from constructors and CI gates.
|
|
232
|
+
*/
|
|
233
|
+
declare function assertLlmRoute(opts: LlmClientOptions, req?: LlmRouteRequirements): void;
|
|
234
|
+
/**
|
|
235
|
+
* Probe whether a model is reachable. Returns latency + null error on
|
|
236
|
+
* success; `ok=false` + error message on any failure (HTTP, timeout,
|
|
237
|
+
* network, parse). Designed for sweep preflights — fail loud at the
|
|
238
|
+
* boundary before burning a 30-leaf run on a misconfigured router.
|
|
239
|
+
*
|
|
240
|
+
* Sends a tiny `ping` message with `maxTokens=64`. Reasoning models
|
|
241
|
+
* (glm-5.1, deepseek-v4) can burn the entire budget on internal reasoning
|
|
242
|
+
* for short prompts, so don't tighten this further. We don't validate
|
|
243
|
+
* content; HTTP 200 means reachable.
|
|
244
|
+
*/
|
|
245
|
+
declare function probeLlm(model: string, opts?: LlmClientOptions & {
|
|
246
|
+
timeoutMs?: number;
|
|
247
|
+
}): Promise<{
|
|
248
|
+
ok: boolean;
|
|
249
|
+
latencyMs: number;
|
|
250
|
+
error: string | null;
|
|
251
|
+
}>;
|
|
252
|
+
/**
|
|
253
|
+
* Stateful client — construct once with defaults, call many times.
|
|
254
|
+
* Thin wrapper around the free functions; exists for callers that want
|
|
255
|
+
* to inject a single configured instance into multiple primitives.
|
|
256
|
+
*/
|
|
257
|
+
declare class LlmClient {
|
|
258
|
+
private readonly opts;
|
|
259
|
+
constructor(opts?: LlmClientOptions);
|
|
260
|
+
call(req: LlmCallRequest, per?: LlmClientOptions): Promise<LlmCallResult>;
|
|
261
|
+
callJson<T = unknown>(req: LlmCallRequest, per?: LlmClientOptions): Promise<{
|
|
262
|
+
value: T;
|
|
263
|
+
result: LlmCallResult;
|
|
264
|
+
}>;
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
/**
|
|
268
|
+
* ChatClient — the single LLM abstraction analysts call.
|
|
269
|
+
*
|
|
270
|
+
* agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
|
|
271
|
+
* graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
|
|
272
|
+
* mixed patterns force every analyst author to pick a transport, which
|
|
273
|
+
* couples analyst code to runtime concerns (cli-bridge vs router vs
|
|
274
|
+
* sandbox-sdk) it shouldn't know about.
|
|
275
|
+
*
|
|
276
|
+
* `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
|
|
277
|
+
* The operator decides at the registry boundary which transport binds
|
|
278
|
+
* to it. Analyst code stays transport-agnostic; swapping production
|
|
279
|
+
* (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
|
|
280
|
+
* line factory call.
|
|
281
|
+
*
|
|
282
|
+
* Designed to coexist: existing `LlmClient` callers and existing
|
|
283
|
+
* `TCloud`-based judges keep working untouched. New analyst code uses
|
|
284
|
+
* `ChatClient`. When old call sites migrate, they pick up budgeting,
|
|
285
|
+
* cancellation, and unified telemetry for free.
|
|
286
|
+
*/
|
|
287
|
+
|
|
288
|
+
/**
|
|
289
|
+
* Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
|
|
290
|
+
* compatible mental model stays. Two methods: a one-shot `chat()` and
|
|
291
|
+
* an `streamChat()` for future agentic loops (not yet exposed).
|
|
292
|
+
*/
|
|
293
|
+
interface ChatClient {
|
|
294
|
+
/** Display name of the bound transport — included in telemetry. */
|
|
295
|
+
readonly transport: ChatTransport;
|
|
296
|
+
/** Default model when caller omits — operators bind this per environment. */
|
|
297
|
+
readonly defaultModel?: string;
|
|
298
|
+
chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
|
|
299
|
+
}
|
|
300
|
+
type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
|
|
301
|
+
interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
|
|
302
|
+
/** Optional — falls back to ChatClient.defaultModel. */
|
|
303
|
+
model?: string;
|
|
304
|
+
}
|
|
305
|
+
type ChatResponse = LlmCallResult;
|
|
306
|
+
interface ChatCallOpts {
|
|
307
|
+
/** Cancel the in-flight request. */
|
|
308
|
+
signal?: AbortSignal;
|
|
309
|
+
/** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
|
|
310
|
+
maxCostUsd?: number;
|
|
311
|
+
/** Correlation tag carried into request headers when the transport allows. */
|
|
312
|
+
correlationId?: string;
|
|
313
|
+
}
|
|
314
|
+
type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
|
|
315
|
+
interface BaseTransportOpts {
|
|
316
|
+
defaultModel?: string;
|
|
317
|
+
}
|
|
318
|
+
interface RouterTransportOpts extends BaseTransportOpts {
|
|
319
|
+
transport: 'router';
|
|
320
|
+
baseUrl?: string;
|
|
321
|
+
apiKey: string;
|
|
322
|
+
}
|
|
323
|
+
interface CliBridgeTransportOpts extends BaseTransportOpts {
|
|
324
|
+
transport: 'cli-bridge';
|
|
325
|
+
baseUrl?: string;
|
|
326
|
+
bearer?: string;
|
|
327
|
+
}
|
|
328
|
+
interface DirectProviderTransportOpts extends BaseTransportOpts {
|
|
329
|
+
transport: 'direct-provider';
|
|
330
|
+
baseUrl: string;
|
|
331
|
+
apiKey: string;
|
|
332
|
+
}
|
|
333
|
+
/**
|
|
334
|
+
* Sandbox-SDK transport. Provided as a thin pass-through: the caller
|
|
335
|
+
* supplies a callable that mimics LlmClient.chat() against an already-
|
|
336
|
+
* configured Sandbox handle. We don't import the SDK here to keep
|
|
337
|
+
* agent-eval dep-free of @tangle-network/sandbox.
|
|
338
|
+
*/
|
|
339
|
+
interface SandboxSdkTransportOpts extends BaseTransportOpts {
|
|
340
|
+
transport: 'sandbox-sdk';
|
|
341
|
+
chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
342
|
+
}
|
|
343
|
+
/**
|
|
344
|
+
* Mock transport for tests. The handler receives the request and returns
|
|
345
|
+
* whatever the test wants. No retries, no JSON-schema degrade.
|
|
346
|
+
*/
|
|
347
|
+
interface MockTransportOpts extends BaseTransportOpts {
|
|
348
|
+
transport: 'mock';
|
|
349
|
+
handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
350
|
+
}
|
|
351
|
+
/**
|
|
352
|
+
* Build a ChatClient bound to a specific transport. The returned client
|
|
353
|
+
* is safe to share across analysts in a single registry run.
|
|
354
|
+
*/
|
|
355
|
+
declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
|
|
356
|
+
|
|
357
|
+
/**
|
|
358
|
+
* Analyst contract — the missing orchestration layer over agent-eval's
|
|
359
|
+
* existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
|
|
360
|
+
* SemanticConceptJudge, JudgeFn, ...).
|
|
361
|
+
*
|
|
362
|
+
* Each existing primitive returns its own output shape. The Analyst
|
|
363
|
+
* contract is the single envelope every primitive lifts into, so a
|
|
364
|
+
* registry can run N analysts against a run and a single renderer can
|
|
365
|
+
* compose findings without knowing which analyzer produced them.
|
|
366
|
+
*
|
|
367
|
+
* The contract is intentionally domain-agnostic: nothing here knows
|
|
368
|
+
* about code, voice, RAG, or any particular agent stack. Analysts
|
|
369
|
+
* declare what INPUT KIND they need (a trace store, an artifact dir,
|
|
370
|
+
* a RunRecord, a JudgeInput, or `custom`), and the registry routes
|
|
371
|
+
* the matching input from `AnalystRunInputs`.
|
|
372
|
+
*/
|
|
373
|
+
|
|
374
|
+
/**
|
|
375
|
+
* Unified envelope every analyst emits. Schema-versioned so renderers
|
|
376
|
+
* and time-series diffs survive future field additions.
|
|
377
|
+
*/
|
|
378
|
+
interface AnalystFinding {
|
|
379
|
+
schema_version: '1.0.0';
|
|
380
|
+
/**
|
|
381
|
+
* Stable hash over identity-defining fields (analyst_id + canonical
|
|
382
|
+
* claim + area + optional subject). Two findings from two runs that
|
|
383
|
+
* "are the same finding" share this id — that's what `diffFindings`
|
|
384
|
+
* uses to compute appeared/disappeared sets across runs.
|
|
385
|
+
*/
|
|
386
|
+
finding_id: string;
|
|
387
|
+
analyst_id: string;
|
|
388
|
+
produced_at: string;
|
|
389
|
+
severity: AnalystSeverity;
|
|
390
|
+
/**
|
|
391
|
+
* Coarse classification. Renderers group by this. Free-form so
|
|
392
|
+
* domain-specific analysts can introduce categories without a
|
|
393
|
+
* schema change ('agent-reasoning', 'verification', 'cost',
|
|
394
|
+
* 'tool-use', 'safety', 'latency', 'data-quality', ...).
|
|
395
|
+
*/
|
|
396
|
+
area: string;
|
|
397
|
+
claim: string;
|
|
398
|
+
rationale?: string;
|
|
399
|
+
evidence_refs: EvidenceRef[];
|
|
400
|
+
recommended_action?: string;
|
|
401
|
+
validation_plan?: string;
|
|
402
|
+
/** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
|
|
403
|
+
confidence: number;
|
|
404
|
+
/**
|
|
405
|
+
* Optional subject the finding is about — leaf id, agent id, request
|
|
406
|
+
* id. Included in finding_id when present so per-subject findings
|
|
407
|
+
* diff cleanly across runs.
|
|
408
|
+
*/
|
|
409
|
+
subject?: string;
|
|
410
|
+
/** FIREWALL provenance (docs/learning-flywheel.md): true iff this finding was
|
|
411
|
+
* lifted from a JUDGE verdict (an acceptance score), not OBSERVED from the
|
|
412
|
+
* agent's behavior. A judge-derived finding must NEVER be admitted as a
|
|
413
|
+
* steering input — that is the held-out judge leaking into the loop. Set at
|
|
414
|
+
* the lift site (createJudgeAdapter); checked by `assertNoJudgeVerdict`.
|
|
415
|
+
* Provenance, not evidence presence, is the correct discriminator: an
|
|
416
|
+
* evidence-less trace-analyst observation legitimately steers, while a judge
|
|
417
|
+
* verdict that happens to cite an artifact must not. */
|
|
418
|
+
derived_from_judge?: boolean;
|
|
419
|
+
/** Analyst-private extras; renderers ignore unless they know the analyst. */
|
|
420
|
+
metadata?: Record<string, unknown>;
|
|
421
|
+
}
|
|
422
|
+
type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
|
|
423
|
+
interface EvidenceRef {
|
|
424
|
+
/**
|
|
425
|
+
* Where the evidence lives. `span` and `event` refer to OTLP trace
|
|
426
|
+
* elements; `artifact` to a file inside the run's artifact tree;
|
|
427
|
+
* `finding` to another AnalystFinding (cross-analyst chaining);
|
|
428
|
+
* `metric` to a named scalar reading the renderer knows how to read.
|
|
429
|
+
*/
|
|
430
|
+
kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
|
|
431
|
+
uri: string;
|
|
432
|
+
excerpt?: string;
|
|
433
|
+
}
|
|
434
|
+
/**
|
|
435
|
+
* The discriminator the registry uses to pass the right input.
|
|
436
|
+
* `custom` is the escape hatch — analysts that need something else
|
|
437
|
+
* (e.g. an embedding cache, a partner SDK handle) read it from
|
|
438
|
+
* `AnalystRunInputs.custom[<analyst id>]`.
|
|
439
|
+
*/
|
|
440
|
+
type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
|
|
441
|
+
interface AnalystCost {
|
|
442
|
+
/** `deterministic` analysts MUST NOT call the LLM. */
|
|
443
|
+
kind: 'deterministic' | 'llm';
|
|
444
|
+
/** Optional declared upper bound; the registry can enforce a budget. */
|
|
445
|
+
est_usd_per_run?: number;
|
|
446
|
+
/** Models the analyst expects to use (informational). */
|
|
447
|
+
models?: string[];
|
|
448
|
+
}
|
|
449
|
+
interface AnalystRequirements {
|
|
450
|
+
/** Min number of shots / samples the analyst needs to produce signal. */
|
|
451
|
+
min_shots?: number;
|
|
452
|
+
/** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
|
|
453
|
+
capabilities?: string[];
|
|
454
|
+
}
|
|
455
|
+
/**
|
|
456
|
+
* What's passed to every analyst call. The registry resolves which
|
|
457
|
+
* field the analyst's `inputKind` selects and asserts it's present.
|
|
458
|
+
*/
|
|
459
|
+
interface AnalystRunInputs {
|
|
460
|
+
traceStore?: TraceAnalysisStore;
|
|
461
|
+
artifactDir?: string;
|
|
462
|
+
runRecord?: RunRecord;
|
|
463
|
+
judgeInput?: JudgeInput;
|
|
464
|
+
/** Keyed by analyst id; populated by callers that registered custom analysts. */
|
|
465
|
+
custom?: Record<string, unknown>;
|
|
466
|
+
}
|
|
467
|
+
interface AnalystContext {
|
|
468
|
+
runId: string;
|
|
469
|
+
/** Stable correlation id so logs from a single registry.run() share a tag. */
|
|
470
|
+
correlationId: string;
|
|
471
|
+
/** Wall-clock deadline (epoch ms). Analysts SHOULD honor for graceful cancel. */
|
|
472
|
+
deadlineMs?: number;
|
|
473
|
+
/** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
|
|
474
|
+
budgetUsd?: number;
|
|
475
|
+
/**
|
|
476
|
+
* Shared chat client. Analysts that call an LLM go through this so
|
|
477
|
+
* the operator picks transport (sandbox-sdk | router | cli-bridge |
|
|
478
|
+
* direct-provider | mock) at the registry boundary without touching
|
|
479
|
+
* analyst code.
|
|
480
|
+
*/
|
|
481
|
+
chat?: ChatClient;
|
|
482
|
+
/**
|
|
483
|
+
* Findings from a prior run the operator wants the analyst to see as
|
|
484
|
+
* retrieval context. Kinds that take advantage of cross-run memory
|
|
485
|
+
* (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
|
|
486
|
+
* page I asked for is still missing") render these into the actor's
|
|
487
|
+
* working set. Filtering is the operator's job: pass the slice that
|
|
488
|
+
* matches the analyst's id, or pass everything and let the kind
|
|
489
|
+
* filter. Empty / absent means no cross-run context.
|
|
490
|
+
*/
|
|
491
|
+
priorFindings?: ReadonlyArray<AnalystFinding>;
|
|
492
|
+
/** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
|
|
493
|
+
tags?: Record<string, string>;
|
|
494
|
+
/** Logger callback — analysts SHOULD prefer this over console.* for testability. */
|
|
495
|
+
log?: (msg: string, fields?: Record<string, unknown>) => void;
|
|
496
|
+
/** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
|
|
497
|
+
signal?: AbortSignal;
|
|
498
|
+
}
|
|
499
|
+
/**
|
|
500
|
+
* The minimal contract. Concrete analysts can refine `TInput` so
|
|
501
|
+
* implementations stay type-safe (e.g. a trace analyst's `TInput` is
|
|
502
|
+
* `TraceAnalysisStore`); the registry passes the right field from
|
|
503
|
+
* `AnalystRunInputs` based on `inputKind`.
|
|
504
|
+
*/
|
|
505
|
+
interface Analyst<TInput = unknown> {
|
|
506
|
+
/** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
|
|
507
|
+
readonly id: string;
|
|
508
|
+
/** Human-readable. One sentence. */
|
|
509
|
+
readonly description: string;
|
|
510
|
+
readonly inputKind: AnalystInputKind;
|
|
511
|
+
readonly cost: AnalystCost;
|
|
512
|
+
readonly requires?: AnalystRequirements;
|
|
513
|
+
/** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
|
|
514
|
+
readonly version: string;
|
|
515
|
+
analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
|
|
516
|
+
}
|
|
517
|
+
/**
|
|
518
|
+
* Compute the stable finding_id from the identity-defining fields.
|
|
519
|
+
* Default implementation hashes {analyst_id, area, subject, normalized claim}.
|
|
520
|
+
* Analysts that emit findings whose claim text varies per run (timestamps,
|
|
521
|
+
* counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
|
|
522
|
+
* or (b) move the variable part into `rationale`/`metadata` and keep the
|
|
523
|
+
* `claim` static.
|
|
524
|
+
*/
|
|
525
|
+
declare function computeFindingId(input: {
|
|
526
|
+
analyst_id: string;
|
|
527
|
+
area: string;
|
|
528
|
+
subject?: string;
|
|
529
|
+
claim: string;
|
|
530
|
+
/** Override the claim for hashing — use when the displayed claim has run-specific bits. */
|
|
531
|
+
id_basis?: string;
|
|
532
|
+
}): string;
|
|
533
|
+
/**
|
|
534
|
+
* Convenience factory: produce a fully-formed AnalystFinding with the
|
|
535
|
+
* id computed automatically. Analyst code stays terse.
|
|
536
|
+
*/
|
|
537
|
+
declare function makeFinding(init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {
|
|
538
|
+
id_basis?: string;
|
|
539
|
+
produced_at?: string;
|
|
540
|
+
}): AnalystFinding;
|
|
541
|
+
interface AnalystRunSummary {
|
|
542
|
+
analyst_id: string;
|
|
543
|
+
status: 'ok' | 'skipped' | 'failed';
|
|
544
|
+
/** Why skipped — missing input, budget exceeded, capability unmet. */
|
|
545
|
+
reason?: string;
|
|
546
|
+
findings_count: number;
|
|
547
|
+
latency_ms: number;
|
|
548
|
+
cost_usd: number;
|
|
549
|
+
/** When `status='failed'`: the error class + message, never the full stack. */
|
|
550
|
+
error?: {
|
|
551
|
+
class: string;
|
|
552
|
+
message: string;
|
|
553
|
+
};
|
|
554
|
+
}
|
|
555
|
+
interface AnalystRunResult {
|
|
556
|
+
run_id: string;
|
|
557
|
+
correlation_id: string;
|
|
558
|
+
started_at: string;
|
|
559
|
+
ended_at: string;
|
|
560
|
+
findings: AnalystFinding[];
|
|
561
|
+
per_analyst: AnalystRunSummary[];
|
|
562
|
+
/** Total LLM cost in USD across all analysts in this registry.run(). */
|
|
563
|
+
total_cost_usd: number;
|
|
564
|
+
}
|
|
565
|
+
/**
|
|
566
|
+
* Events emitted by `AnalystRegistry.runStream(...)` in real time as
|
|
567
|
+
* the registry executes. UIs subscribe via `for await (const ev of
|
|
568
|
+
* registry.runStream(...))`; `registry.run(...)` is a thin collector
|
|
569
|
+
* over the same stream, so the two surfaces share their invariants.
|
|
570
|
+
*
|
|
571
|
+
* Per-finding events are intentionally omitted — analyzers are batch
|
|
572
|
+
* operations (an Ax actor returns the full `findings:json[]` at the
|
|
573
|
+
* end of the responder), so streaming inside one analyst would only
|
|
574
|
+
* emit partial JSON consumers can't render. The kind-completion event
|
|
575
|
+
* is the right granularity; subscribers wanting per-finding rendering
|
|
576
|
+
* iterate `event.findings` themselves.
|
|
577
|
+
*/
|
|
578
|
+
type AnalystRunEvent = {
|
|
579
|
+
type: 'run-started';
|
|
580
|
+
run_id: string;
|
|
581
|
+
correlation_id: string;
|
|
582
|
+
started_at: string;
|
|
583
|
+
/** The ordered list of analyst ids the registry will run. */
|
|
584
|
+
analyst_ids: ReadonlyArray<string>;
|
|
585
|
+
} | {
|
|
586
|
+
type: 'analyst-skipped';
|
|
587
|
+
summary: AnalystRunSummary;
|
|
588
|
+
} | {
|
|
589
|
+
type: 'analyst-started';
|
|
590
|
+
analyst_id: string;
|
|
591
|
+
started_at: string;
|
|
592
|
+
} | {
|
|
593
|
+
type: 'analyst-completed';
|
|
594
|
+
/** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
|
|
595
|
+
summary: AnalystRunSummary;
|
|
596
|
+
findings: ReadonlyArray<AnalystFinding>;
|
|
597
|
+
} | {
|
|
598
|
+
type: 'run-completed';
|
|
599
|
+
result: AnalystRunResult;
|
|
600
|
+
};
|
|
601
|
+
|
|
602
|
+
type PolicyEditSchemaVersion = 'policy-edit/v1';
|
|
603
|
+
declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
|
|
604
|
+
type PolicyEditAxis = (typeof POLICY_EDIT_AXES)[number];
|
|
605
|
+
declare const POLICY_EDIT_TARGET_SURFACES: readonly ["prompt", "tool-contract", "runtime-config", "memory", "agent-profile", "code", "deployment"];
|
|
606
|
+
type PolicyEditTargetSurface = (typeof POLICY_EDIT_TARGET_SURFACES)[number];
|
|
607
|
+
type PolicyEditRisk = 'low' | 'medium' | 'high' | 'unknown';
|
|
608
|
+
type PolicyEditGainDirection = 'increase' | 'decrease';
|
|
609
|
+
type PolicyEditGainUnit = 'absolute' | 'relative' | 'percent' | 'score';
|
|
610
|
+
interface PolicyEditTarget {
|
|
611
|
+
surface: PolicyEditTargetSurface;
|
|
612
|
+
/** Stable path inside the target surface, for example `system-prompt:tools`
|
|
613
|
+
* or `budget.maxTurns`. */
|
|
614
|
+
path?: string;
|
|
615
|
+
/** Optional canonical deployment identity. Store the existing cell, not a
|
|
616
|
+
* local profile shape. */
|
|
617
|
+
agentProfileCell?: AgentProfileCell;
|
|
618
|
+
/** Human label when the path is not enough for a readable audit trail. */
|
|
619
|
+
label?: string;
|
|
620
|
+
}
|
|
621
|
+
type PolicyEditChange = {
|
|
622
|
+
kind: 'text';
|
|
623
|
+
mode: 'append' | 'prepend' | 'replace';
|
|
624
|
+
value: string;
|
|
625
|
+
/** Required when `mode === 'replace'`; exact match only. */
|
|
626
|
+
find?: string;
|
|
627
|
+
} | {
|
|
628
|
+
kind: 'json';
|
|
629
|
+
mode: 'set' | 'merge' | 'remove';
|
|
630
|
+
path: string;
|
|
631
|
+
value?: AgentProfileJson;
|
|
632
|
+
};
|
|
633
|
+
interface PolicyEditExpectedGain {
|
|
634
|
+
/** Metric this edit is expected to move, e.g. `holdout.composite`. */
|
|
635
|
+
metric: string;
|
|
636
|
+
direction: PolicyEditGainDirection;
|
|
637
|
+
/** Positive magnitude in the metric's native units. */
|
|
638
|
+
amount: number;
|
|
639
|
+
unit?: PolicyEditGainUnit;
|
|
640
|
+
rationale?: string;
|
|
641
|
+
}
|
|
642
|
+
interface PolicyEditSource {
|
|
643
|
+
findingIds: string[];
|
|
644
|
+
analystIds: string[];
|
|
645
|
+
evidenceRefs: EvidenceRef[];
|
|
646
|
+
/** Mirrors `AnalystFinding.derived_from_judge`; admission rejects it. */
|
|
647
|
+
derivedFromJudge?: boolean;
|
|
648
|
+
}
|
|
649
|
+
interface PolicyEdit {
|
|
650
|
+
schemaVersion: PolicyEditSchemaVersion;
|
|
651
|
+
editId: string;
|
|
652
|
+
axis: PolicyEditAxis;
|
|
653
|
+
target: PolicyEditTarget;
|
|
654
|
+
change: PolicyEditChange;
|
|
655
|
+
claim: string;
|
|
656
|
+
expectedGain: PolicyEditExpectedGain;
|
|
657
|
+
confidence: number;
|
|
658
|
+
risk: PolicyEditRisk;
|
|
659
|
+
source: PolicyEditSource;
|
|
660
|
+
rationale?: string;
|
|
661
|
+
validationPlan?: string;
|
|
662
|
+
metadata?: Record<string, unknown>;
|
|
663
|
+
}
|
|
664
|
+
declare const POLICY_EDIT_CANDIDATE_RECORD_SCHEMA: "tangle.policy-edit-candidate.v1";
|
|
665
|
+
/** JSON-safe attribution carried with a measured candidate and its scores. */
|
|
666
|
+
interface PolicyEditCandidateRecord {
|
|
667
|
+
schema: typeof POLICY_EDIT_CANDIDATE_RECORD_SCHEMA;
|
|
668
|
+
policyEdit: PolicyEdit;
|
|
669
|
+
}
|
|
670
|
+
type PolicyEditInit = Omit<PolicyEdit, 'schemaVersion' | 'editId'> & {
|
|
671
|
+
schemaVersion?: PolicyEditSchemaVersion;
|
|
672
|
+
editId?: string;
|
|
673
|
+
};
|
|
674
|
+
declare class PolicyEditValidationError extends ValidationError {
|
|
675
|
+
readonly path: string;
|
|
676
|
+
constructor(message: string, path?: string);
|
|
677
|
+
}
|
|
678
|
+
interface FindingToPolicyEditOptions {
|
|
679
|
+
expectedGain?: PolicyEditExpectedGain | ((finding: AnalystFinding) => PolicyEditExpectedGain | null | undefined);
|
|
680
|
+
risk?: PolicyEditRisk | ((finding: AnalystFinding) => PolicyEditRisk);
|
|
681
|
+
defaultAxis?: PolicyEditAxis;
|
|
682
|
+
defaultTargetSurface?: PolicyEditTargetSurface;
|
|
683
|
+
}
|
|
684
|
+
interface PolicyEditAdmissionOptions {
|
|
685
|
+
minScore?: number;
|
|
686
|
+
minExpectedGain?: number;
|
|
687
|
+
allowHighRisk?: boolean;
|
|
688
|
+
requireEvidence?: boolean;
|
|
689
|
+
}
|
|
690
|
+
interface PolicyEditAdmission {
|
|
691
|
+
edit: PolicyEdit;
|
|
692
|
+
decision: 'admit' | 'reject';
|
|
693
|
+
score: number;
|
|
694
|
+
reasons: string[];
|
|
695
|
+
}
|
|
696
|
+
declare function makePolicyEdit(init: PolicyEditInit): PolicyEdit;
|
|
697
|
+
declare function computePolicyEditId(edit: Omit<PolicyEdit, 'editId'> | PolicyEdit): string;
|
|
698
|
+
declare function validatePolicyEdit(input: unknown): PolicyEdit;
|
|
699
|
+
declare function makePolicyEditCandidateRecord(edit: PolicyEdit): PolicyEditCandidateRecord;
|
|
700
|
+
declare function validatePolicyEditCandidateRecord(input: unknown): PolicyEditCandidateRecord;
|
|
701
|
+
declare function isPolicyEdit(input: unknown): input is PolicyEdit;
|
|
702
|
+
declare function policyEditsFromFindings(findings: ReadonlyArray<AnalystFinding>, opts?: FindingToPolicyEditOptions): PolicyEdit[];
|
|
703
|
+
declare function policyEditFromFinding(finding: AnalystFinding, opts?: FindingToPolicyEditOptions): PolicyEdit | null;
|
|
704
|
+
declare function scorePolicyEditReadiness(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): number;
|
|
705
|
+
declare function admitPolicyEdit(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): PolicyEditAdmission;
|
|
706
|
+
declare function applyPolicyEditToSurface(surface: unknown, edit: PolicyEdit): unknown;
|
|
707
|
+
|
|
708
|
+
export { makePolicyEditCandidateRecord as $, type Analyst as A, type PolicyEditGainDirection as B, type ChatClient as C, type DirectProviderTransportOpts as D, type EvidenceRef as E, type FindingToPolicyEditOptions as F, type PolicyEditGainUnit as G, type PolicyEditInit as H, type PolicyEditRisk as I, type PolicyEditSchemaVersion as J, type PolicyEditSource as K, type LlmClientOptions as L, type MockTransportOpts as M, type PolicyEditTarget as N, type PolicyEditTargetSurface as O, type PolicyEditCandidateRecord as P, PolicyEditValidationError as Q, type RouterTransportOpts as R, type SandboxSdkTransportOpts as S, admitPolicyEdit as T, applyPolicyEditToSurface as U, computeFindingId as V, computePolicyEditId as W, createChatClient as X, isPolicyEdit as Y, makeFinding as Z, makePolicyEdit as _, type LlmRouteRequirements as a, policyEditFromFinding as a0, policyEditsFromFindings as a1, scorePolicyEditReadiness as a2, validatePolicyEdit as a3, validatePolicyEditCandidateRecord as a4, LlmCallError as a5, type LlmCallRequest as a6, type LlmCallResult as a7, LlmClient as a8, type LlmMessage as a9, LlmRouteAssertionError as aa, type LlmUsage as ab, assertLlmRoute as ac, backoffMs as ad, callLlm as ae, callLlmJson as af, isTransientLlmError as ag, probeLlm as ah, stripFencedJson as ai, type AnalystContext as b, type AnalystRunSummary as c, type AnalystFinding as d, type AnalystRunResult as e, type AnalystRunInputs as f, type AnalystRunEvent as g, type AnalystCost as h, type AnalystSeverity as i, type AnalystInputKind as j, type AnalystRequirements as k, type ChatCallOpts as l, type ChatRequest as m, type ChatResponse as n, type ChatTransport as o, type CliBridgeTransportOpts as p, type CreateChatClientOpts as q, POLICY_EDIT_AXES as r, POLICY_EDIT_CANDIDATE_RECORD_SCHEMA as s, POLICY_EDIT_TARGET_SURFACES as t, type PolicyEdit as u, type PolicyEditAdmission as v, type PolicyEditAdmissionOptions as w, type PolicyEditAxis as x, type PolicyEditChange as y, type PolicyEditExpectedGain as z };
|