@tangle-network/agent-eval 0.115.3 → 0.116.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/dist/analyst/index.d.ts +8 -10
  3. package/dist/analyst/index.js +27 -22
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyst-C8HHvfJp.d.ts → analyst-CFBc14Wc.d.ts} +1 -1
  6. package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs-0rz_m29H.d.ts} +3 -3
  7. package/dist/belief-state/index.d.ts +3 -3
  8. package/dist/benchmarks/index.d.ts +7 -3
  9. package/dist/benchmarks/index.js +4 -4
  10. package/dist/campaign/index.d.ts +212 -23
  11. package/dist/campaign/index.js +19 -4
  12. package/dist/{chunk-ADYLPOSX.js → chunk-3274WNK7.js} +427 -93
  13. package/dist/chunk-3274WNK7.js.map +1 -0
  14. package/dist/{chunk-I6LVHOV3.js → chunk-7GKEAIAD.js} +2 -2
  15. package/dist/{chunk-5S5NJ63F.js → chunk-CIUOICJT.js} +747 -2
  16. package/dist/chunk-CIUOICJT.js.map +1 -0
  17. package/dist/{chunk-KG4TD7EQ.js → chunk-GSW3OBHK.js} +1283 -181
  18. package/dist/chunk-GSW3OBHK.js.map +1 -0
  19. package/dist/chunk-MPHTT5HE.js +74 -0
  20. package/dist/chunk-MPHTT5HE.js.map +1 -0
  21. package/dist/{chunk-WSBUZMBU.js → chunk-NBSS5NDZ.js} +3 -3
  22. package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CdxteG0y.d.ts} +1 -1
  23. package/dist/contract/index.d.ts +19 -19
  24. package/dist/contract/index.js +5 -3
  25. package/dist/contract/index.js.map +1 -1
  26. package/dist/{control-CcBiAEnn.d.ts → control-DbcDxouY.d.ts} +1 -1
  27. package/dist/control.d.ts +2 -2
  28. package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DDfv22MQ.d.ts} +2 -1
  29. package/dist/{gepa-dne9JDPL.d.ts → gepa-CQelRtuC.d.ts} +10 -8
  30. package/dist/hosted/index.d.ts +8 -4
  31. package/dist/{index-BTEpx9He.d.ts → index-DbCXJfZ1.d.ts} +2 -2
  32. package/dist/index.d.ts +27 -30
  33. package/dist/index.js +26 -20
  34. package/dist/index.js.map +1 -1
  35. package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-oMVxDTxl.d.ts} +1 -1
  36. package/dist/{integrity-qemeBAyx.d.ts → integrity-C6PZ73iC.d.ts} +1 -1
  37. package/dist/kind-factory-DWOvXjR_.d.ts +171 -0
  38. package/dist/meta-eval/index.d.ts +2 -2
  39. package/dist/multishot/index.d.ts +6 -2
  40. package/dist/openapi.json +1 -1
  41. package/dist/policy-edit-Clb2v6Oa.d.ts +708 -0
  42. package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration--vU0mMtD.d.ts} +4 -4
  43. package/dist/{provenance-Bibyg1U9.d.ts → provenance-BbVagC68.d.ts} +26 -14
  44. package/dist/{release-report-CCtzajxP.d.ts → release-report-CamNDe90.d.ts} +2 -2
  45. package/dist/reporting.d.ts +4 -4
  46. package/dist/{researcher-Dq-EtpbE.d.ts → researcher-Dwbo_Fxx.d.ts} +5 -5
  47. package/dist/rl.d.ts +11 -9
  48. package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-BIdf9h4R.d.ts} +1 -1
  49. package/dist/{run-record-B7RTi_ix.d.ts → run-record-CZmcpWPo.d.ts} +1 -1
  50. package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-CC0jx9ql.d.ts} +1 -1
  51. package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CKjePUMh.d.ts} +3 -3
  52. package/dist/{store-C1YxJDEK.d.ts → store-9cAScOcb.d.ts} +132 -1
  53. package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-DTNgQycC.d.ts} +1 -1
  54. package/dist/traces.d.ts +6 -8
  55. package/dist/{types-C5gJrOVT.d.ts → types-Ca_63YSD.d.ts} +59 -2
  56. package/docs/design/loop-taxonomy.md +1 -2
  57. package/package.json +1 -1
  58. package/dist/chunk-5S5NJ63F.js.map +0 -1
  59. package/dist/chunk-ADYLPOSX.js.map +0 -1
  60. package/dist/chunk-KG4TD7EQ.js.map +0 -1
  61. package/dist/chunk-QMXXSNC4.js +0 -761
  62. package/dist/chunk-QMXXSNC4.js.map +0 -1
  63. package/dist/kind-factory-DcNg13sZ.d.ts +0 -508
  64. package/dist/llm-client-DyqEH4jH.d.ts +0 -265
  65. package/dist/policy-edit-RLn8GWof.d.ts +0 -103
  66. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  67. /package/dist/{chunk-I6LVHOV3.js.map → chunk-7GKEAIAD.js.map} +0 -0
  68. /package/dist/{chunk-WSBUZMBU.js.map → chunk-NBSS5NDZ.js.map} +0 -0
@@ -1,265 +0,0 @@
1
- import { A as AgentEvalError, C as CaptureIntegrityError } from './errors-oeQrLqXC.js';
2
- import { R as RawProviderSink, P as ProviderRedactor } from './raw-provider-sink-C46HDghv.js';
3
-
4
- /**
5
- * LLM client with graceful degrade.
6
- *
7
- * OpenAI-compatible `/v1/chat/completions` client with:
8
- * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
9
- * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
10
- * - Graceful json_schema → json_object degrade on 400 with schema-reject body.
11
- * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
12
- * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
13
- * directly, cli-bridge subscriptions, and any router that speaks the spec.
14
- *
15
- * Usage:
16
- * const { value, result } = await callLlmJson<MyType>(
17
- * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
18
- * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
19
- * )
20
- *
21
- * This is THE llm-calling seam for agent-eval primitives that need structured
22
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
23
- * that need free-form text use `callLlm` and parse output themselves.
24
- */
25
-
26
- interface LlmMessage {
27
- role: 'system' | 'user' | 'assistant';
28
- /**
29
- * Either a plain text content string OR a multimodal content array
30
- * (text + image_url parts) for vision-capable models.
31
- */
32
- content: string | Array<{
33
- type: 'text';
34
- text: string;
35
- } | {
36
- type: 'image_url';
37
- image_url: {
38
- url: string;
39
- detail?: 'auto' | 'low' | 'high';
40
- };
41
- }>;
42
- }
43
- interface LlmCallRequest {
44
- model: string;
45
- messages: LlmMessage[];
46
- /** Optional JSON-mode response format (response_format: json_object). */
47
- jsonMode?: boolean;
48
- /** Optional structured output via JSON Schema. Falls back to json_object on 400. */
49
- jsonSchema?: {
50
- name: string;
51
- schema: Record<string, unknown>;
52
- };
53
- temperature?: number;
54
- maxTokens?: number;
55
- /** Per-call timeout, default 300s. */
56
- timeoutMs?: number;
57
- }
58
- interface LlmUsage {
59
- promptTokens: number;
60
- completionTokens: number;
61
- totalTokens: number;
62
- /** Proxies populate this when prompt caching is on. */
63
- cachedPromptTokens?: number;
64
- }
65
- interface LlmCallResult {
66
- /** The text content of the first choice. Empty string if none. */
67
- content: string;
68
- usage: LlmUsage;
69
- /**
70
- * Cost in USD. Pulled from proxy's `_response_cost` field when present;
71
- * `null` when neither the proxy nor the caller can derive it.
72
- */
73
- costUsd: number | null;
74
- /** Model name actually used (echoed from response). */
75
- model: string;
76
- /** Wall-clock duration of the HTTP call (last attempt, if retried). */
77
- durationMs: number;
78
- /**
79
- * `finish_reason` echoed from the first choice (`stop`, `length`,
80
- * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
81
- * Exposed so a free-form `callLlm` caller CAN detect a truncated answer
82
- * (`length`) instead of treating a cut-off completion as complete. Note:
83
- * `callLlm` does not itself reject on it — acting on this signal is the
84
- * caller's responsibility (in-repo free-form drivers do not yet enforce it).
85
- */
86
- finishReason?: string | null;
87
- /**
88
- * True when `content.trim()` is empty. An empty completion is a silent zero
89
- * for free-form `callLlm` callers; this flag is the signal a caller can
90
- * inspect to fail loud rather than proceed on an empty string. `callLlm`
91
- * surfaces it but does not throw on it.
92
- */
93
- contentEmpty?: boolean;
94
- /** Raw response body. */
95
- raw: Record<string, unknown>;
96
- }
97
- declare class LlmCallError extends AgentEvalError {
98
- readonly status: number;
99
- readonly body: string;
100
- readonly model: string;
101
- constructor(message: string, status: number, body: string, model: string);
102
- }
103
- interface LlmClientOptions {
104
- /** Base URL (without trailing slash). Must end at the `/v1` prefix. */
105
- baseUrl?: string;
106
- /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
107
- apiKey?: string;
108
- bearer?: string;
109
- /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
110
- authHeader?: {
111
- name: string;
112
- value: string;
113
- };
114
- /** Default timeout in ms. Per-call can override. */
115
- defaultTimeoutMs?: number;
116
- /**
117
- * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
118
- * each attempt's per-attempt timeout controller, so aborting it cancels
119
- * the in-flight fetch. A caller abort is FATAL: it is not retried even
120
- * though an AbortError otherwise matches the transient patterns.
121
- */
122
- signal?: AbortSignal;
123
- /**
124
- * Cross-attempt wall-clock budget in ms, measured from the first attempt.
125
- * Before launching each attempt the loop checks the remaining budget and
126
- * stops retrying once it is exhausted, rather than waiting the full
127
- * per-attempt timeout on every retry. Bounds total time independent of
128
- * `maxRetries` × `timeoutMs`.
129
- */
130
- deadlineMs?: number;
131
- /** Max retry attempts on retriable errors. Default 3 (1 initial + 2 retries). */
132
- maxRetries?: number;
133
- /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
134
- fetch?: typeof fetch;
135
- /**
136
- * Optional raw HTTP capture sink. When provided, every request, response,
137
- * and error (across all retry attempts) is recorded to the sink, with auth
138
- * headers and credential-shaped body fields redacted by default. This is
139
- * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
140
- * raw events record what actually crossed the wire.
141
- */
142
- rawSink?: RawProviderSink;
143
- /**
144
- * Logical provider id attached to raw events. When omitted, derived from
145
- * `baseUrl` via `providerFromBaseUrl`.
146
- */
147
- provider?: string;
148
- /** Trace context attached to raw events; populated by emitter-aware callers. */
149
- traceContext?: {
150
- runId?: string;
151
- spanId?: string;
152
- };
153
- /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
154
- redactor?: ProviderRedactor;
155
- }
156
- /**
157
- * True when an error is a transient transport/network fault worth retrying,
158
- * as opposed to a deterministic failure (4xx schema reject, JSON parse) that
159
- * a retry cannot fix. Inspects `LlmCallError.status`, then the error's
160
- * name/message/code, then recurses into `error.cause` — undici nests the
161
- * real socket fault one or more levels under `.cause`.
162
- *
163
- * This is THE retry classifier for the package: `callLlm` and
164
- * `withJudgeRetry` both route through it, so a connection-class error is
165
- * treated identically whether it surfaces in the HTTP client or a
166
- * TCloud-backed judge.
167
- */
168
- declare function isTransientLlmError(err: unknown): boolean;
169
- /** Exponential backoff: 500ms, 1s, 2s, 4s, ... capped at 16s. Attempt is 0-indexed. */
170
- declare function backoffMs(attempt: number): number;
171
- /**
172
- * Strip a ```json / ``` code fence if the model emitted one.
173
- * Idempotent for naked JSON. Some models (claude-code via router, certain
174
- * deepseek models) wrap output even under json_object.
175
- */
176
- declare function stripFencedJson(raw: string): string;
177
- /**
178
- * Low-level call. Returns raw content + usage + cost. Retries on transient
179
- * failures; does NOT degrade schema here — callers that want graceful
180
- * degrade use `callLlmJson`.
181
- */
182
- declare function callLlm(req: LlmCallRequest, opts?: LlmClientOptions): Promise<LlmCallResult>;
183
- /**
184
- * Structured-output call. Returns parsed JSON plus the raw result envelope.
185
- * Degrades `jsonSchema` → `jsonMode` on a 400 that names the schema param —
186
- * critical for deepseek-v3/v4, kimi-k2.6, and other models that don't accept
187
- * the `response_format.json_schema` shape but DO accept `json_object`.
188
- */
189
- declare function callLlmJson<T = unknown>(req: LlmCallRequest, opts?: LlmClientOptions): Promise<{
190
- value: T;
191
- result: LlmCallResult;
192
- }>;
193
- type LlmRouteAssertionReason = 'no_explicit_base_url' | 'base_url_blocked' | 'base_url_not_allowed' | 'no_auth' | 'wrong_provider';
194
- declare class LlmRouteAssertionError extends CaptureIntegrityError {
195
- readonly reason: LlmRouteAssertionReason;
196
- readonly baseUrl: string;
197
- constructor(message: string, reason: LlmRouteAssertionReason, baseUrl: string);
198
- }
199
- interface LlmRouteRequirements {
200
- /**
201
- * Throw if `opts.baseUrl` is undefined, i.e. the call would fall back to
202
- * `DEFAULT_BASE_URL`. Set this for evaluation runs where silently using
203
- * the public/free-tier router is a defect — the launch reviewer needs to
204
- * know exactly which provider answered.
205
- */
206
- requireExplicitBaseUrl?: boolean;
207
- /**
208
- * Allowlist of acceptable base URLs. Strings match by prefix
209
- * (case-insensitive); RegExps test against the full base URL.
210
- */
211
- allowedBaseUrls?: Array<string | RegExp>;
212
- /** Blocklist that takes precedence over `allowedBaseUrls`. */
213
- blockedBaseUrls?: Array<string | RegExp>;
214
- /** Throw if no auth header / api key is configured. */
215
- requireAuth?: boolean;
216
- /**
217
- * Logical provider id the configured `baseUrl` is expected to match (via
218
- * `providerFromBaseUrl`). Mainly useful when paired with `requireExplicitBaseUrl`.
219
- */
220
- expectedProvider?: string;
221
- }
222
- /**
223
- * Fail-loud assertion that the configured LLM client points at the route
224
- * the caller intends. Designed for the matrix-runner preflight: invoke
225
- * once before any LLM call to catch misconfiguration before a sweep burns
226
- * dollars on the wrong provider.
227
- *
228
- * Throws `LlmRouteAssertionError`. Pure — no I/O — so it's safe to call
229
- * from constructors and CI gates.
230
- */
231
- declare function assertLlmRoute(opts: LlmClientOptions, req?: LlmRouteRequirements): void;
232
- /**
233
- * Probe whether a model is reachable. Returns latency + null error on
234
- * success; `ok=false` + error message on any failure (HTTP, timeout,
235
- * network, parse). Designed for sweep preflights — fail loud at the
236
- * boundary before burning a 30-leaf run on a misconfigured router.
237
- *
238
- * Sends a tiny `ping` message with `maxTokens=64`. Reasoning models
239
- * (glm-5.1, deepseek-v4) can burn the entire budget on internal reasoning
240
- * for short prompts, so don't tighten this further. We don't validate
241
- * content; HTTP 200 means reachable.
242
- */
243
- declare function probeLlm(model: string, opts?: LlmClientOptions & {
244
- timeoutMs?: number;
245
- }): Promise<{
246
- ok: boolean;
247
- latencyMs: number;
248
- error: string | null;
249
- }>;
250
- /**
251
- * Stateful client — construct once with defaults, call many times.
252
- * Thin wrapper around the free functions; exists for callers that want
253
- * to inject a single configured instance into multiple primitives.
254
- */
255
- declare class LlmClient {
256
- private readonly opts;
257
- constructor(opts?: LlmClientOptions);
258
- call(req: LlmCallRequest, per?: LlmClientOptions): Promise<LlmCallResult>;
259
- callJson<T = unknown>(req: LlmCallRequest, per?: LlmClientOptions): Promise<{
260
- value: T;
261
- result: LlmCallResult;
262
- }>;
263
- }
264
-
265
- export { type LlmClientOptions as L, type LlmRouteRequirements as a, type LlmCallRequest as b, type LlmCallResult as c, LlmCallError as d, LlmClient as e, type LlmMessage as f, LlmRouteAssertionError as g, type LlmUsage as h, assertLlmRoute as i, backoffMs as j, callLlm as k, callLlmJson as l, isTransientLlmError as m, probeLlm as p, stripFencedJson as s };
@@ -1,103 +0,0 @@
1
- import { A as AgentProfileCell, a as AgentProfileJson } from './run-record-B7RTi_ix.js';
2
- import { V as ValidationError } from './errors-oeQrLqXC.js';
3
- import { A as AnalystFinding, E as EvidenceRef } from './kind-factory-DcNg13sZ.js';
4
-
5
- type PolicyEditSchemaVersion = 'policy-edit/v1';
6
- declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
7
- type PolicyEditAxis = (typeof POLICY_EDIT_AXES)[number];
8
- declare const POLICY_EDIT_TARGET_SURFACES: readonly ["prompt", "tool-contract", "runtime-config", "memory", "agent-profile", "code", "deployment"];
9
- type PolicyEditTargetSurface = (typeof POLICY_EDIT_TARGET_SURFACES)[number];
10
- type PolicyEditRisk = 'low' | 'medium' | 'high' | 'unknown';
11
- type PolicyEditGainDirection = 'increase' | 'decrease';
12
- type PolicyEditGainUnit = 'absolute' | 'relative' | 'percent' | 'score';
13
- interface PolicyEditTarget {
14
- surface: PolicyEditTargetSurface;
15
- /** Stable path inside the target surface, for example `system-prompt:tools`
16
- * or `budget.maxTurns`. */
17
- path?: string;
18
- /** Optional canonical deployment identity. Store the existing cell, not a
19
- * local profile shape. */
20
- agentProfileCell?: AgentProfileCell;
21
- /** Human label when the path is not enough for a readable audit trail. */
22
- label?: string;
23
- }
24
- type PolicyEditChange = {
25
- kind: 'text';
26
- mode: 'append' | 'prepend' | 'replace';
27
- value: string;
28
- /** Required when `mode === 'replace'`; exact match only. */
29
- find?: string;
30
- } | {
31
- kind: 'json';
32
- mode: 'set' | 'merge' | 'remove';
33
- path: string;
34
- value?: AgentProfileJson;
35
- };
36
- interface PolicyEditExpectedGain {
37
- /** Metric this edit is expected to move, e.g. `holdout.composite`. */
38
- metric: string;
39
- direction: PolicyEditGainDirection;
40
- /** Positive magnitude in the metric's native units. */
41
- amount: number;
42
- unit?: PolicyEditGainUnit;
43
- rationale?: string;
44
- }
45
- interface PolicyEditSource {
46
- findingIds: string[];
47
- analystIds: string[];
48
- evidenceRefs: EvidenceRef[];
49
- /** Mirrors `AnalystFinding.derived_from_judge`; admission rejects it. */
50
- derivedFromJudge?: boolean;
51
- }
52
- interface PolicyEdit {
53
- schemaVersion: PolicyEditSchemaVersion;
54
- editId: string;
55
- axis: PolicyEditAxis;
56
- target: PolicyEditTarget;
57
- change: PolicyEditChange;
58
- claim: string;
59
- expectedGain: PolicyEditExpectedGain;
60
- confidence: number;
61
- risk: PolicyEditRisk;
62
- source: PolicyEditSource;
63
- rationale?: string;
64
- validationPlan?: string;
65
- metadata?: Record<string, unknown>;
66
- }
67
- type PolicyEditInit = Omit<PolicyEdit, 'schemaVersion' | 'editId'> & {
68
- schemaVersion?: PolicyEditSchemaVersion;
69
- editId?: string;
70
- };
71
- declare class PolicyEditValidationError extends ValidationError {
72
- readonly path: string;
73
- constructor(message: string, path?: string);
74
- }
75
- interface FindingToPolicyEditOptions {
76
- expectedGain?: PolicyEditExpectedGain | ((finding: AnalystFinding) => PolicyEditExpectedGain | null | undefined);
77
- risk?: PolicyEditRisk | ((finding: AnalystFinding) => PolicyEditRisk);
78
- defaultAxis?: PolicyEditAxis;
79
- defaultTargetSurface?: PolicyEditTargetSurface;
80
- }
81
- interface PolicyEditAdmissionOptions {
82
- minScore?: number;
83
- minExpectedGain?: number;
84
- allowHighRisk?: boolean;
85
- requireEvidence?: boolean;
86
- }
87
- interface PolicyEditAdmission {
88
- edit: PolicyEdit;
89
- decision: 'admit' | 'reject';
90
- score: number;
91
- reasons: string[];
92
- }
93
- declare function makePolicyEdit(init: PolicyEditInit): PolicyEdit;
94
- declare function computePolicyEditId(edit: Omit<PolicyEdit, 'editId'> | PolicyEdit): string;
95
- declare function validatePolicyEdit(input: unknown): PolicyEdit;
96
- declare function isPolicyEdit(input: unknown): input is PolicyEdit;
97
- declare function policyEditsFromFindings(findings: ReadonlyArray<AnalystFinding>, opts?: FindingToPolicyEditOptions): PolicyEdit[];
98
- declare function policyEditFromFinding(finding: AnalystFinding, opts?: FindingToPolicyEditOptions): PolicyEdit | null;
99
- declare function scorePolicyEditReadiness(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): number;
100
- declare function admitPolicyEdit(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): PolicyEditAdmission;
101
- declare function applyPolicyEditToSurface(surface: unknown, edit: PolicyEdit): unknown;
102
-
103
- export { type FindingToPolicyEditOptions as F, POLICY_EDIT_AXES as P, POLICY_EDIT_TARGET_SURFACES as a, type PolicyEdit as b, type PolicyEditAdmission as c, type PolicyEditAdmissionOptions as d, type PolicyEditAxis as e, type PolicyEditChange as f, type PolicyEditExpectedGain as g, type PolicyEditGainDirection as h, type PolicyEditGainUnit as i, type PolicyEditInit as j, type PolicyEditRisk as k, type PolicyEditSchemaVersion as l, type PolicyEditSource as m, type PolicyEditTarget as n, type PolicyEditTargetSurface as o, PolicyEditValidationError as p, admitPolicyEdit as q, applyPolicyEditToSurface as r, computePolicyEditId as s, isPolicyEdit as t, makePolicyEdit as u, policyEditFromFinding as v, policyEditsFromFindings as w, scorePolicyEditReadiness as x, validatePolicyEdit as y };
@@ -1,132 +0,0 @@
1
- /**
2
- * RawProviderSink — first-class persistence for the actual HTTP-level
3
- * request/response bodies of every LLM provider call.
4
- *
5
- * Why this is a separate sink from the structured `LlmSpan`:
6
- *
7
- * - `LlmSpan` records the *intent* — model name, messages, output text,
8
- * usage. It's what dashboards read; it's NOT enough for forensics.
9
- * - When a downstream consumer reports "the verifier used the wrong route"
10
- * or "tokens look right but reasoning was missing," the only way to
11
- * answer is the raw HTTP body. Span fields can lie (a proxy can echo
12
- * a different `model` value than what actually answered); the raw
13
- * response is ground truth.
14
- *
15
- * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
16
- * matrix runner / BuilderSession sets it up automatically) and every
17
- * request, response, and error is recorded — including retries, with the
18
- * attempt index attached so a flaky call's full event chain is recoverable.
19
- *
20
- * Redaction is enforced at sink time. The default redactor strips
21
- * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
22
- * payload field whose key matches `apiKey | api_key | bearer | password |
23
- * secret | token` (case-insensitive). Override via the sink constructor or
24
- * the per-call `redactor`. The `redactedFields` array on the persisted
25
- * event lets a reviewer see what was stripped without exposing the values.
26
- */
27
- type RawProviderDirection = 'request' | 'response' | 'error';
28
- interface RawProviderEvent {
29
- /** Stable id. Generated by the sink if omitted. */
30
- eventId: string;
31
- /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
32
- runId?: string;
33
- spanId?: string;
34
- /**
35
- * Logical provider name. Free-form so callers can use whatever id matches
36
- * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
37
- * omitted, derived from `baseUrl` in `LlmClientOptions`.
38
- */
39
- provider: string;
40
- model: string;
41
- /** Endpoint path, e.g. `'/v1/chat/completions'`. */
42
- endpoint: string;
43
- /** Base URL used for the call (already-normalised — no trailing slash). */
44
- baseUrl: string;
45
- /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
46
- attemptIndex: number;
47
- direction: RawProviderDirection;
48
- /** Unix ms. */
49
- timestamp: number;
50
- /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
51
- durationMs?: number;
52
- statusCode?: number;
53
- requestHeaders?: Record<string, string>;
54
- requestBody?: unknown;
55
- responseHeaders?: Record<string, string>;
56
- responseBody?: unknown;
57
- /** Set on `direction: 'error'` events. */
58
- errorMessage?: string;
59
- /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
60
- redactedFields: string[];
61
- }
62
- interface RawProviderSinkFilter {
63
- runId?: string;
64
- spanId?: string;
65
- direction?: RawProviderDirection;
66
- attemptIndex?: number;
67
- }
68
- interface RawProviderSink {
69
- record(event: RawProviderEvent): Promise<void>;
70
- /** Optional listing — implementations that durably persist (file, db) should support this. */
71
- list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
72
- /** Optional teardown for backed implementations. */
73
- close?(): Promise<void>;
74
- }
75
- type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
76
- /**
77
- * Default redactor — strips well-known auth headers and any body field whose
78
- * key matches the credential pattern. Records every redacted path on
79
- * `event.redactedFields` so a downstream reviewer can see what was removed.
80
- */
81
- declare function defaultProviderRedactor(event: RawProviderEvent): RawProviderEvent;
82
- interface InMemoryRawProviderSinkOptions {
83
- redactor?: ProviderRedactor;
84
- }
85
- declare class InMemoryRawProviderSink implements RawProviderSink {
86
- private events;
87
- private redactor;
88
- constructor(opts?: InMemoryRawProviderSinkOptions);
89
- record(event: RawProviderEvent): Promise<void>;
90
- list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
91
- size(): number;
92
- }
93
- declare class NoopRawProviderSink implements RawProviderSink {
94
- record(): Promise<void>;
95
- /**
96
- * Returns an empty array. Implemented so `assertRunCaptured` does not
97
- * trip the `no_raw_sink` issue when a caller explicitly opts out of
98
- * capture by passing this sink — opt-out is a deliberate choice, not a
99
- * misconfiguration.
100
- */
101
- list(): Promise<RawProviderEvent[]>;
102
- }
103
- interface FileSystemRawProviderSinkOptions {
104
- /** Directory the NDJSON file is written into. Created if missing. */
105
- dir: string;
106
- /** File name; default `'raw-provider-events.ndjson'`. */
107
- fileName?: string;
108
- /** Bytes after which the writer rolls over to a new file (default 32 MiB). */
109
- rollAtBytes?: number;
110
- redactor?: ProviderRedactor;
111
- }
112
- declare class FileSystemRawProviderSink implements RawProviderSink {
113
- private dir;
114
- private fileName;
115
- private rollAtBytes;
116
- private redactor;
117
- private bytesWritten;
118
- private rollIndex;
119
- private initPromise;
120
- constructor(opts: FileSystemRawProviderSinkOptions);
121
- private ensureInit;
122
- private currentPath;
123
- record(event: RawProviderEvent): Promise<void>;
124
- list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
125
- }
126
- /**
127
- * Best-effort provider id from a base URL. Falls back to the URL host when
128
- * none of the well-known patterns match.
129
- */
130
- declare function providerFromBaseUrl(baseUrl: string): string;
131
-
132
- export { FileSystemRawProviderSink as F, InMemoryRawProviderSink as I, NoopRawProviderSink as N, type ProviderRedactor as P, type RawProviderSink as R, type FileSystemRawProviderSinkOptions as a, type InMemoryRawProviderSinkOptions as b, type RawProviderDirection as c, type RawProviderEvent as d, type RawProviderSinkFilter as e, defaultProviderRedactor as f, providerFromBaseUrl as p };