@tangle-network/agent-eval 0.120.2 → 0.120.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/dist/analyst/index.d.ts +3111 -0
- package/dist/analyst/index.js +403 -0
- package/dist/analyst/index.js.map +1 -0
- package/dist/authenticity/index.d.ts +161 -0
- package/dist/authenticity/index.js +215 -0
- package/dist/authenticity/index.js.map +1 -0
- package/dist/belief-state/index.d.ts +1301 -0
- package/dist/belief-state/index.js +2152 -0
- package/dist/belief-state/index.js.map +1 -0
- package/dist/benchmarks/index.d.ts +974 -0
- package/dist/benchmarks/index.js +60 -0
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +695 -0
- package/dist/builder-eval/index.js +366 -0
- package/dist/builder-eval/index.js.map +1 -0
- package/dist/campaign/index.d.ts +7454 -0
- package/dist/campaign/index.js +272 -0
- package/dist/campaign/index.js.map +1 -0
- package/dist/chunk-3CDFMEMO.js +3878 -0
- package/dist/chunk-3CDFMEMO.js.map +1 -0
- package/dist/chunk-3RF76KTD.js +84 -0
- package/dist/chunk-3RF76KTD.js.map +1 -0
- package/dist/chunk-3XH4Y2SS.js +750 -0
- package/dist/chunk-3XH4Y2SS.js.map +1 -0
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/chunk-5BYTIDZ7.js +550 -0
- package/dist/chunk-5BYTIDZ7.js.map +1 -0
- package/dist/chunk-5CVUPHJ4.js +2668 -0
- package/dist/chunk-5CVUPHJ4.js.map +1 -0
- package/dist/chunk-ARU2PZFM.js +312 -0
- package/dist/chunk-ARU2PZFM.js.map +1 -0
- package/dist/chunk-BOD4O7OF.js +40 -0
- package/dist/chunk-BOD4O7OF.js.map +1 -0
- package/dist/chunk-CVJP5TMD.js +766 -0
- package/dist/chunk-CVJP5TMD.js.map +1 -0
- package/dist/chunk-DPZAEKA6.js +880 -0
- package/dist/chunk-DPZAEKA6.js.map +1 -0
- package/dist/chunk-DTJ6QUQB.js +131 -0
- package/dist/chunk-DTJ6QUQB.js.map +1 -0
- package/dist/chunk-GGE4NNQT.js +65 -0
- package/dist/chunk-GGE4NNQT.js.map +1 -0
- package/dist/chunk-H5UD2323.js +286 -0
- package/dist/chunk-H5UD2323.js.map +1 -0
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/chunk-HKUCJ437.js +787 -0
- package/dist/chunk-HKUCJ437.js.map +1 -0
- package/dist/chunk-JHCHEVET.js +274 -0
- package/dist/chunk-JHCHEVET.js.map +1 -0
- package/dist/chunk-K4DBDHLK.js +158 -0
- package/dist/chunk-K4DBDHLK.js.map +1 -0
- package/dist/chunk-K6N6XJJX.js +306 -0
- package/dist/chunk-K6N6XJJX.js.map +1 -0
- package/dist/chunk-MA6HLL3S.js +65 -0
- package/dist/chunk-MA6HLL3S.js.map +1 -0
- package/dist/chunk-MAZ26DC7.js +99 -0
- package/dist/chunk-MAZ26DC7.js.map +1 -0
- package/dist/chunk-MOXWMGPC.js +577 -0
- package/dist/chunk-MOXWMGPC.js.map +1 -0
- package/dist/chunk-NJC7U437.js +626 -0
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/chunk-NMN4WGSJ.js +1030 -0
- package/dist/chunk-NMN4WGSJ.js.map +1 -0
- package/dist/chunk-NPCTHQIO.js +91 -0
- package/dist/chunk-NPCTHQIO.js.map +1 -0
- package/dist/chunk-ONWEPEDO.js +57 -0
- package/dist/chunk-ONWEPEDO.js.map +1 -0
- package/dist/chunk-OYZAPX5G.js +1526 -0
- package/dist/chunk-OYZAPX5G.js.map +1 -0
- package/dist/chunk-P5MGQ2FY.js +7958 -0
- package/dist/chunk-P5MGQ2FY.js.map +1 -0
- package/dist/chunk-PC4UYEBM.js +166 -0
- package/dist/chunk-PC4UYEBM.js.map +1 -0
- package/dist/chunk-PJQFMIOX.js +1182 -0
- package/dist/chunk-PJQFMIOX.js.map +1 -0
- package/dist/chunk-PXD6ZFNY.js +1107 -0
- package/dist/chunk-PXD6ZFNY.js.map +1 -0
- package/dist/chunk-PXE2VKMX.js +140 -0
- package/dist/chunk-PXE2VKMX.js.map +1 -0
- package/dist/chunk-PZ5AY32C.js +10 -0
- package/dist/chunk-PZ5AY32C.js.map +1 -0
- package/dist/chunk-QBRSJK47.js +622 -0
- package/dist/chunk-QBRSJK47.js.map +1 -0
- package/dist/chunk-S3UZOQ5Y.js +328 -0
- package/dist/chunk-S3UZOQ5Y.js.map +1 -0
- package/dist/chunk-SQQED7ZH.js +998 -0
- package/dist/chunk-SQQED7ZH.js.map +1 -0
- package/dist/chunk-SYV364BL.js +1266 -0
- package/dist/chunk-SYV364BL.js.map +1 -0
- package/dist/chunk-T4SQEITX.js +95 -0
- package/dist/chunk-T4SQEITX.js.map +1 -0
- package/dist/chunk-TT4KNT67.js +124 -0
- package/dist/chunk-TT4KNT67.js.map +1 -0
- package/dist/chunk-U5CHZ5M3.js +357 -0
- package/dist/chunk-U5CHZ5M3.js.map +1 -0
- package/dist/chunk-ULOKLHIQ.js +1937 -0
- package/dist/chunk-ULOKLHIQ.js.map +1 -0
- package/dist/chunk-VI2UW6B6.js +162 -0
- package/dist/chunk-VI2UW6B6.js.map +1 -0
- package/dist/chunk-VQMK5FMP.js +247 -0
- package/dist/chunk-VQMK5FMP.js.map +1 -0
- package/dist/chunk-VSMTAMNK.js +53 -0
- package/dist/chunk-VSMTAMNK.js.map +1 -0
- package/dist/chunk-VZSRQ272.js +149 -0
- package/dist/chunk-VZSRQ272.js.map +1 -0
- package/dist/chunk-WW2A73HW.js +159 -0
- package/dist/chunk-WW2A73HW.js.map +1 -0
- package/dist/chunk-X4UCIOTZ.js +136 -0
- package/dist/chunk-X4UCIOTZ.js.map +1 -0
- package/dist/chunk-XJYR7XFV.js +317 -0
- package/dist/chunk-XJYR7XFV.js.map +1 -0
- package/dist/chunk-ZET2UAYW.js +89 -0
- package/dist/chunk-ZET2UAYW.js.map +1 -0
- package/dist/chunk-ZMXDQ4K7.js +870 -0
- package/dist/chunk-ZMXDQ4K7.js.map +1 -0
- package/dist/chunk-ZZUXHH3R.js +99 -0
- package/dist/chunk-ZZUXHH3R.js.map +1 -0
- package/dist/cli.d.ts +1 -0
- package/dist/cli.js +112 -0
- package/dist/cli.js.map +1 -0
- package/dist/contract/index.d.ts +4972 -0
- package/dist/contract/index.js +1654 -0
- package/dist/contract/index.js.map +1 -0
- package/dist/control.d.ts +1013 -0
- package/dist/control.js +34 -0
- package/dist/control.js.map +1 -0
- package/dist/fuzz.d.ts +759 -0
- package/dist/fuzz.js +714 -0
- package/dist/fuzz.js.map +1 -0
- package/dist/hosted/index.d.ts +730 -0
- package/dist/hosted/index.js +14 -0
- package/dist/hosted/index.js.map +1 -0
- package/dist/index.d.ts +16780 -0
- package/dist/index.js +12168 -0
- package/dist/index.js.map +1 -0
- package/dist/matrix/index.d.ts +155 -0
- package/dist/matrix/index.js +8 -0
- package/dist/matrix/index.js.map +1 -0
- package/dist/meta-eval/index.d.ts +1030 -0
- package/dist/meta-eval/index.js +417 -0
- package/dist/meta-eval/index.js.map +1 -0
- package/dist/multishot/index.d.ts +579 -0
- package/dist/multishot/index.js +589 -0
- package/dist/multishot/index.js.map +1 -0
- package/dist/openapi.json +992 -0
- package/dist/pipelines/index.d.ts +567 -0
- package/dist/pipelines/index.js +515 -0
- package/dist/pipelines/index.js.map +1 -0
- package/dist/reporting.d.ts +1277 -0
- package/dist/reporting.js +48 -0
- package/dist/reporting.js.map +1 -0
- package/dist/rl.d.ts +4092 -0
- package/dist/rl.js +1724 -0
- package/dist/rl.js.map +1 -0
- package/dist/run-campaign-75RTPVV5.js +14 -0
- package/dist/run-campaign-75RTPVV5.js.map +1 -0
- package/dist/storyboard/index.d.ts +279 -0
- package/dist/storyboard/index.js +767 -0
- package/dist/storyboard/index.js.map +1 -0
- package/dist/trace-attributes.d.ts +52 -0
- package/dist/trace-attributes.js +62 -0
- package/dist/trace-attributes.js.map +1 -0
- package/dist/traces.d.ts +2343 -0
- package/dist/traces.js +249 -0
- package/dist/traces.js.map +1 -0
- package/dist/wire/index.d.ts +1252 -0
- package/dist/wire/index.js +81 -0
- package/dist/wire/index.js.map +1 -0
- package/package.json +1 -1
|
@@ -0,0 +1,3111 @@
|
|
|
1
|
+
import { TCloud } from '@tangle-network/tcloud';
|
|
2
|
+
import { AxAIArgs, AxAIService, AxFunction } from '@ax-llm/ax';
|
|
3
|
+
import { z } from 'zod';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Validator-output verdict — substrate primitive for "did this output pass,
|
|
7
|
+
* and how well?"
|
|
8
|
+
*
|
|
9
|
+
* Used by:
|
|
10
|
+
* - `@tangle-network/agent-eval/matrix` — verdict per cell in the cartesian.
|
|
11
|
+
* - `@tangle-network/agent-runtime` — Validator<Output, Verdict = DefaultVerdict>.
|
|
12
|
+
* Runtime keeps `Validator` because it's coupled to runtime-shaped
|
|
13
|
+
* `ValidationCtx` (iteration, signal, traceEmitter); the verdict TYPE
|
|
14
|
+
* itself is a substrate concept and lives here.
|
|
15
|
+
*
|
|
16
|
+
* Repo layering: agent-eval is the substrate (no upward deps). Both
|
|
17
|
+
* agent-runtime and agent-knowledge consume this type FROM agent-eval —
|
|
18
|
+
* never the other way around. See CLAUDE.md "Repo layering" for the rule.
|
|
19
|
+
*/
|
|
20
|
+
/**
|
|
21
|
+
* Minimal verdict shape — `valid` + `score` are required; `scores` +
|
|
22
|
+
* `notes` are optional surface. Validators that need richer shapes
|
|
23
|
+
* parameterise `Validator<Output, MyVerdict>` with their own type.
|
|
24
|
+
*
|
|
25
|
+
* Need structured extras? Extend DefaultVerdict with typed fields — never
|
|
26
|
+
* serialize extras into `notes`.
|
|
27
|
+
*/
|
|
28
|
+
interface DefaultVerdict {
|
|
29
|
+
/** Whether the output meets the validator's pass criteria. */
|
|
30
|
+
valid: boolean;
|
|
31
|
+
/** Aggregate score in [0, 1]. Drivers use this for winner selection. */
|
|
32
|
+
score: number;
|
|
33
|
+
/** Per-dimension scores. Free-form; weighted into `score` by the validator. */
|
|
34
|
+
scores?: Record<string, number>;
|
|
35
|
+
/** Human-readable rationale; surfaces in trace + final-result `winner.verdict`. */
|
|
36
|
+
notes?: string;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Multi-layer verifier — ordered pipeline of verification layers.
|
|
41
|
+
*
|
|
42
|
+
* Different contract from {@link JudgeRunner} (which runs parallel
|
|
43
|
+
* specs against a sandbox). MultiLayerVerifier is a DAG of layers
|
|
44
|
+
* (install → typecheck → build → lint → serve → semantic → …) with
|
|
45
|
+
* dependency-based skip, per-layer findings, soft-fail semantics, and
|
|
46
|
+
* an aggregated `blendedScore` across all passed layers.
|
|
47
|
+
*
|
|
48
|
+
* Use when you want:
|
|
49
|
+
* - ordered stages where a failing upstream stage skips downstream ones
|
|
50
|
+
* - each stage produces rich `findings` (severity + message + evidence)
|
|
51
|
+
* - a single composite score across stages with per-stage weights
|
|
52
|
+
* - soft-fail stages whose failure doesn't abort the pipeline
|
|
53
|
+
*
|
|
54
|
+
* Use {@link JudgeRunner} when you want:
|
|
55
|
+
* - N independent judges running in parallel against the same artifact
|
|
56
|
+
* - no inter-judge dependencies
|
|
57
|
+
* - boolean `passed` per judge + overall
|
|
58
|
+
*
|
|
59
|
+
* Both primitives compose — JudgeRunner can be invoked as a single
|
|
60
|
+
* layer inside a MultiLayerVerifier if that suits the caller.
|
|
61
|
+
*/
|
|
62
|
+
|
|
63
|
+
type LayerStatus = 'pass' | 'fail' | 'skipped' | 'error' | 'timeout';
|
|
64
|
+
type Severity = 'critical' | 'major' | 'minor' | 'info';
|
|
65
|
+
interface Finding {
|
|
66
|
+
severity: Severity;
|
|
67
|
+
message: string;
|
|
68
|
+
evidence?: string;
|
|
69
|
+
/** Optional layer name the finding belongs to (set by the verifier if omitted). */
|
|
70
|
+
layer?: string;
|
|
71
|
+
/**
|
|
72
|
+
* Free-form structured payload — used by `multiToolchainLayer` to attach
|
|
73
|
+
* `{ adapter: 'pnpm' }`, by judges to attach evidence pointers, etc.
|
|
74
|
+
* Renderers MAY interrogate; agent-eval primitives never assume shape.
|
|
75
|
+
*/
|
|
76
|
+
detail?: Record<string, unknown>;
|
|
77
|
+
}
|
|
78
|
+
interface LayerResult {
|
|
79
|
+
layer: string;
|
|
80
|
+
status: LayerStatus;
|
|
81
|
+
/** 0..1 score, optional — layers that don't produce a numeric score omit. */
|
|
82
|
+
score?: number;
|
|
83
|
+
durationMs: number;
|
|
84
|
+
findings: Finding[];
|
|
85
|
+
/** Short human-readable summary (one line). */
|
|
86
|
+
reason?: string;
|
|
87
|
+
/**
|
|
88
|
+
* Numeric layer-level diagnostics: error counts, warning counts,
|
|
89
|
+
* cyclomatic complexity, total adapter wall-time, etc. Keyed by
|
|
90
|
+
* diagnostic name; null = "diagnostic not applicable / not measured."
|
|
91
|
+
* Renderers that know the keys can display them; ones that don't,
|
|
92
|
+
* ignore. Free-form on purpose — consumers type the value shape in
|
|
93
|
+
* their own namespace.
|
|
94
|
+
*/
|
|
95
|
+
diagnostics?: Record<string, number | null>;
|
|
96
|
+
/** Any rich per-layer detail — rendered as-is by consumers that know the layer. */
|
|
97
|
+
detail?: Record<string, unknown>;
|
|
98
|
+
}
|
|
99
|
+
interface VerifyContext<Env = unknown> {
|
|
100
|
+
/** Per-run opaque context the caller provides. Layers destructure what they need. */
|
|
101
|
+
env: Env;
|
|
102
|
+
/** Previously-computed results from layers that already ran. */
|
|
103
|
+
prior: Record<string, LayerResult>;
|
|
104
|
+
/** Signal — if aborted, layers MUST bail within reasonable wall. */
|
|
105
|
+
signal: AbortSignal;
|
|
106
|
+
}
|
|
107
|
+
interface Layer<Env = unknown> {
|
|
108
|
+
name: string;
|
|
109
|
+
/** Stages that must have `status: 'pass'` before this layer runs. */
|
|
110
|
+
dependsOn?: string[];
|
|
111
|
+
/**
|
|
112
|
+
* Weight in the composite `blendedScore`. Default 1.0. Layers with weight 0
|
|
113
|
+
* contribute findings but not score.
|
|
114
|
+
*/
|
|
115
|
+
weight?: number;
|
|
116
|
+
/**
|
|
117
|
+
* If true, a `fail` status contributes to `blendedScore` (as 0) instead of
|
|
118
|
+
* being dropped — use for layers whose failure is a real signal. Default:
|
|
119
|
+
* fail drops from numerator + denominator, matching VB's existing semantics.
|
|
120
|
+
*/
|
|
121
|
+
failContributesToScore?: boolean;
|
|
122
|
+
/** Optional per-layer wall-cap in ms. Honored by the verifier (AbortSignal). */
|
|
123
|
+
capMs?: number;
|
|
124
|
+
run: (ctx: VerifyContext<Env>) => Promise<LayerResult> | LayerResult;
|
|
125
|
+
}
|
|
126
|
+
interface VerifyOptions<Env = unknown> {
|
|
127
|
+
env: Env;
|
|
128
|
+
/**
|
|
129
|
+
* Overall wall cap. Default: sum of layer capMs, or Infinity if any layer
|
|
130
|
+
* omits a cap. The verifier short-circuits remaining layers on overall cap.
|
|
131
|
+
*/
|
|
132
|
+
overallCapMs?: number;
|
|
133
|
+
/** Called with each layer result as it completes. */
|
|
134
|
+
onLayer?: (result: LayerResult) => void;
|
|
135
|
+
}
|
|
136
|
+
/** Extends the substrate verdict spine: `valid` = `allPass` and `score` =
|
|
137
|
+
* `blendedScore` — derived where the report is aggregated, so spine
|
|
138
|
+
* consumers (drivers, gates) read this report without an adapter. */
|
|
139
|
+
interface VerificationReport extends DefaultVerdict {
|
|
140
|
+
layers: LayerResult[];
|
|
141
|
+
passCount: number;
|
|
142
|
+
failCount: number;
|
|
143
|
+
skippedCount: number;
|
|
144
|
+
errorCount: number;
|
|
145
|
+
/** True iff at least one scored layer ran AND every scored layer passed. */
|
|
146
|
+
allPass: boolean;
|
|
147
|
+
/**
|
|
148
|
+
* Weighted mean of `score` across contributing layers. 0 when no layers
|
|
149
|
+
* contributed. See {@link Layer.failContributesToScore} for fail semantics.
|
|
150
|
+
*/
|
|
151
|
+
blendedScore: number;
|
|
152
|
+
durationMs: number;
|
|
153
|
+
startedAt: string;
|
|
154
|
+
finishedAt: string;
|
|
155
|
+
}
|
|
156
|
+
/**
|
|
157
|
+
* Ordered DAG of verification layers with dependency-based skipping, per-layer findings, soft-fail semantics, and a blended composite score across all passed layers.
|
|
158
|
+
*/
|
|
159
|
+
declare class MultiLayerVerifier<Env = unknown> {
|
|
160
|
+
private readonly layers;
|
|
161
|
+
constructor(layers: Layer<Env>[]);
|
|
162
|
+
run(opts: VerifyOptions<Env>): Promise<VerificationReport>;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
interface RunScore {
|
|
166
|
+
success: number;
|
|
167
|
+
goalProgress: number;
|
|
168
|
+
repoGroundedness: number;
|
|
169
|
+
driftPenalty: number;
|
|
170
|
+
toolUseQuality: number;
|
|
171
|
+
patchQuality: number;
|
|
172
|
+
testReality: number;
|
|
173
|
+
finalGate: number;
|
|
174
|
+
reviewerBlockers: number;
|
|
175
|
+
costUsd: number;
|
|
176
|
+
wallSeconds: number;
|
|
177
|
+
notes?: string[];
|
|
178
|
+
}
|
|
179
|
+
interface RunScoreWeights {
|
|
180
|
+
success: number;
|
|
181
|
+
goalProgress: number;
|
|
182
|
+
repoGroundedness: number;
|
|
183
|
+
driftPenalty: number;
|
|
184
|
+
toolUseQuality: number;
|
|
185
|
+
patchQuality: number;
|
|
186
|
+
testReality: number;
|
|
187
|
+
finalGate: number;
|
|
188
|
+
reviewerBlockers: number;
|
|
189
|
+
costUsd: number;
|
|
190
|
+
wallSeconds: number;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* Error taxonomy for `@tangle-network/agent-eval`.
|
|
195
|
+
*
|
|
196
|
+
* Every error this package throws as part of its *public contract* extends
|
|
197
|
+
* `AgentEvalError`. Consumers can pattern-match by `instanceof <Subclass>` or
|
|
198
|
+
* by the stable string `code` carried on the base class.
|
|
199
|
+
*
|
|
200
|
+
* The codes are stable across minor versions; new codes can be added, but
|
|
201
|
+
* existing codes never change meaning. New subclasses are non-breaking.
|
|
202
|
+
*
|
|
203
|
+
* Internal invariant guards (`throw new Error('this should never happen')`)
|
|
204
|
+
* remain plain `Error`s on purpose — they're programmer-mistake assertions,
|
|
205
|
+
* not consumer-catchable contract failures.
|
|
206
|
+
*/
|
|
207
|
+
type AgentEvalErrorCode = 'validation' | 'not_found' | 'config' | 'capture_integrity' | 'judge' | 'verification' | 'replay' | 'backend_integrity' | 'profile_matrix';
|
|
208
|
+
/**
|
|
209
|
+
* Base class for every contract error this package throws — carries the stable
|
|
210
|
+
* string `code` taxonomy so consumers can `instanceof`-match or switch on `code`.
|
|
211
|
+
*/
|
|
212
|
+
declare class AgentEvalError extends Error {
|
|
213
|
+
/** Stable string code. Survives minification; safe to switch on. */
|
|
214
|
+
readonly code: AgentEvalErrorCode;
|
|
215
|
+
constructor(code: AgentEvalErrorCode, message: string, options?: {
|
|
216
|
+
cause?: unknown;
|
|
217
|
+
});
|
|
218
|
+
}
|
|
219
|
+
/** Caller passed invalid arguments (out of range, mutually-exclusive options, bad shape). */
|
|
220
|
+
declare class ValidationError extends AgentEvalError {
|
|
221
|
+
constructor(message: string, options?: {
|
|
222
|
+
cause?: unknown;
|
|
223
|
+
});
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
|
|
227
|
+
type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | {
|
|
228
|
+
[key: string]: AgentProfileJson;
|
|
229
|
+
};
|
|
230
|
+
type AgentProfileDimensionValue = string | number | boolean | null;
|
|
231
|
+
interface AgentProfileSource {
|
|
232
|
+
/** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
|
|
233
|
+
kind: string;
|
|
234
|
+
/** sha256 over the canonical source profile object. */
|
|
235
|
+
hash: string;
|
|
236
|
+
}
|
|
237
|
+
interface AgentProfileHarness {
|
|
238
|
+
id: string;
|
|
239
|
+
version?: string;
|
|
240
|
+
hash?: string;
|
|
241
|
+
}
|
|
242
|
+
interface AgentProfileCell {
|
|
243
|
+
schemaVersion: AgentProfileCellSchemaVersion;
|
|
244
|
+
cellId: string;
|
|
245
|
+
profileId: string;
|
|
246
|
+
sourceProfile: AgentProfileSource;
|
|
247
|
+
harness?: AgentProfileHarness;
|
|
248
|
+
model?: string;
|
|
249
|
+
promptHash?: string;
|
|
250
|
+
dimensions?: Record<string, AgentProfileDimensionValue>;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
|
|
254
|
+
interface BudgetSpec {
|
|
255
|
+
tokens?: number;
|
|
256
|
+
wallMs?: number;
|
|
257
|
+
calls?: number;
|
|
258
|
+
usd?: number;
|
|
259
|
+
}
|
|
260
|
+
interface RunOutcome$1 {
|
|
261
|
+
score?: number;
|
|
262
|
+
pass?: boolean;
|
|
263
|
+
failureClass?: FailureClass;
|
|
264
|
+
notes?: string;
|
|
265
|
+
}
|
|
266
|
+
/**
|
|
267
|
+
* Layer — optional classification in a nested build workflow.
|
|
268
|
+
* `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
|
|
269
|
+
* `app-build`: sandbox harness that compiled + tested the generated scaffold.
|
|
270
|
+
* `app-runtime`: a run of the generated agent against a domain scenario.
|
|
271
|
+
* `meta`: any meta-eval (judge replay, correlation analysis).
|
|
272
|
+
*/
|
|
273
|
+
type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
|
|
274
|
+
interface Run {
|
|
275
|
+
runId: string;
|
|
276
|
+
/**
|
|
277
|
+
* Stable identifier of the scenario being executed.
|
|
278
|
+
*
|
|
279
|
+
* Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
|
|
280
|
+
* input WITHOUT this field, substituting a sensible default
|
|
281
|
+
* (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
|
|
282
|
+
* curated scenario to anchor to (runtime / operator / meta-eval runs). This
|
|
283
|
+
* keeps the persisted shape unambiguous for downstream filters + aggregations
|
|
284
|
+
* while removing the boilerplate of inventing placeholder ids at the call site.
|
|
285
|
+
*/
|
|
286
|
+
scenarioId: string;
|
|
287
|
+
variantId?: string;
|
|
288
|
+
datasetVersion?: string;
|
|
289
|
+
/** Git SHA of agent code at run time. */
|
|
290
|
+
codeSha?: string;
|
|
291
|
+
/** Hash of the prompt template + any system prompt. */
|
|
292
|
+
promptSha?: string;
|
|
293
|
+
/** Model id + date + system-prompt hash, concatenated. */
|
|
294
|
+
modelFingerprint?: string;
|
|
295
|
+
seed?: number;
|
|
296
|
+
/** Arbitrary environment markers (shell, docker version, tz). */
|
|
297
|
+
envFingerprint?: Record<string, string>;
|
|
298
|
+
/** Version of the redaction rules applied to this run. */
|
|
299
|
+
redactionVersion?: string;
|
|
300
|
+
/** Parent run in a nested build workflow. A builder run's children are
|
|
301
|
+
* app-build runs; those children are app-runtime runs. */
|
|
302
|
+
parentRunId?: string;
|
|
303
|
+
/** Stable project identifier — groups runs across chats + sessions. */
|
|
304
|
+
projectId?: string;
|
|
305
|
+
/** Chat/conversation identifier within a project. */
|
|
306
|
+
chatId?: string;
|
|
307
|
+
/** Layer classification — hint for aggregation; not enforced. */
|
|
308
|
+
layer?: RunLayer;
|
|
309
|
+
startedAt: number;
|
|
310
|
+
endedAt?: number;
|
|
311
|
+
status: RunStatus;
|
|
312
|
+
outcome?: RunOutcome$1;
|
|
313
|
+
budget?: BudgetSpec;
|
|
314
|
+
/** Free-form labels for downstream grouping. */
|
|
315
|
+
tags?: Record<string, string>;
|
|
316
|
+
}
|
|
317
|
+
type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
|
|
318
|
+
type SpanStatus = 'ok' | 'error';
|
|
319
|
+
interface SpanBase {
|
|
320
|
+
spanId: string;
|
|
321
|
+
parentSpanId?: string;
|
|
322
|
+
runId: string;
|
|
323
|
+
kind: SpanKind;
|
|
324
|
+
name: string;
|
|
325
|
+
startedAt: number;
|
|
326
|
+
endedAt?: number;
|
|
327
|
+
status?: SpanStatus;
|
|
328
|
+
error?: string;
|
|
329
|
+
/** Anything not covered by typed fields. Kept deliberately free-form. */
|
|
330
|
+
attributes?: Record<string, unknown>;
|
|
331
|
+
}
|
|
332
|
+
interface Message {
|
|
333
|
+
role: 'system' | 'user' | 'assistant' | 'tool';
|
|
334
|
+
content: string;
|
|
335
|
+
tokens?: number;
|
|
336
|
+
/** Multi-modal content descriptors; blobs themselves live in Artifacts. */
|
|
337
|
+
images?: Array<{
|
|
338
|
+
artifactId?: string;
|
|
339
|
+
url?: string;
|
|
340
|
+
mime?: string;
|
|
341
|
+
}>;
|
|
342
|
+
}
|
|
343
|
+
interface LlmSpan extends SpanBase {
|
|
344
|
+
kind: 'llm';
|
|
345
|
+
model: string;
|
|
346
|
+
messages: Message[];
|
|
347
|
+
output?: string;
|
|
348
|
+
inputTokens?: number;
|
|
349
|
+
/** All generated tokens, including the reasoning subset when present. */
|
|
350
|
+
outputTokens?: number;
|
|
351
|
+
cachedTokens?: number;
|
|
352
|
+
cacheWriteTokens?: number;
|
|
353
|
+
/** Reasoning-token subset of `outputTokens`. */
|
|
354
|
+
reasoningTokens?: number;
|
|
355
|
+
costUsd?: number;
|
|
356
|
+
finishReason?: string;
|
|
357
|
+
}
|
|
358
|
+
interface ToolSpan extends SpanBase {
|
|
359
|
+
kind: 'tool';
|
|
360
|
+
toolName: string;
|
|
361
|
+
args: unknown;
|
|
362
|
+
/** False when the source observed the call but did not capture its arguments. */
|
|
363
|
+
argsCaptured?: boolean;
|
|
364
|
+
result?: unknown;
|
|
365
|
+
latencyMs?: number;
|
|
366
|
+
}
|
|
367
|
+
interface RetrievalSpan extends SpanBase {
|
|
368
|
+
kind: 'retrieval';
|
|
369
|
+
query: string;
|
|
370
|
+
hits: Array<{
|
|
371
|
+
docId: string;
|
|
372
|
+
score: number;
|
|
373
|
+
content?: string;
|
|
374
|
+
}>;
|
|
375
|
+
}
|
|
376
|
+
interface JudgeSpan extends SpanBase {
|
|
377
|
+
kind: 'judge';
|
|
378
|
+
judgeId: string;
|
|
379
|
+
/** Span this judgment applies to. */
|
|
380
|
+
targetSpanId: string;
|
|
381
|
+
dimension: string;
|
|
382
|
+
/** Numeric score (free-range; interpretation up to the judge). */
|
|
383
|
+
score: number;
|
|
384
|
+
rationale?: string;
|
|
385
|
+
evidence?: string;
|
|
386
|
+
}
|
|
387
|
+
interface SandboxSpan extends SpanBase {
|
|
388
|
+
kind: 'sandbox';
|
|
389
|
+
image?: string;
|
|
390
|
+
command?: string;
|
|
391
|
+
exitCode?: number;
|
|
392
|
+
testsTotal?: number;
|
|
393
|
+
testsPassed?: number;
|
|
394
|
+
stdoutHash?: string;
|
|
395
|
+
stderrHash?: string;
|
|
396
|
+
/** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
|
|
397
|
+
wallMs?: number;
|
|
398
|
+
}
|
|
399
|
+
interface GenericSpan extends SpanBase {
|
|
400
|
+
kind: 'agent' | 'custom';
|
|
401
|
+
}
|
|
402
|
+
type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
|
|
403
|
+
type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
|
|
404
|
+
interface TraceEvent {
|
|
405
|
+
eventId: string;
|
|
406
|
+
runId: string;
|
|
407
|
+
spanId?: string;
|
|
408
|
+
kind: EventKind;
|
|
409
|
+
timestamp: number;
|
|
410
|
+
payload: Record<string, unknown>;
|
|
411
|
+
}
|
|
412
|
+
interface BudgetLedgerEntry {
|
|
413
|
+
runId: string;
|
|
414
|
+
dimension: keyof BudgetSpec;
|
|
415
|
+
limit: number;
|
|
416
|
+
consumed: number;
|
|
417
|
+
remaining: number;
|
|
418
|
+
timestamp: number;
|
|
419
|
+
breached: boolean;
|
|
420
|
+
/** Span that triggered this entry, if any. */
|
|
421
|
+
spanId?: string;
|
|
422
|
+
}
|
|
423
|
+
interface Artifact {
|
|
424
|
+
artifactId: string;
|
|
425
|
+
runId: string;
|
|
426
|
+
spanId?: string;
|
|
427
|
+
contentType: string;
|
|
428
|
+
sizeBytes: number;
|
|
429
|
+
/** sha256 in hex. */
|
|
430
|
+
hash: string;
|
|
431
|
+
/** External storage URL (R2, S3, filesystem path). */
|
|
432
|
+
storageUrl?: string;
|
|
433
|
+
/** Inline content for small blobs — keep under ~64KB. */
|
|
434
|
+
inlineContent?: string;
|
|
435
|
+
}
|
|
436
|
+
type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
|
|
437
|
+
|
|
438
|
+
/**
|
|
439
|
+
* Paper-grade RunRecord schema + runtime validator.
|
|
440
|
+
*
|
|
441
|
+
* Every run that participates in a promotion gate, paper table, or
|
|
442
|
+
* researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
|
|
443
|
+
* fields are exactly those the paper "Two Loops, Three Roles" requires
|
|
444
|
+
* for reproducibility: who/what/when/cost/seed/hash, plus the search vs
|
|
445
|
+
* holdout split tag and either a `searchScore` or a `holdoutScore`.
|
|
446
|
+
*
|
|
447
|
+
* This is intentionally NOT a replacement for the rich `Run` /
|
|
448
|
+
* `ProposeReviewReport` / `ScenarioResult` types already in the
|
|
449
|
+
* package. Those are runtime structures with full provenance. A
|
|
450
|
+
* `RunRecord` is the analysis-time projection — the JSON-friendly
|
|
451
|
+
* row you'd put in a parquet file or paste into a notebook.
|
|
452
|
+
*
|
|
453
|
+
* Validate at the boundary:
|
|
454
|
+
*
|
|
455
|
+
* const rec = validateRunRecord(rawJson) // throws on missing
|
|
456
|
+
* const ok = isRunRecord(rawJson) // boolean check
|
|
457
|
+
* const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
|
|
458
|
+
*
|
|
459
|
+
* The validator runs in pure TS — zod is intentionally NOT a
|
|
460
|
+
* dependency. Round-trip tested in `tests/run-record.test.ts`.
|
|
461
|
+
*/
|
|
462
|
+
|
|
463
|
+
/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
|
|
464
|
+
* combined train+test pool that the optimizer is allowed to read. */
|
|
465
|
+
type RunSplitTag = 'search' | 'dev' | 'holdout';
|
|
466
|
+
interface RunTokenUsage {
|
|
467
|
+
input: number;
|
|
468
|
+
/** All generated tokens charged as output, including reasoning tokens. */
|
|
469
|
+
output: number;
|
|
470
|
+
/** Reasoning-token subset of `output`, when the provider reports it. */
|
|
471
|
+
reasoning?: number;
|
|
472
|
+
/** Prompt tokens served from a provider cache. */
|
|
473
|
+
cached?: number;
|
|
474
|
+
/** Prompt tokens written into a provider cache. */
|
|
475
|
+
cacheWrite?: number;
|
|
476
|
+
}
|
|
477
|
+
/**
|
|
478
|
+
* How a run's USD amount was obtained.
|
|
479
|
+
*
|
|
480
|
+
* `costUsd` remains mandatory for wire compatibility. New producers should
|
|
481
|
+
* always populate this discriminated union so a missing bill is never
|
|
482
|
+
* mistaken for an observed zero-dollar run. For `uncaptured`, `costUsd` uses
|
|
483
|
+
* the legacy `0` sentinel while this field carries the truthful null.
|
|
484
|
+
*/
|
|
485
|
+
type RunCostProvenance = {
|
|
486
|
+
kind: 'observed';
|
|
487
|
+
usd: number;
|
|
488
|
+
} | {
|
|
489
|
+
kind: 'estimated';
|
|
490
|
+
usd: number;
|
|
491
|
+
} | {
|
|
492
|
+
kind: 'uncaptured';
|
|
493
|
+
usd: null;
|
|
494
|
+
};
|
|
495
|
+
interface RunJudgeMetadata {
|
|
496
|
+
model: string;
|
|
497
|
+
promptVersion: string;
|
|
498
|
+
/** [0,1] confidence the judge declared. Constant judge confidence
|
|
499
|
+
* across many runs is a fallback signal (see `canary.ts`). */
|
|
500
|
+
confidence: number;
|
|
501
|
+
/** True if the judge degraded to a fallback path (rules-only,
|
|
502
|
+
* prior-call cache, etc.). The canary uses this to alert. */
|
|
503
|
+
fallback: boolean;
|
|
504
|
+
}
|
|
505
|
+
/**
|
|
506
|
+
* Per-judge / per-dimension breakdown for runs scored by an ensemble of
|
|
507
|
+
* judges over a multi-dimensional rubric.
|
|
508
|
+
*
|
|
509
|
+
* The collapsed `outcome.searchScore` / `holdoutScore` carries the
|
|
510
|
+
* composite the gate uses. The full breakdown belongs here so consumers
|
|
511
|
+
* can answer "which judge disagreed?", "which dimension dragged the
|
|
512
|
+
* composite down?", and "did half the panel fail?" without re-running.
|
|
513
|
+
*
|
|
514
|
+
* `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
|
|
515
|
+
* `composite` are convenience projections — derivable but precomputed so
|
|
516
|
+
* downstream IRR primitives (`interRaterReliability`,
|
|
517
|
+
* `corpusInterRaterAgreement`) and reporters don't pay the same
|
|
518
|
+
* aggregation twice.
|
|
519
|
+
*
|
|
520
|
+
* Fail-loud discipline: judges that errored out land in `failedJudges`
|
|
521
|
+
* by id. A missing key in `perJudge` is ambiguous (silent zero vs not
|
|
522
|
+
* run); the explicit list makes a partial-failure recorded as such.
|
|
523
|
+
*/
|
|
524
|
+
interface JudgeScoresRecord {
|
|
525
|
+
/** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
|
|
526
|
+
perJudge: Record<string, Record<string, number>>;
|
|
527
|
+
/** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
|
|
528
|
+
perDimMean: Record<string, number>;
|
|
529
|
+
/** Composite mean across all dims and judges. Mirrors the score
|
|
530
|
+
* the gate sees on `outcome.searchScore` / `holdoutScore`. */
|
|
531
|
+
composite: number;
|
|
532
|
+
/** Judges that errored or returned an unparseable verdict. Recorded
|
|
533
|
+
* by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
|
|
534
|
+
* not inferred from missing keys in `perJudge`. */
|
|
535
|
+
failedJudges?: string[];
|
|
536
|
+
/** Free-form notes the judges emitted (joined across judges or
|
|
537
|
+
* first-judge only — consumer's choice). */
|
|
538
|
+
notes?: string;
|
|
539
|
+
}
|
|
540
|
+
interface RunOutcome {
|
|
541
|
+
/** Score on the search/optimization split. Optional because a
|
|
542
|
+
* holdout-only evaluation only fills `holdoutScore`. */
|
|
543
|
+
searchScore?: number;
|
|
544
|
+
/** Score on the held-out split. Optional because a search-only run
|
|
545
|
+
* only fills `searchScore`. At least one must be present. */
|
|
546
|
+
holdoutScore?: number;
|
|
547
|
+
/** Bag of any other metric the run produced — judge dimensions,
|
|
548
|
+
* pass/fail counters, latency stats, etc. Numeric only — keeps
|
|
549
|
+
* reporters honest. */
|
|
550
|
+
raw: Record<string, number>;
|
|
551
|
+
/** Per-judge / per-dim breakdown. Consumers writing ensemble
|
|
552
|
+
* judgements populate this; substrate primitives like
|
|
553
|
+
* `interRaterReliability` and `corpusInterRaterAgreement` accept
|
|
554
|
+
* these records as input. Optional — single-judge or scalar-only
|
|
555
|
+
* runs leave it unset. */
|
|
556
|
+
judgeScores?: JudgeScoresRecord;
|
|
557
|
+
/** Authenticity / realness verdict — did the run build the REAL thing on the
|
|
558
|
+
* intended infra, or fake it (see `./authenticity`)? Optional: only domains
|
|
559
|
+
* with an authenticity config populate it. Carried in the corpus so the
|
|
560
|
+
* flywheel / off-policy learning can optimize for real completion, not gamed
|
|
561
|
+
* pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
|
|
562
|
+
* must not count as a real success regardless of `score`. */
|
|
563
|
+
realness?: {
|
|
564
|
+
score: number;
|
|
565
|
+
gated: boolean;
|
|
566
|
+
reason?: string;
|
|
567
|
+
};
|
|
568
|
+
}
|
|
569
|
+
/**
|
|
570
|
+
* Mandatory paper-grade fields for a single evaluation run. Optional
|
|
571
|
+
* fields are extension points; mandatory fields throw if missing.
|
|
572
|
+
*
|
|
573
|
+
* Hash discipline:
|
|
574
|
+
* - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
|
|
575
|
+
* model (after any steering bundle merge).
|
|
576
|
+
* - `configHash` is the sha256 of the effective run config (model,
|
|
577
|
+
* temperature, tools, judges, splits). The pair (promptHash,
|
|
578
|
+
* configHash) uniquely identifies an experiment cell.
|
|
579
|
+
*
|
|
580
|
+
* Model snapshot discipline:
|
|
581
|
+
* - `model` MUST encode a snapshot version. Bare aliases like
|
|
582
|
+
* `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
|
|
583
|
+
* Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
|
|
584
|
+
*/
|
|
585
|
+
interface RunRecord {
|
|
586
|
+
/** UUID for the run. */
|
|
587
|
+
runId: string;
|
|
588
|
+
/** Logical experiment grouping (a treatment vs a baseline within
|
|
589
|
+
* the same sweep should share `experimentId`). */
|
|
590
|
+
experimentId: string;
|
|
591
|
+
/** Stable identifier for the candidate (variant) being run. The
|
|
592
|
+
* promotion gate compares two `candidateId`s on matched items. */
|
|
593
|
+
candidateId: string;
|
|
594
|
+
/** RNG seed for the run. Always recorded — silent re-seeding is
|
|
595
|
+
* the most common cause of non-reproducible numbers. */
|
|
596
|
+
seed: number;
|
|
597
|
+
/** Model identifier WITH snapshot version. */
|
|
598
|
+
model: string;
|
|
599
|
+
/** sha256 of the effective prompt (post-steering). */
|
|
600
|
+
promptHash: string;
|
|
601
|
+
/** sha256 of the effective config. */
|
|
602
|
+
configHash: string;
|
|
603
|
+
/** Git SHA the harness was run from. */
|
|
604
|
+
commitSha: string;
|
|
605
|
+
/** End-to-end wall-clock duration in milliseconds. */
|
|
606
|
+
wallMs: number;
|
|
607
|
+
/** Time spent queued before execution started, if known. */
|
|
608
|
+
queueMs?: number;
|
|
609
|
+
/** Total USD cost. Mandatory — runs without a cost number are
|
|
610
|
+
* unbounded by definition and must not be admitted into the gate.
|
|
611
|
+
* `0` is retained as the compatibility sentinel for an uncaptured amount;
|
|
612
|
+
* inspect `costProvenance` before treating it as observed. */
|
|
613
|
+
costUsd: number;
|
|
614
|
+
/** Observed, model-priced estimate, or genuinely uncaptured USD amount.
|
|
615
|
+
* Optional only so existing serialized RunRecords remain valid. */
|
|
616
|
+
costProvenance?: RunCostProvenance;
|
|
617
|
+
/** Token usage breakdown. */
|
|
618
|
+
tokenUsage: RunTokenUsage;
|
|
619
|
+
/** Judge-side metadata, if a judge was used. */
|
|
620
|
+
judgeMetadata?: RunJudgeMetadata;
|
|
621
|
+
/** Per-split scores + raw bag. */
|
|
622
|
+
outcome: RunOutcome;
|
|
623
|
+
/** Canonical, cross-agent failure class drawn from the shared
|
|
624
|
+
* `FAILURE_CLASSES` taxonomy. This is the aggregation key that makes
|
|
625
|
+
* "which failure dominates across the whole fleet" answerable in ONE
|
|
626
|
+
* vocabulary — every agent classifies against the same enum. Producers
|
|
627
|
+
* set it via the substrate classifier; leave unset only when the failure
|
|
628
|
+
* genuinely can't be classified. */
|
|
629
|
+
failureClass?: FailureClass;
|
|
630
|
+
/** Free-form domain-specific failure detail, scoped UNDER `failureClass`
|
|
631
|
+
* (e.g. failureClass='tool_recovery_failure', failureMode='forge_build_unsatisfied').
|
|
632
|
+
* The within-agent drill-down; `failureClass` is the cross-agent key. */
|
|
633
|
+
failureMode?: string;
|
|
634
|
+
/** Which split this run was drawn from. */
|
|
635
|
+
splitTag: RunSplitTag;
|
|
636
|
+
/**
|
|
637
|
+
* Stable scenario identifier the run was scored against. Optional for
|
|
638
|
+
* backwards compatibility, but **strongly recommended**: every primitive
|
|
639
|
+
* that pairs runs by scenario (preferences, paired stats, BT tournament)
|
|
640
|
+
* keys on this. The campaign artifact populates it canonically; legacy
|
|
641
|
+
* runs without it fall back to inference from `outcome.raw.scenario_id`
|
|
642
|
+
* or `experimentId`.
|
|
643
|
+
*/
|
|
644
|
+
scenarioId?: string;
|
|
645
|
+
/**
|
|
646
|
+
* Canonical identity for the agent profile cell that produced this row:
|
|
647
|
+
* profile artifact hash plus optional harness/model/prompt/reporting
|
|
648
|
+
* dimensions. Use `agentProfile.cellId` to group persona sweeps and
|
|
649
|
+
* longitudinal reports by the complete source profile, not by a loose
|
|
650
|
+
* candidate label or opaque config hash.
|
|
651
|
+
*/
|
|
652
|
+
agentProfile?: AgentProfileCell;
|
|
653
|
+
}
|
|
654
|
+
|
|
655
|
+
/**
|
|
656
|
+
* RawProviderSink — first-class persistence for the actual HTTP-level
|
|
657
|
+
* request/response bodies of every LLM provider call.
|
|
658
|
+
*
|
|
659
|
+
* Why this is a separate sink from the structured `LlmSpan`:
|
|
660
|
+
*
|
|
661
|
+
* - `LlmSpan` records the *intent* — model name, messages, output text,
|
|
662
|
+
* usage. It's what dashboards read; it's NOT enough for forensics.
|
|
663
|
+
* - When a downstream consumer reports "the verifier used the wrong route"
|
|
664
|
+
* or "tokens look right but reasoning was missing," the only way to
|
|
665
|
+
* answer is the raw HTTP body. Span fields can lie (a proxy can echo
|
|
666
|
+
* a different `model` value than what actually answered); the raw
|
|
667
|
+
* response is ground truth.
|
|
668
|
+
*
|
|
669
|
+
* Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
|
|
670
|
+
* matrix runner / BuilderSession sets it up automatically) and every
|
|
671
|
+
* request, response, and error is recorded — including retries, with the
|
|
672
|
+
* attempt index attached so a flaky call's full event chain is recoverable.
|
|
673
|
+
*
|
|
674
|
+
* Redaction is enforced at sink time. The default redactor strips
|
|
675
|
+
* `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
|
|
676
|
+
* payload field whose key matches `apiKey | api_key | bearer | password |
|
|
677
|
+
* secret | token` (case-insensitive). Override via the sink constructor or
|
|
678
|
+
* the per-call `redactor`. The `redactedFields` array on the persisted
|
|
679
|
+
* event lets a reviewer see what was stripped without exposing the values.
|
|
680
|
+
*/
|
|
681
|
+
type RawProviderDirection = 'request' | 'response' | 'error';
|
|
682
|
+
interface RawProviderEvent {
|
|
683
|
+
/** Stable id. Generated by the sink if omitted. */
|
|
684
|
+
eventId: string;
|
|
685
|
+
/** Trace context populated by `LlmClient` when the call is wrapped in a span. */
|
|
686
|
+
runId?: string;
|
|
687
|
+
spanId?: string;
|
|
688
|
+
/**
|
|
689
|
+
* Logical provider name. Free-form so callers can use whatever id matches
|
|
690
|
+
* their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
|
|
691
|
+
* omitted, derived from `baseUrl` in `LlmClientOptions`.
|
|
692
|
+
*/
|
|
693
|
+
provider: string;
|
|
694
|
+
model: string;
|
|
695
|
+
/** Endpoint path, e.g. `'/v1/chat/completions'`. */
|
|
696
|
+
endpoint: string;
|
|
697
|
+
/** Base URL used for the call (already-normalised — no trailing slash). */
|
|
698
|
+
baseUrl: string;
|
|
699
|
+
/** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
|
|
700
|
+
attemptIndex: number;
|
|
701
|
+
direction: RawProviderDirection;
|
|
702
|
+
/** Unix ms. */
|
|
703
|
+
timestamp: number;
|
|
704
|
+
/** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
|
|
705
|
+
durationMs?: number;
|
|
706
|
+
statusCode?: number;
|
|
707
|
+
requestHeaders?: Record<string, string>;
|
|
708
|
+
requestBody?: unknown;
|
|
709
|
+
responseHeaders?: Record<string, string>;
|
|
710
|
+
responseBody?: unknown;
|
|
711
|
+
/** Set on `direction: 'error'` events. */
|
|
712
|
+
errorMessage?: string;
|
|
713
|
+
/** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
|
|
714
|
+
redactedFields: string[];
|
|
715
|
+
}
|
|
716
|
+
interface RawProviderSinkFilter {
|
|
717
|
+
runId?: string;
|
|
718
|
+
spanId?: string;
|
|
719
|
+
direction?: RawProviderDirection;
|
|
720
|
+
attemptIndex?: number;
|
|
721
|
+
}
|
|
722
|
+
interface RawProviderSink {
|
|
723
|
+
record(event: RawProviderEvent): Promise<void>;
|
|
724
|
+
/** Optional listing — implementations that durably persist (file, db) should support this. */
|
|
725
|
+
list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
726
|
+
/** Optional teardown for backed implementations. */
|
|
727
|
+
close?(): Promise<void>;
|
|
728
|
+
}
|
|
729
|
+
type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
|
|
730
|
+
|
|
731
|
+
interface RunFilter {
|
|
732
|
+
scenarioId?: string;
|
|
733
|
+
variantId?: string;
|
|
734
|
+
status?: RunStatus;
|
|
735
|
+
since?: number;
|
|
736
|
+
until?: number;
|
|
737
|
+
tag?: {
|
|
738
|
+
key: string;
|
|
739
|
+
value: string;
|
|
740
|
+
};
|
|
741
|
+
parentRunId?: string;
|
|
742
|
+
projectId?: string;
|
|
743
|
+
chatId?: string;
|
|
744
|
+
layer?: RunLayer;
|
|
745
|
+
}
|
|
746
|
+
interface SpanFilter {
|
|
747
|
+
runId?: string;
|
|
748
|
+
parentSpanId?: string;
|
|
749
|
+
kind?: SpanKind;
|
|
750
|
+
name?: string;
|
|
751
|
+
toolName?: string;
|
|
752
|
+
judgeId?: string;
|
|
753
|
+
since?: number;
|
|
754
|
+
until?: number;
|
|
755
|
+
}
|
|
756
|
+
interface EventFilter {
|
|
757
|
+
runId?: string;
|
|
758
|
+
spanId?: string;
|
|
759
|
+
kind?: EventKind;
|
|
760
|
+
since?: number;
|
|
761
|
+
until?: number;
|
|
762
|
+
}
|
|
763
|
+
interface TraceStore {
|
|
764
|
+
appendRun(run: Run): Promise<void>;
|
|
765
|
+
updateRun(runId: string, patch: Partial<Run>): Promise<void>;
|
|
766
|
+
appendSpan(span: Span): Promise<void>;
|
|
767
|
+
updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
|
|
768
|
+
appendEvent(event: TraceEvent): Promise<void>;
|
|
769
|
+
appendArtifact(artifact: Artifact): Promise<void>;
|
|
770
|
+
appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
|
|
771
|
+
getRun(runId: string): Promise<Run | undefined>;
|
|
772
|
+
listRuns(filter?: RunFilter): Promise<Run[]>;
|
|
773
|
+
spans(filter?: SpanFilter): Promise<Span[]>;
|
|
774
|
+
events(filter?: EventFilter): Promise<TraceEvent[]>;
|
|
775
|
+
budget(runId: string): Promise<BudgetLedgerEntry[]>;
|
|
776
|
+
artifacts(runId: string): Promise<Artifact[]>;
|
|
777
|
+
}
|
|
778
|
+
|
|
779
|
+
interface RunTrace {
|
|
780
|
+
run: Run;
|
|
781
|
+
spans: Span[];
|
|
782
|
+
events: TraceEvent[];
|
|
783
|
+
artifacts: Artifact[];
|
|
784
|
+
budget: BudgetLedgerEntry[];
|
|
785
|
+
}
|
|
786
|
+
interface RunCriticOptions {
|
|
787
|
+
weights?: Partial<RunScoreWeights>;
|
|
788
|
+
driftPatterns?: RegExp[];
|
|
789
|
+
}
|
|
790
|
+
declare class RunCritic {
|
|
791
|
+
private readonly weights?;
|
|
792
|
+
private readonly driftPatterns;
|
|
793
|
+
constructor(options?: RunCriticOptions);
|
|
794
|
+
score(store: TraceStore, runId: string): Promise<RunScore>;
|
|
795
|
+
scoreTrace(trace: RunTrace): RunScore;
|
|
796
|
+
rank(score: RunScore): number;
|
|
797
|
+
private isDrift;
|
|
798
|
+
}
|
|
799
|
+
|
|
800
|
+
type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
|
|
801
|
+
interface CostUsage {
|
|
802
|
+
inputTokens: number;
|
|
803
|
+
/** Includes reasoning tokens when the provider bills them as output. */
|
|
804
|
+
outputTokens: number;
|
|
805
|
+
/** Reasoning-token subset of outputTokens, when reported. */
|
|
806
|
+
reasoningTokens?: number;
|
|
807
|
+
/** Prompt tokens served from a provider cache. */
|
|
808
|
+
cachedTokens?: number;
|
|
809
|
+
/** Prompt tokens written into a provider cache. */
|
|
810
|
+
cacheWriteTokens?: number;
|
|
811
|
+
}
|
|
812
|
+
interface CostCallBase {
|
|
813
|
+
callId: string;
|
|
814
|
+
channel: CostChannel;
|
|
815
|
+
phase: string;
|
|
816
|
+
actor: string;
|
|
817
|
+
model: string;
|
|
818
|
+
maximumCostUsd?: number;
|
|
819
|
+
tags?: Record<string, string>;
|
|
820
|
+
timestamp: number;
|
|
821
|
+
}
|
|
822
|
+
interface CostReceipt extends CostCallBase, CostUsage {
|
|
823
|
+
status: 'settled';
|
|
824
|
+
costUsd: number;
|
|
825
|
+
costUnknown: boolean;
|
|
826
|
+
usageUnknown?: boolean;
|
|
827
|
+
pricing?: {
|
|
828
|
+
inputUsdPerThousand: number;
|
|
829
|
+
outputUsdPerThousand: number;
|
|
830
|
+
};
|
|
831
|
+
actualCostUsd?: number;
|
|
832
|
+
error?: string;
|
|
833
|
+
}
|
|
834
|
+
interface CostReceiptInput extends CostUsage {
|
|
835
|
+
model: string;
|
|
836
|
+
actualCostUsd?: number;
|
|
837
|
+
costUnknown?: boolean;
|
|
838
|
+
usageUnknown?: boolean;
|
|
839
|
+
}
|
|
840
|
+
type MaximumCharge = {
|
|
841
|
+
externallyEnforcedMaximumUsd: number;
|
|
842
|
+
} | ({
|
|
843
|
+
model: string;
|
|
844
|
+
} & CostUsage);
|
|
845
|
+
interface RunPaidCallInput<T> {
|
|
846
|
+
callId?: string;
|
|
847
|
+
channel: CostChannel;
|
|
848
|
+
phase: string;
|
|
849
|
+
actor: string;
|
|
850
|
+
/** Used before a provider receipt exists and on failures without one. */
|
|
851
|
+
model?: string;
|
|
852
|
+
tags?: Record<string, string>;
|
|
853
|
+
signal?: AbortSignal;
|
|
854
|
+
/** Provider-enforced dollar maximum, or maximum priced token usage. Required when capped. */
|
|
855
|
+
maximumCharge?: MaximumCharge;
|
|
856
|
+
/** `callId` can be forwarded as the provider's idempotency key. */
|
|
857
|
+
execute(signal: AbortSignal, callId: string): Promise<T>;
|
|
858
|
+
receipt(value: T): CostReceiptInput;
|
|
859
|
+
receiptFromError?(error: Error): CostReceiptInput | undefined;
|
|
860
|
+
}
|
|
861
|
+
type PaidCallResult<T> = {
|
|
862
|
+
succeeded: true;
|
|
863
|
+
callId: string;
|
|
864
|
+
value: T;
|
|
865
|
+
receipt: CostReceipt;
|
|
866
|
+
} | {
|
|
867
|
+
succeeded: false;
|
|
868
|
+
callId?: string;
|
|
869
|
+
error: Error;
|
|
870
|
+
receipt?: CostReceipt;
|
|
871
|
+
};
|
|
872
|
+
interface ChannelRollup {
|
|
873
|
+
channel: CostChannel;
|
|
874
|
+
calls: number;
|
|
875
|
+
inputTokens: number;
|
|
876
|
+
outputTokens: number;
|
|
877
|
+
reasoningTokens?: number;
|
|
878
|
+
cachedTokens: number;
|
|
879
|
+
cacheWriteTokens?: number;
|
|
880
|
+
costUsd: number;
|
|
881
|
+
unpricedCalls: number;
|
|
882
|
+
unknownUsageCalls: number;
|
|
883
|
+
}
|
|
884
|
+
interface CostLedgerSummary {
|
|
885
|
+
totalCalls: number;
|
|
886
|
+
pendingCalls: number;
|
|
887
|
+
unresolvedCalls: number;
|
|
888
|
+
reservedCostUsd: number;
|
|
889
|
+
inputTokens: number;
|
|
890
|
+
outputTokens: number;
|
|
891
|
+
reasoningTokens?: number;
|
|
892
|
+
cachedTokens: number;
|
|
893
|
+
cacheWriteTokens?: number;
|
|
894
|
+
totalCostUsd: number;
|
|
895
|
+
byChannel: ChannelRollup[];
|
|
896
|
+
unpricedModels: string[];
|
|
897
|
+
fullyPriced: boolean;
|
|
898
|
+
usageComplete: boolean;
|
|
899
|
+
accountingComplete: boolean;
|
|
900
|
+
incompleteReasons: string[];
|
|
901
|
+
}
|
|
902
|
+
interface CostLedgerFilter {
|
|
903
|
+
channel?: CostChannel;
|
|
904
|
+
phase?: string;
|
|
905
|
+
tags?: Record<string, string>;
|
|
906
|
+
}
|
|
907
|
+
interface CostLedgerWaitOptions {
|
|
908
|
+
/** Maximum time to wait for active provider calls. Default 5 seconds. */
|
|
909
|
+
timeoutMs?: number;
|
|
910
|
+
}
|
|
911
|
+
/** Append-only storage. `append` must atomically reject stale revisions. */
|
|
912
|
+
interface CostLedgerPersistence {
|
|
913
|
+
read(): {
|
|
914
|
+
revision: string;
|
|
915
|
+
events: string;
|
|
916
|
+
};
|
|
917
|
+
append(expectedRevision: string, event: string): string | undefined;
|
|
918
|
+
}
|
|
919
|
+
interface CostLedgerOptions {
|
|
920
|
+
costCeilingUsd?: number;
|
|
921
|
+
persistence?: CostLedgerPersistence;
|
|
922
|
+
/** Import already-settled receipts without admitting new paid work. */
|
|
923
|
+
receipts?: readonly CostReceipt[];
|
|
924
|
+
}
|
|
925
|
+
/** Run-wide paid-call admission, durable call state, receipts, and summaries. */
|
|
926
|
+
declare class CostLedger {
|
|
927
|
+
private readonly records;
|
|
928
|
+
private readonly activeCallIds;
|
|
929
|
+
private readonly lateCallIds;
|
|
930
|
+
private readonly idleWaiters;
|
|
931
|
+
private completedTasks;
|
|
932
|
+
private revision;
|
|
933
|
+
private costLimitPersisted;
|
|
934
|
+
readonly costCeilingUsd?: number;
|
|
935
|
+
private readonly persistence?;
|
|
936
|
+
constructor(input?: number | CostLedgerOptions);
|
|
937
|
+
runPaidCall<T>(input: RunPaidCallInput<T>): Promise<PaidCallResult<T>>;
|
|
938
|
+
/** Wait until every call started by this ledger has produced a durable outcome. */
|
|
939
|
+
waitForIdle(options?: CostLedgerWaitOptions): Promise<boolean>;
|
|
940
|
+
/** Settle a call left pending by a crashed process after reconciling with the provider. */
|
|
941
|
+
reconcile(callId: string, observed: CostReceiptInput, options?: {
|
|
942
|
+
error?: string;
|
|
943
|
+
}): CostReceipt;
|
|
944
|
+
list(filter?: CostLedgerFilter): CostReceipt[];
|
|
945
|
+
summary(filter?: CostLedgerFilter): CostLedgerSummary;
|
|
946
|
+
markCompleted(count?: number): void;
|
|
947
|
+
costPerCompletedTask(): number | null;
|
|
948
|
+
private execute;
|
|
949
|
+
private captureLateOutcome;
|
|
950
|
+
private releaseActiveCall;
|
|
951
|
+
private commitOutcome;
|
|
952
|
+
private captureFailure;
|
|
953
|
+
private commitReceipt;
|
|
954
|
+
private resolveMaximum;
|
|
955
|
+
private hasIncompleteSettledCall;
|
|
956
|
+
private appendRecord;
|
|
957
|
+
private ensureCostLimitPersisted;
|
|
958
|
+
private appendEvent;
|
|
959
|
+
}
|
|
960
|
+
/** Public callback surface for a shared cost ledger.
|
|
961
|
+
*
|
|
962
|
+
* Declaration bundles may expose this type through multiple package subpaths.
|
|
963
|
+
* Keeping callback contracts structural lets those subpaths compose while the
|
|
964
|
+
* concrete {@link CostLedger} retains its private durable state.
|
|
965
|
+
*/
|
|
966
|
+
type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'waitForIdle'>> & Partial<Pick<CostLedger, 'waitForIdle'>>;
|
|
967
|
+
|
|
968
|
+
/**
|
|
969
|
+
* LLM client with graceful degrade.
|
|
970
|
+
*
|
|
971
|
+
* OpenAI-compatible `/v1/chat/completions` client with:
|
|
972
|
+
* - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
|
|
973
|
+
* - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
|
|
974
|
+
* - Graceful json_schema → json_object degrade on 400 with schema-reject body.
|
|
975
|
+
* - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
|
|
976
|
+
* - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
|
|
977
|
+
* directly, cli-bridge subscriptions, and any router that speaks the spec.
|
|
978
|
+
*
|
|
979
|
+
* Usage:
|
|
980
|
+
* const { value, result } = await callLlmJson<MyType>(
|
|
981
|
+
* { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
|
|
982
|
+
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
983
|
+
* )
|
|
984
|
+
*
|
|
985
|
+
* This is THE llm-calling seam for agent-eval primitives that need structured
|
|
986
|
+
* output (semantic concept judge, reviewer directives, critic scores). Primitives
|
|
987
|
+
* that need free-form text use `callLlm` and parse output themselves.
|
|
988
|
+
*/
|
|
989
|
+
|
|
990
|
+
interface LlmMessage {
|
|
991
|
+
role: 'system' | 'user' | 'assistant';
|
|
992
|
+
/**
|
|
993
|
+
* Either a plain text content string OR a multimodal content array
|
|
994
|
+
* (text + image_url parts) for vision-capable models.
|
|
995
|
+
*/
|
|
996
|
+
content: string | Array<{
|
|
997
|
+
type: 'text';
|
|
998
|
+
text: string;
|
|
999
|
+
} | {
|
|
1000
|
+
type: 'image_url';
|
|
1001
|
+
image_url: {
|
|
1002
|
+
url: string;
|
|
1003
|
+
detail?: 'auto' | 'low' | 'high';
|
|
1004
|
+
};
|
|
1005
|
+
}>;
|
|
1006
|
+
}
|
|
1007
|
+
interface LlmCallRequest {
|
|
1008
|
+
model: string;
|
|
1009
|
+
messages: LlmMessage[];
|
|
1010
|
+
/** Optional JSON-mode response format (response_format: json_object). */
|
|
1011
|
+
jsonMode?: boolean;
|
|
1012
|
+
/** Optional structured output via JSON Schema. Falls back to json_object on 400. */
|
|
1013
|
+
jsonSchema?: {
|
|
1014
|
+
name: string;
|
|
1015
|
+
schema: Record<string, unknown>;
|
|
1016
|
+
};
|
|
1017
|
+
temperature?: number;
|
|
1018
|
+
maxTokens?: number;
|
|
1019
|
+
/** Per-call timeout, default 300s. */
|
|
1020
|
+
timeoutMs?: number;
|
|
1021
|
+
}
|
|
1022
|
+
interface LlmUsage {
|
|
1023
|
+
promptTokens: number;
|
|
1024
|
+
completionTokens: number;
|
|
1025
|
+
totalTokens: number;
|
|
1026
|
+
/** False when the provider omitted or malformed prompt/completion usage. */
|
|
1027
|
+
captured?: boolean;
|
|
1028
|
+
/** Proxies populate this when prompt caching is on. */
|
|
1029
|
+
cachedPromptTokens?: number;
|
|
1030
|
+
}
|
|
1031
|
+
interface LlmCallResult {
|
|
1032
|
+
/** The text content of the first choice. Empty string if none. */
|
|
1033
|
+
content: string;
|
|
1034
|
+
usage: LlmUsage;
|
|
1035
|
+
/**
|
|
1036
|
+
* Cost in USD. Pulled from proxy's `_response_cost` field when present;
|
|
1037
|
+
* `null` when neither the proxy nor the caller can derive it.
|
|
1038
|
+
*/
|
|
1039
|
+
costUsd: number | null;
|
|
1040
|
+
/** Model name actually used (echoed from response). */
|
|
1041
|
+
model: string;
|
|
1042
|
+
/** Wall-clock duration of the HTTP call (last attempt, if retried). */
|
|
1043
|
+
durationMs: number;
|
|
1044
|
+
/**
|
|
1045
|
+
* `finish_reason` echoed from the first choice (`stop`, `length`,
|
|
1046
|
+
* `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
|
|
1047
|
+
* Exposed so a free-form `callLlm` caller CAN detect a truncated answer
|
|
1048
|
+
* (`length`) instead of treating a cut-off completion as complete. Note:
|
|
1049
|
+
* `callLlm` does not itself reject on it — acting on this signal is the
|
|
1050
|
+
* caller's responsibility (in-repo free-form drivers do not yet enforce it).
|
|
1051
|
+
*/
|
|
1052
|
+
finishReason?: string | null;
|
|
1053
|
+
/**
|
|
1054
|
+
* True when `content.trim()` is empty. An empty completion is a silent zero
|
|
1055
|
+
* for free-form `callLlm` callers; this flag is the signal a caller can
|
|
1056
|
+
* inspect to fail loud rather than proceed on an empty string. `callLlm`
|
|
1057
|
+
* surfaces it but does not throw on it.
|
|
1058
|
+
*/
|
|
1059
|
+
contentEmpty?: boolean;
|
|
1060
|
+
/** Raw response body. */
|
|
1061
|
+
raw: Record<string, unknown>;
|
|
1062
|
+
}
|
|
1063
|
+
interface LlmClientOptions {
|
|
1064
|
+
/** Base URL (without trailing slash). Must end at the `/v1` prefix. */
|
|
1065
|
+
baseUrl?: string;
|
|
1066
|
+
/** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
|
|
1067
|
+
apiKey?: string;
|
|
1068
|
+
bearer?: string;
|
|
1069
|
+
/** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
|
|
1070
|
+
authHeader?: {
|
|
1071
|
+
name: string;
|
|
1072
|
+
value: string;
|
|
1073
|
+
};
|
|
1074
|
+
/** Stable provider idempotency key, reused across retries of this logical call. */
|
|
1075
|
+
idempotencyKey?: string;
|
|
1076
|
+
/** Default timeout in ms. Per-call can override. */
|
|
1077
|
+
defaultTimeoutMs?: number;
|
|
1078
|
+
/**
|
|
1079
|
+
* Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
|
|
1080
|
+
* each attempt's per-attempt timeout controller, so aborting it cancels
|
|
1081
|
+
* the in-flight fetch. A caller abort is FATAL: it is not retried even
|
|
1082
|
+
* though an AbortError otherwise matches the transient patterns.
|
|
1083
|
+
*/
|
|
1084
|
+
signal?: AbortSignal;
|
|
1085
|
+
/**
|
|
1086
|
+
* Cross-attempt wall-clock budget in ms, measured from the first attempt.
|
|
1087
|
+
* Before launching each attempt the loop checks the remaining budget and
|
|
1088
|
+
* stops retrying once it is exhausted, rather than waiting the full
|
|
1089
|
+
* per-attempt timeout on every retry. Bounds total time independent of
|
|
1090
|
+
* total attempts × `timeoutMs`.
|
|
1091
|
+
*/
|
|
1092
|
+
deadlineMs?: number;
|
|
1093
|
+
/** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
|
|
1094
|
+
maxRetries?: number;
|
|
1095
|
+
/** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
|
|
1096
|
+
fetch?: typeof fetch;
|
|
1097
|
+
/**
|
|
1098
|
+
* Optional raw HTTP capture sink. When provided, every request, response,
|
|
1099
|
+
* and error (across all retry attempts) is recorded to the sink, with auth
|
|
1100
|
+
* headers and credential-shaped body fields redacted by default. This is
|
|
1101
|
+
* the layer-1 forensics primitive: structured `LlmSpan`s record intent,
|
|
1102
|
+
* raw events record what actually crossed the wire.
|
|
1103
|
+
*/
|
|
1104
|
+
rawSink?: RawProviderSink;
|
|
1105
|
+
/**
|
|
1106
|
+
* Logical provider id attached to raw events. When omitted, derived from
|
|
1107
|
+
* `baseUrl` via `providerFromBaseUrl`.
|
|
1108
|
+
*/
|
|
1109
|
+
provider?: string;
|
|
1110
|
+
/** Trace context attached to raw events; populated by emitter-aware callers. */
|
|
1111
|
+
traceContext?: {
|
|
1112
|
+
runId?: string;
|
|
1113
|
+
spanId?: string;
|
|
1114
|
+
};
|
|
1115
|
+
/** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
|
|
1116
|
+
redactor?: ProviderRedactor;
|
|
1117
|
+
}
|
|
1118
|
+
|
|
1119
|
+
/**
|
|
1120
|
+
* Semantic concept judge — "does the built artifact actually implement
|
|
1121
|
+
* the features the user asked for?"
|
|
1122
|
+
*
|
|
1123
|
+
* Distinct from the domain/code/coherence judges in `judges.ts`:
|
|
1124
|
+
* - those judges score free-form conversational agent outputs along
|
|
1125
|
+
* quality dimensions (accuracy, depth, etc.)
|
|
1126
|
+
* - this judge scores a *built artifact* (served HTML + source files)
|
|
1127
|
+
* against an explicit list of expected concepts, returning per-concept
|
|
1128
|
+
* {present, score 0-10, evidence, severity}.
|
|
1129
|
+
*
|
|
1130
|
+
* The judge is strict about distinguishing (a) a working implementation
|
|
1131
|
+
* from (b) a keyword-present stub. "// TODO: mint button" is NOT present.
|
|
1132
|
+
* Only real, functional, wired-up code counts.
|
|
1133
|
+
*
|
|
1134
|
+
* Use via {@link createSemanticConceptJudge} or directly via
|
|
1135
|
+
* {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM
|
|
1136
|
+
* or JSON-parse errors so the caller can treat that as "layer skipped"
|
|
1137
|
+
* rather than "layer failed" in a multi-layer pipeline.
|
|
1138
|
+
*/
|
|
1139
|
+
|
|
1140
|
+
/**
|
|
1141
|
+
* Implementation complexity class for weighted scoring.
|
|
1142
|
+
*
|
|
1143
|
+
* - `render` (default): the concept is a UI surface that displays static
|
|
1144
|
+
* data — render a list, show a counter, lay out a button. Single-file
|
|
1145
|
+
* work, no external integration.
|
|
1146
|
+
* - `integrate`: the concept requires wiring a real external system —
|
|
1147
|
+
* wallet connect (wagmi + RainbowKit + chain config), payment provider
|
|
1148
|
+
* (Stripe Elements + intent + webhook), an API client with auth.
|
|
1149
|
+
* Multi-file, library-knowledge, runtime correctness matters.
|
|
1150
|
+
* - `compute`: the concept requires algorithmic work — solver, simulator,
|
|
1151
|
+
* constraint propagation, ML inference. Correctness > UI polish.
|
|
1152
|
+
*
|
|
1153
|
+
* Default weights (when applied via `weightConcepts: 'complexity'`):
|
|
1154
|
+
* render=1.0, integrate=2.0, compute=2.5
|
|
1155
|
+
*
|
|
1156
|
+
* Cross-vertical scoring without complexity weighting silently inflates
|
|
1157
|
+
* the rate of UI-heavy verticals (healthcare, fintech dashboards) vs
|
|
1158
|
+
* integration-heavy verticals (DeFi, wallets) — all concepts treated
|
|
1159
|
+
* equally even though the agent does 2-3x the work for `integrate`.
|
|
1160
|
+
*/
|
|
1161
|
+
type ConceptComplexity = 'render' | 'integrate' | 'compute';
|
|
1162
|
+
interface ConceptSpec {
|
|
1163
|
+
name: string;
|
|
1164
|
+
/** Short hints that help the judge; not used for matching. */
|
|
1165
|
+
keywords?: string[];
|
|
1166
|
+
/** Optional explicit weight; default 1.0. Overrides complexity-derived weight. */
|
|
1167
|
+
weight?: number;
|
|
1168
|
+
/** Implementation complexity class. Default `render`. */
|
|
1169
|
+
complexity?: ConceptComplexity;
|
|
1170
|
+
}
|
|
1171
|
+
interface SemanticConceptJudgeInput {
|
|
1172
|
+
/** Full natural-language prompt the agent was handed. */
|
|
1173
|
+
userRequest: string;
|
|
1174
|
+
/** Rendered HTML the preview returns (UI artifacts). Optional. */
|
|
1175
|
+
servedHtml?: string;
|
|
1176
|
+
/** Top-level source files from the agent's workdir. */
|
|
1177
|
+
sourceFiles: Array<{
|
|
1178
|
+
path: string;
|
|
1179
|
+
content: string;
|
|
1180
|
+
}>;
|
|
1181
|
+
/** The expected concept list. */
|
|
1182
|
+
expectedConcepts: ConceptSpec[];
|
|
1183
|
+
/** Free-form metadata (id, difficulty) to inject into the prompt. */
|
|
1184
|
+
artifactLabel?: string;
|
|
1185
|
+
artifactDescription?: string;
|
|
1186
|
+
}
|
|
1187
|
+
/**
|
|
1188
|
+
* Score-aggregation strategy. `mean` averages 0-10 scores uniformly.
|
|
1189
|
+
* `complexity` applies the default weight table (render=1, integrate=2,
|
|
1190
|
+
* compute=2.5) unless a concept has an explicit `weight`. `explicit`
|
|
1191
|
+
* honors only `weight` (defaulting to 1 for unspecified).
|
|
1192
|
+
*/
|
|
1193
|
+
type ConceptWeightStrategy = 'mean' | 'complexity' | 'explicit';
|
|
1194
|
+
interface SemanticConceptJudgeOptions {
|
|
1195
|
+
/** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */
|
|
1196
|
+
model?: string;
|
|
1197
|
+
/** Per-call timeout. Default 300s. */
|
|
1198
|
+
timeoutMs?: number;
|
|
1199
|
+
/** Provider-enforced output limit. Default 16000. */
|
|
1200
|
+
maxTokens?: number;
|
|
1201
|
+
/** Pipeline budget for the prompt (source blob truncation). Default 45000. */
|
|
1202
|
+
maxSourceChars?: number;
|
|
1203
|
+
/** Per-file cap before inclusion. Default 20000. */
|
|
1204
|
+
maxPerFileChars?: number;
|
|
1205
|
+
/** HTML cap. Default 30000. */
|
|
1206
|
+
maxHtmlChars?: number;
|
|
1207
|
+
/** LlmClient config (baseUrl, apiKey, authHeader, …). */
|
|
1208
|
+
llm?: LlmClientOptions;
|
|
1209
|
+
costLedger?: CostLedgerHandle;
|
|
1210
|
+
costPhase?: string;
|
|
1211
|
+
signal?: AbortSignal;
|
|
1212
|
+
/**
|
|
1213
|
+
* Score aggregation strategy. Default `mean` — uniform average across
|
|
1214
|
+
* concepts. Cross-vertical comparisons should use `complexity` to
|
|
1215
|
+
* neutralize the integrate-vs-render asymmetry.
|
|
1216
|
+
*/
|
|
1217
|
+
weightConcepts?: ConceptWeightStrategy;
|
|
1218
|
+
/** Override the default complexity → weight table. */
|
|
1219
|
+
complexityWeights?: Partial<Record<ConceptComplexity, number>>;
|
|
1220
|
+
}
|
|
1221
|
+
|
|
1222
|
+
interface Scenario {
|
|
1223
|
+
id: string;
|
|
1224
|
+
persona: string;
|
|
1225
|
+
label: string;
|
|
1226
|
+
thesis: string;
|
|
1227
|
+
dimensions: string[];
|
|
1228
|
+
turns: Turn[];
|
|
1229
|
+
artifactChecks: ArtifactCheck[];
|
|
1230
|
+
systemPromptAppend?: string;
|
|
1231
|
+
}
|
|
1232
|
+
interface Turn {
|
|
1233
|
+
user: string;
|
|
1234
|
+
expectedBehaviors: string[];
|
|
1235
|
+
adversarial?: boolean;
|
|
1236
|
+
feedbackType?: 'correction' | 'rejection' | 'vague' | 'contradictory' | 'escalation';
|
|
1237
|
+
}
|
|
1238
|
+
interface ArtifactCheck {
|
|
1239
|
+
type: 'vault_file_exists' | 'vault_file_contains' | 'block_extracted' | 'code_valid' | 'generation_produced' | 'tool_created' | string;
|
|
1240
|
+
target: string;
|
|
1241
|
+
contains?: string;
|
|
1242
|
+
minCount?: number;
|
|
1243
|
+
description: string;
|
|
1244
|
+
}
|
|
1245
|
+
interface TurnResult {
|
|
1246
|
+
turnIndex: number;
|
|
1247
|
+
userMessage: string;
|
|
1248
|
+
agentResponse: string;
|
|
1249
|
+
durationMs: number;
|
|
1250
|
+
blocksExtracted: {
|
|
1251
|
+
type: string;
|
|
1252
|
+
title: string;
|
|
1253
|
+
}[];
|
|
1254
|
+
containsCode: boolean;
|
|
1255
|
+
containsToolCall: boolean;
|
|
1256
|
+
}
|
|
1257
|
+
interface JudgeScore {
|
|
1258
|
+
judgeName: string;
|
|
1259
|
+
dimension: string;
|
|
1260
|
+
score: number;
|
|
1261
|
+
reasoning: string;
|
|
1262
|
+
evidence?: string;
|
|
1263
|
+
}
|
|
1264
|
+
interface CollectedArtifacts {
|
|
1265
|
+
vaultFiles: {
|
|
1266
|
+
path: string;
|
|
1267
|
+
content: string;
|
|
1268
|
+
}[];
|
|
1269
|
+
blocksExtracted: {
|
|
1270
|
+
type: string;
|
|
1271
|
+
fields: Record<string, string>;
|
|
1272
|
+
}[];
|
|
1273
|
+
codeBlocks: {
|
|
1274
|
+
language: string;
|
|
1275
|
+
code: string;
|
|
1276
|
+
}[];
|
|
1277
|
+
toolCalls: string[];
|
|
1278
|
+
}
|
|
1279
|
+
interface JudgeInput {
|
|
1280
|
+
scenario: Scenario;
|
|
1281
|
+
turns: TurnResult[];
|
|
1282
|
+
artifacts: CollectedArtifacts;
|
|
1283
|
+
/** Shared ledger for paid built-in judges. Direct calls default to an uncapped ledger. */
|
|
1284
|
+
costLedger?: CostLedgerHandle;
|
|
1285
|
+
costPhase?: string;
|
|
1286
|
+
costTags?: Record<string, string>;
|
|
1287
|
+
signal?: AbortSignal;
|
|
1288
|
+
/** Exact maximum provider attempts configured on the supplied TCloud client. */
|
|
1289
|
+
tcloudMaximumAttempts?: number;
|
|
1290
|
+
}
|
|
1291
|
+
type JudgeFn = (tc: TCloud, input: JudgeInput) => Promise<JudgeScore[]>;
|
|
1292
|
+
|
|
1293
|
+
/**
|
|
1294
|
+
* Shared types for the trace-analyst module.
|
|
1295
|
+
*
|
|
1296
|
+
* Wire format. The store interface speaks `OtlpSpanLike` rows — one JSONL
|
|
1297
|
+
* line per span, OTLP-shaped. We do NOT depend on a specific tracing
|
|
1298
|
+
* vendor at the type level. Adapter
|
|
1299
|
+
* layers map upstream shapes onto this interface.
|
|
1300
|
+
*
|
|
1301
|
+
* Design constraint. Every read operation that can return arbitrary
|
|
1302
|
+
* payload must carry a byte budget so the agent's tool result stays
|
|
1303
|
+
* bounded regardless of input trace size. Oversized responses
|
|
1304
|
+
* substitute a deterministic summary instead of bytes — see
|
|
1305
|
+
* `ViewTraceOversized`.
|
|
1306
|
+
*/
|
|
1307
|
+
/** OTLP span kind (subset we actually use). */
|
|
1308
|
+
type TraceAnalystSpanKind = 'AGENT' | 'LLM' | 'TOOL' | 'CHAIN' | 'GUARDRAIL' | 'SPAN' | 'UNKNOWN';
|
|
1309
|
+
type TraceAnalystSpanStatus = 'OK' | 'ERROR' | 'UNSET';
|
|
1310
|
+
/** Subset of OTLP span fields the analyst exposes to the agent. The
|
|
1311
|
+
* store's job is to project upstream's full span shape down to this
|
|
1312
|
+
* view — the analyst never sees vendor extensions directly. */
|
|
1313
|
+
interface TraceAnalystSpan {
|
|
1314
|
+
trace_id: string;
|
|
1315
|
+
span_id: string;
|
|
1316
|
+
parent_span_id: string | null;
|
|
1317
|
+
name: string;
|
|
1318
|
+
kind: TraceAnalystSpanKind;
|
|
1319
|
+
start_time: string;
|
|
1320
|
+
end_time: string;
|
|
1321
|
+
duration_ms: number;
|
|
1322
|
+
status: TraceAnalystSpanStatus;
|
|
1323
|
+
status_message?: string;
|
|
1324
|
+
service_name: string | null;
|
|
1325
|
+
agent_name: string | null;
|
|
1326
|
+
model_name: string | null;
|
|
1327
|
+
tool_name: string | null;
|
|
1328
|
+
/** Raw JSON-serialisable attribute map. May contain large strings;
|
|
1329
|
+
* callers must respect the per-attribute byte cap. */
|
|
1330
|
+
attributes: Record<string, unknown>;
|
|
1331
|
+
}
|
|
1332
|
+
interface TraceAnalystTraceSummary {
|
|
1333
|
+
trace_id: string;
|
|
1334
|
+
service_name: string | null;
|
|
1335
|
+
agent_name: string | null;
|
|
1336
|
+
span_count: number;
|
|
1337
|
+
has_errors: boolean;
|
|
1338
|
+
start_time: string;
|
|
1339
|
+
end_time: string;
|
|
1340
|
+
duration_ms: number;
|
|
1341
|
+
raw_jsonl_bytes: number;
|
|
1342
|
+
models: string[];
|
|
1343
|
+
tools: string[];
|
|
1344
|
+
}
|
|
1345
|
+
interface TraceAnalystFilters {
|
|
1346
|
+
/** Restrict to traces that contain at least one error span. */
|
|
1347
|
+
has_errors?: boolean;
|
|
1348
|
+
/** Match if any span's `service.name` is in this list. */
|
|
1349
|
+
service_names?: string[];
|
|
1350
|
+
/** Match if any span's `agent.name` is in this list. */
|
|
1351
|
+
agent_names?: string[];
|
|
1352
|
+
/** Match if any LLM span's `llm.model_name` is in this list. */
|
|
1353
|
+
model_names?: string[];
|
|
1354
|
+
/** Match if any tool span's `tool.name` is in this list. */
|
|
1355
|
+
tool_names?: string[];
|
|
1356
|
+
/** ISO-8601 lower bound on the trace's earliest start time. */
|
|
1357
|
+
start_time_after?: string;
|
|
1358
|
+
/** ISO-8601 upper bound on the trace's earliest start time. */
|
|
1359
|
+
start_time_before?: string;
|
|
1360
|
+
/** Single regex applied to raw JSONL bytes for the trace. Opt-in;
|
|
1361
|
+
* expensive on large datasets. Use the indexed filters above first. */
|
|
1362
|
+
regex_pattern?: string;
|
|
1363
|
+
}
|
|
1364
|
+
/** One distinct error signature across the dataset — the deterministic unit of
|
|
1365
|
+
* failure coverage. Signatures normalize volatile tokens (digits, hex/uuids,
|
|
1366
|
+
* paths, durations) out of the span `status_message` so semantically identical
|
|
1367
|
+
* failures collapse into one cluster. An analyst that accounts for every
|
|
1368
|
+
* cluster has, by construction, covered every distinct failure mode. */
|
|
1369
|
+
interface ErrorCluster {
|
|
1370
|
+
/** Normalized status_message — the cluster key. */
|
|
1371
|
+
signature: string;
|
|
1372
|
+
/** A verbatim, un-normalized exemplar message (for exact-string citation). */
|
|
1373
|
+
status_message_sample: string;
|
|
1374
|
+
/** The span name that most often carries this signature, if any. */
|
|
1375
|
+
span_name: string | null;
|
|
1376
|
+
/** The tool that most often carries this signature, if any. */
|
|
1377
|
+
tool_name: string | null;
|
|
1378
|
+
trace_count: number;
|
|
1379
|
+
span_count: number;
|
|
1380
|
+
/** trace_count / total error traces in the matched set (0..1). */
|
|
1381
|
+
prevalence: number;
|
|
1382
|
+
/** Real trace ids carrying this signature (capped), passable to view/search. */
|
|
1383
|
+
exemplar_trace_ids: string[];
|
|
1384
|
+
/** Real span ids carrying this signature (capped). */
|
|
1385
|
+
exemplar_span_ids: string[];
|
|
1386
|
+
}
|
|
1387
|
+
interface DatasetOverview {
|
|
1388
|
+
total_traces: number;
|
|
1389
|
+
raw_jsonl_bytes: number;
|
|
1390
|
+
services: string[];
|
|
1391
|
+
agents: string[];
|
|
1392
|
+
models: string[];
|
|
1393
|
+
tool_names: string[];
|
|
1394
|
+
/** Up to 20 real trace ids the agent may pass to view/search tools. */
|
|
1395
|
+
sample_trace_ids: string[];
|
|
1396
|
+
errors: {
|
|
1397
|
+
trace_count: number;
|
|
1398
|
+
span_count: number;
|
|
1399
|
+
};
|
|
1400
|
+
/** The COMPLETE deterministic error-signature population, sorted by
|
|
1401
|
+
* trace_count desc. This is the failure-coverage checklist: an analysis is
|
|
1402
|
+
* complete only when every cluster here is accounted for. Empty when the
|
|
1403
|
+
* matched set has no error spans. */
|
|
1404
|
+
error_clusters: ErrorCluster[];
|
|
1405
|
+
time_range: {
|
|
1406
|
+
earliest: string;
|
|
1407
|
+
latest: string;
|
|
1408
|
+
} | null;
|
|
1409
|
+
}
|
|
1410
|
+
interface QueryTracesPage {
|
|
1411
|
+
traces: TraceAnalystTraceSummary[];
|
|
1412
|
+
total: number;
|
|
1413
|
+
has_more: boolean;
|
|
1414
|
+
}
|
|
1415
|
+
/** Full-trace view. When the response would exceed the per-call byte
|
|
1416
|
+
* budget, `oversized` is populated INSTEAD of `spans` so the agent
|
|
1417
|
+
* knows to switch to `searchTrace` / `viewSpans`. */
|
|
1418
|
+
interface ViewTraceResult {
|
|
1419
|
+
trace_id: string;
|
|
1420
|
+
spans?: TraceAnalystSpan[];
|
|
1421
|
+
oversized?: ViewTraceOversized;
|
|
1422
|
+
}
|
|
1423
|
+
interface ViewTraceOversized {
|
|
1424
|
+
span_count: number;
|
|
1425
|
+
/** Names with their counts, sorted desc. Capped at 20 entries. */
|
|
1426
|
+
top_span_names: Array<[string, number]>;
|
|
1427
|
+
/** Largest single span body (bytes after attribute-cap projection). */
|
|
1428
|
+
span_response_bytes_max: number;
|
|
1429
|
+
error_span_count: number;
|
|
1430
|
+
}
|
|
1431
|
+
interface ViewSpansResult {
|
|
1432
|
+
trace_id: string;
|
|
1433
|
+
spans: TraceAnalystSpan[];
|
|
1434
|
+
/** Number of requested span ids that were not found in the trace. */
|
|
1435
|
+
missing_span_ids: string[];
|
|
1436
|
+
/** Number of attribute fields truncated to fit the per-attribute cap. */
|
|
1437
|
+
truncated_attribute_count: number;
|
|
1438
|
+
}
|
|
1439
|
+
interface SpanMatchRecord {
|
|
1440
|
+
trace_id: string;
|
|
1441
|
+
span_id: string;
|
|
1442
|
+
span_name: string;
|
|
1443
|
+
span_kind: TraceAnalystSpanKind;
|
|
1444
|
+
/** JSON pointer-style path to the matched value, e.g.
|
|
1445
|
+
* `attributes."llm.input_messages"[2].content`. */
|
|
1446
|
+
attribute_path: string;
|
|
1447
|
+
matched_text: string;
|
|
1448
|
+
context_before: string;
|
|
1449
|
+
context_after: string;
|
|
1450
|
+
match_offset: number;
|
|
1451
|
+
}
|
|
1452
|
+
interface SearchTraceResult {
|
|
1453
|
+
trace_id: string;
|
|
1454
|
+
hits: SpanMatchRecord[];
|
|
1455
|
+
total_matches: number;
|
|
1456
|
+
has_more: boolean;
|
|
1457
|
+
}
|
|
1458
|
+
interface SearchSpanResult {
|
|
1459
|
+
trace_id: string;
|
|
1460
|
+
span_id: string;
|
|
1461
|
+
hits: SpanMatchRecord[];
|
|
1462
|
+
total_matches: number;
|
|
1463
|
+
has_more: boolean;
|
|
1464
|
+
}
|
|
1465
|
+
|
|
1466
|
+
/**
|
|
1467
|
+
* `TraceAnalysisStore` — read-side interface the trace-analyst calls
|
|
1468
|
+
* through. Six operations, all bounded:
|
|
1469
|
+
*
|
|
1470
|
+
* - `getOverview(filters?)` — dataset rollup + sample trace ids.
|
|
1471
|
+
* - `queryTraces(filters?, limit, offset)` — paginated summaries.
|
|
1472
|
+
* - `countTraces(filters?)` — cheap count without materialisation.
|
|
1473
|
+
* - `viewTrace(trace_id, perAttrCap)` — full span list, oversized → summary.
|
|
1474
|
+
* - `viewSpans(trace_id, span_ids, perAttrCap)` — surgical span fetch.
|
|
1475
|
+
* - `searchTrace(trace_id, regex, max_matches)` — bounded regex hits.
|
|
1476
|
+
* - `searchSpan(trace_id, span_id, regex, max_matches)` — single-span search.
|
|
1477
|
+
*
|
|
1478
|
+
* Multiple implementations ship in the core (`OtlpFileTraceStore`).
|
|
1479
|
+
* Downstream callers can supply their own — e.g. a DuckDB-backed
|
|
1480
|
+
* adapter or an in-memory adapter for tests — by implementing this
|
|
1481
|
+
* interface.
|
|
1482
|
+
*
|
|
1483
|
+
* Filters compose with AND semantics. Empty/undefined fields impose
|
|
1484
|
+
* no constraint. `regex_pattern` is the only opt-in raw-bytes scan —
|
|
1485
|
+
* implementations may skip it via `count`/`overview` when not set.
|
|
1486
|
+
*/
|
|
1487
|
+
|
|
1488
|
+
interface TraceAnalysisStore {
|
|
1489
|
+
getOverview(filters?: TraceAnalystFilters): Promise<DatasetOverview>;
|
|
1490
|
+
queryTraces(opts: {
|
|
1491
|
+
filters?: TraceAnalystFilters;
|
|
1492
|
+
limit: number;
|
|
1493
|
+
offset?: number;
|
|
1494
|
+
}): Promise<QueryTracesPage>;
|
|
1495
|
+
countTraces(filters?: TraceAnalystFilters): Promise<number>;
|
|
1496
|
+
viewTrace(opts: {
|
|
1497
|
+
trace_id: string;
|
|
1498
|
+
/** Override per-attribute byte cap. Defaults to discovery budget. */
|
|
1499
|
+
per_attribute_byte_cap?: number;
|
|
1500
|
+
}): Promise<ViewTraceResult>;
|
|
1501
|
+
viewSpans(opts: {
|
|
1502
|
+
trace_id: string;
|
|
1503
|
+
span_ids: readonly string[];
|
|
1504
|
+
/** Override per-attribute byte cap. Defaults to surgical budget. */
|
|
1505
|
+
per_attribute_byte_cap?: number;
|
|
1506
|
+
}): Promise<ViewSpansResult>;
|
|
1507
|
+
searchTrace(opts: {
|
|
1508
|
+
trace_id: string;
|
|
1509
|
+
regex_pattern: string;
|
|
1510
|
+
/** Hard cap on matches returned. Default 50. */
|
|
1511
|
+
max_matches?: number;
|
|
1512
|
+
}): Promise<SearchTraceResult>;
|
|
1513
|
+
searchSpan(opts: {
|
|
1514
|
+
trace_id: string;
|
|
1515
|
+
span_id: string;
|
|
1516
|
+
regex_pattern: string;
|
|
1517
|
+
max_matches?: number;
|
|
1518
|
+
}): Promise<SearchSpanResult>;
|
|
1519
|
+
}
|
|
1520
|
+
|
|
1521
|
+
/**
|
|
1522
|
+
* ChatClient — the single LLM abstraction analysts call.
|
|
1523
|
+
*
|
|
1524
|
+
* agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
|
|
1525
|
+
* graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
|
|
1526
|
+
* mixed patterns force every analyst author to pick a transport, which
|
|
1527
|
+
* couples analyst code to runtime concerns (cli-bridge vs router vs
|
|
1528
|
+
* sandbox-sdk) it shouldn't know about.
|
|
1529
|
+
*
|
|
1530
|
+
* `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
|
|
1531
|
+
* The operator decides at the registry boundary which transport binds
|
|
1532
|
+
* to it. Analyst code stays transport-agnostic; swapping production
|
|
1533
|
+
* (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
|
|
1534
|
+
* line factory call.
|
|
1535
|
+
*
|
|
1536
|
+
* Designed to coexist: existing `LlmClient` callers and existing
|
|
1537
|
+
* `TCloud`-based judges keep working untouched. New analyst code uses
|
|
1538
|
+
* `ChatClient`. When old call sites migrate, they pick up budgeting,
|
|
1539
|
+
* cancellation, and unified telemetry for free.
|
|
1540
|
+
*/
|
|
1541
|
+
|
|
1542
|
+
/**
|
|
1543
|
+
* Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
|
|
1544
|
+
* compatible mental model stays. Two methods: a one-shot `chat()` and
|
|
1545
|
+
* an `streamChat()` for future agentic loops (not yet exposed).
|
|
1546
|
+
*/
|
|
1547
|
+
interface ChatClient {
|
|
1548
|
+
/** Display name of the bound transport — included in telemetry. */
|
|
1549
|
+
readonly transport: ChatTransport;
|
|
1550
|
+
/** Default model when caller omits — operators bind this per environment. */
|
|
1551
|
+
readonly defaultModel?: string;
|
|
1552
|
+
/** Total provider attempts this transport can make for one chat call. */
|
|
1553
|
+
readonly maximumAttempts?: number;
|
|
1554
|
+
/** Implementations must enforce `req.maxTokens` when it is present. */
|
|
1555
|
+
chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
|
|
1556
|
+
}
|
|
1557
|
+
type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
|
|
1558
|
+
interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
|
|
1559
|
+
/** Optional — falls back to ChatClient.defaultModel. */
|
|
1560
|
+
model?: string;
|
|
1561
|
+
}
|
|
1562
|
+
type ChatResponse = LlmCallResult;
|
|
1563
|
+
interface ChatCallOpts {
|
|
1564
|
+
/** Cancel the in-flight request. */
|
|
1565
|
+
signal?: AbortSignal;
|
|
1566
|
+
/** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
|
|
1567
|
+
maxCostUsd?: number;
|
|
1568
|
+
/** Correlation tag carried into request headers when the transport allows. */
|
|
1569
|
+
correlationId?: string;
|
|
1570
|
+
/** Stable provider idempotency key for retries/redrives of one paid call. */
|
|
1571
|
+
idempotencyKey?: string;
|
|
1572
|
+
}
|
|
1573
|
+
type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
|
|
1574
|
+
interface BaseTransportOpts {
|
|
1575
|
+
defaultModel?: string;
|
|
1576
|
+
/** Total provider attempts. Required for opaque transports used in capped runs. */
|
|
1577
|
+
maximumAttempts?: number;
|
|
1578
|
+
}
|
|
1579
|
+
interface RouterTransportOpts extends BaseTransportOpts {
|
|
1580
|
+
transport: 'router';
|
|
1581
|
+
baseUrl?: string;
|
|
1582
|
+
apiKey: string;
|
|
1583
|
+
}
|
|
1584
|
+
interface CliBridgeTransportOpts extends BaseTransportOpts {
|
|
1585
|
+
transport: 'cli-bridge';
|
|
1586
|
+
baseUrl?: string;
|
|
1587
|
+
bearer?: string;
|
|
1588
|
+
}
|
|
1589
|
+
interface DirectProviderTransportOpts extends BaseTransportOpts {
|
|
1590
|
+
transport: 'direct-provider';
|
|
1591
|
+
baseUrl: string;
|
|
1592
|
+
apiKey: string;
|
|
1593
|
+
}
|
|
1594
|
+
/**
|
|
1595
|
+
* Sandbox-SDK transport. Provided as a thin pass-through: the caller
|
|
1596
|
+
* supplies a callable that mimics LlmClient.chat() against an already-
|
|
1597
|
+
* configured Sandbox handle. We don't import the SDK here to keep
|
|
1598
|
+
* agent-eval dep-free of @tangle-network/sandbox.
|
|
1599
|
+
*/
|
|
1600
|
+
interface SandboxSdkTransportOpts extends BaseTransportOpts {
|
|
1601
|
+
transport: 'sandbox-sdk';
|
|
1602
|
+
chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1603
|
+
}
|
|
1604
|
+
/**
|
|
1605
|
+
* Mock transport for tests. The handler receives the request and returns
|
|
1606
|
+
* whatever the test wants. No retries, no JSON-schema degrade.
|
|
1607
|
+
*/
|
|
1608
|
+
interface MockTransportOpts extends BaseTransportOpts {
|
|
1609
|
+
transport: 'mock';
|
|
1610
|
+
handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1611
|
+
}
|
|
1612
|
+
/**
|
|
1613
|
+
* Build a ChatClient bound to a specific transport. The returned client
|
|
1614
|
+
* is safe to share across analysts in a single registry run.
|
|
1615
|
+
*/
|
|
1616
|
+
declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
|
|
1617
|
+
|
|
1618
|
+
/**
|
|
1619
|
+
* Analyst contract — the missing orchestration layer over agent-eval's
|
|
1620
|
+
* existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
|
|
1621
|
+
* SemanticConceptJudge, JudgeFn, ...).
|
|
1622
|
+
*
|
|
1623
|
+
* Each existing primitive returns its own output shape. The Analyst
|
|
1624
|
+
* contract is the single envelope every primitive lifts into, so a
|
|
1625
|
+
* registry can run N analysts against a run and a single renderer can
|
|
1626
|
+
* compose findings without knowing which analyzer produced them.
|
|
1627
|
+
*
|
|
1628
|
+
* The contract is intentionally domain-agnostic: nothing here knows
|
|
1629
|
+
* about code, voice, RAG, or any particular agent stack. Analysts
|
|
1630
|
+
* declare what INPUT KIND they need (a trace store, an artifact dir,
|
|
1631
|
+
* a RunRecord, a JudgeInput, or `custom`), and the registry routes
|
|
1632
|
+
* the matching input from `AnalystRunInputs`.
|
|
1633
|
+
*/
|
|
1634
|
+
|
|
1635
|
+
/**
|
|
1636
|
+
* Unified envelope every analyst emits. Schema-versioned so renderers
|
|
1637
|
+
* and time-series diffs survive future field additions.
|
|
1638
|
+
*/
|
|
1639
|
+
interface AnalystFinding {
|
|
1640
|
+
schema_version: '1.0.0';
|
|
1641
|
+
/**
|
|
1642
|
+
* Stable hash over identity-defining fields (analyst_id + canonical
|
|
1643
|
+
* claim + area + optional subject). Two findings from two runs that
|
|
1644
|
+
* "are the same finding" share this id — that's what `diffFindings`
|
|
1645
|
+
* uses to compute appeared/disappeared sets across runs.
|
|
1646
|
+
*/
|
|
1647
|
+
finding_id: string;
|
|
1648
|
+
analyst_id: string;
|
|
1649
|
+
produced_at: string;
|
|
1650
|
+
severity: AnalystSeverity;
|
|
1651
|
+
/**
|
|
1652
|
+
* Coarse classification. Renderers group by this. Free-form so
|
|
1653
|
+
* domain-specific analysts can introduce categories without a
|
|
1654
|
+
* schema change ('agent-reasoning', 'verification', 'cost',
|
|
1655
|
+
* 'tool-use', 'safety', 'latency', 'data-quality', ...).
|
|
1656
|
+
*/
|
|
1657
|
+
area: string;
|
|
1658
|
+
claim: string;
|
|
1659
|
+
rationale?: string;
|
|
1660
|
+
evidence_refs: EvidenceRef[];
|
|
1661
|
+
recommended_action?: string;
|
|
1662
|
+
validation_plan?: string;
|
|
1663
|
+
/** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
|
|
1664
|
+
confidence: number;
|
|
1665
|
+
/**
|
|
1666
|
+
* Optional subject the finding is about — leaf id, agent id, request
|
|
1667
|
+
* id. Included in finding_id when present so per-subject findings
|
|
1668
|
+
* diff cleanly across runs.
|
|
1669
|
+
*/
|
|
1670
|
+
subject?: string;
|
|
1671
|
+
/** FIREWALL provenance (docs/learning-flywheel.md): true iff this finding was
|
|
1672
|
+
* lifted from a JUDGE verdict (an acceptance score), not OBSERVED from the
|
|
1673
|
+
* agent's behavior. A judge-derived finding must NEVER be admitted as a
|
|
1674
|
+
* steering input — that is the held-out judge leaking into the loop. Set at
|
|
1675
|
+
* the lift site (createJudgeAdapter); checked by `assertNoJudgeVerdict`.
|
|
1676
|
+
* Provenance, not evidence presence, is the correct discriminator: an
|
|
1677
|
+
* evidence-less trace-analyst observation legitimately steers, while a judge
|
|
1678
|
+
* verdict that happens to cite an artifact must not. */
|
|
1679
|
+
derived_from_judge?: boolean;
|
|
1680
|
+
/** Analyst-private extras; renderers ignore unless they know the analyst. */
|
|
1681
|
+
metadata?: Record<string, unknown>;
|
|
1682
|
+
}
|
|
1683
|
+
type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
|
|
1684
|
+
interface EvidenceRef {
|
|
1685
|
+
/**
|
|
1686
|
+
* Where the evidence lives. `span` and `event` refer to OTLP trace
|
|
1687
|
+
* elements; `artifact` to a file inside the run's artifact tree;
|
|
1688
|
+
* `finding` to another AnalystFinding (cross-analyst chaining);
|
|
1689
|
+
* `metric` to a named scalar reading the renderer knows how to read.
|
|
1690
|
+
*/
|
|
1691
|
+
kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
|
|
1692
|
+
uri: string;
|
|
1693
|
+
excerpt?: string;
|
|
1694
|
+
}
|
|
1695
|
+
/**
|
|
1696
|
+
* The discriminator the registry uses to pass the right input.
|
|
1697
|
+
* `custom` is the escape hatch — analysts that need something else
|
|
1698
|
+
* (e.g. an embedding cache, a partner SDK handle) read it from
|
|
1699
|
+
* `AnalystRunInputs.custom[<analyst id>]`.
|
|
1700
|
+
*/
|
|
1701
|
+
type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
|
|
1702
|
+
interface AnalystCost {
|
|
1703
|
+
/** `deterministic` analysts MUST NOT call the LLM. */
|
|
1704
|
+
kind: 'deterministic' | 'llm';
|
|
1705
|
+
/** Optional declared upper bound; the registry can enforce a budget. */
|
|
1706
|
+
est_usd_per_run?: number;
|
|
1707
|
+
/** Models the analyst expects to use (informational). */
|
|
1708
|
+
models?: string[];
|
|
1709
|
+
/** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */
|
|
1710
|
+
settlement_timeout_ms?: number;
|
|
1711
|
+
}
|
|
1712
|
+
interface AnalystRequirements {
|
|
1713
|
+
/** Min number of shots / samples the analyst needs to produce signal. */
|
|
1714
|
+
min_shots?: number;
|
|
1715
|
+
/** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
|
|
1716
|
+
capabilities?: string[];
|
|
1717
|
+
}
|
|
1718
|
+
/**
|
|
1719
|
+
* What's passed to every analyst call. The registry resolves which
|
|
1720
|
+
* field the analyst's `inputKind` selects and asserts it's present.
|
|
1721
|
+
*/
|
|
1722
|
+
interface AnalystRunInputs {
|
|
1723
|
+
traceStore?: TraceAnalysisStore;
|
|
1724
|
+
artifactDir?: string;
|
|
1725
|
+
runRecord?: RunRecord;
|
|
1726
|
+
judgeInput?: JudgeInput;
|
|
1727
|
+
/** Keyed by analyst id; populated by callers that registered custom analysts. */
|
|
1728
|
+
custom?: Record<string, unknown>;
|
|
1729
|
+
}
|
|
1730
|
+
interface AnalystContext {
|
|
1731
|
+
runId: string;
|
|
1732
|
+
/** Stable correlation id so logs from a single registry.run() share a tag. */
|
|
1733
|
+
correlationId: string;
|
|
1734
|
+
/** Enforced wall-clock deadline (epoch ms). */
|
|
1735
|
+
deadlineMs?: number;
|
|
1736
|
+
/** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
|
|
1737
|
+
budgetUsd?: number;
|
|
1738
|
+
/** Shared paid-call account when the analyst runs inside a larger campaign. */
|
|
1739
|
+
costLedger?: CostLedgerHandle;
|
|
1740
|
+
/** Attribution phase used when writing to the shared paid-call account. */
|
|
1741
|
+
costPhase?: string;
|
|
1742
|
+
/**
|
|
1743
|
+
* Shared chat client. Analysts that call an LLM go through this so
|
|
1744
|
+
* the operator picks transport (sandbox-sdk | router | cli-bridge |
|
|
1745
|
+
* direct-provider | mock) at the registry boundary without touching
|
|
1746
|
+
* analyst code.
|
|
1747
|
+
*/
|
|
1748
|
+
chat?: ChatClient;
|
|
1749
|
+
/**
|
|
1750
|
+
* Findings from a prior run the operator wants the analyst to see as
|
|
1751
|
+
* retrieval context. Kinds that take advantage of cross-run memory
|
|
1752
|
+
* (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
|
|
1753
|
+
* page I asked for is still missing") render these into the actor's
|
|
1754
|
+
* working set. Filtering is the operator's job: pass the slice that
|
|
1755
|
+
* matches the analyst's id, or pass everything and let the kind
|
|
1756
|
+
* filter. Empty / absent means no cross-run context.
|
|
1757
|
+
*/
|
|
1758
|
+
priorFindings?: ReadonlyArray<AnalystFinding>;
|
|
1759
|
+
/**
|
|
1760
|
+
* Findings emitted by analysts that completed earlier in this registry run.
|
|
1761
|
+
* This is separate from `priorFindings`: upstream findings are dependency
|
|
1762
|
+
* context for the current pass, while prior findings are cross-run memory.
|
|
1763
|
+
* The registry populates this only when `RegistryRunOpts.chainFindings` is on.
|
|
1764
|
+
*/
|
|
1765
|
+
upstreamFindings?: ReadonlyArray<AnalystFinding>;
|
|
1766
|
+
/**
|
|
1767
|
+
* Report metered work independently of findings. This keeps an empty finding
|
|
1768
|
+
* set from erasing token/cost telemetry. Multiple receipts are accumulated.
|
|
1769
|
+
*/
|
|
1770
|
+
recordUsage?: (receipt: AnalystUsageReceipt) => void;
|
|
1771
|
+
/** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
|
|
1772
|
+
tags?: Record<string, string>;
|
|
1773
|
+
/** Logger callback — analysts SHOULD prefer this over console.* for testability. */
|
|
1774
|
+
log?: (msg: string, fields?: Record<string, unknown>) => void;
|
|
1775
|
+
/** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
|
|
1776
|
+
signal?: AbortSignal;
|
|
1777
|
+
}
|
|
1778
|
+
/**
|
|
1779
|
+
* The minimal contract. Concrete analysts can refine `TInput` so
|
|
1780
|
+
* implementations stay type-safe (e.g. a trace analyst's `TInput` is
|
|
1781
|
+
* `TraceAnalysisStore`); the registry passes the right field from
|
|
1782
|
+
* `AnalystRunInputs` based on `inputKind`.
|
|
1783
|
+
*/
|
|
1784
|
+
interface Analyst<TInput = unknown> {
|
|
1785
|
+
/** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
|
|
1786
|
+
readonly id: string;
|
|
1787
|
+
/** Human-readable. One sentence. */
|
|
1788
|
+
readonly description: string;
|
|
1789
|
+
readonly inputKind: AnalystInputKind;
|
|
1790
|
+
readonly cost: AnalystCost;
|
|
1791
|
+
readonly requires?: AnalystRequirements;
|
|
1792
|
+
/** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
|
|
1793
|
+
readonly version: string;
|
|
1794
|
+
analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
|
|
1795
|
+
}
|
|
1796
|
+
/** Metered work performed by one analyst call. */
|
|
1797
|
+
interface AnalystUsageReceipt {
|
|
1798
|
+
/** Number of model-usage records observed at the provider boundary. */
|
|
1799
|
+
calls: number | null;
|
|
1800
|
+
/** Null when the provider did not return token accounting. */
|
|
1801
|
+
tokens: RunTokenUsage | null;
|
|
1802
|
+
/** Observed, estimated, or explicitly uncaptured dollar cost. */
|
|
1803
|
+
cost: RunCostProvenance;
|
|
1804
|
+
/** Known lower bound when one or more calls have uncaptured cost. */
|
|
1805
|
+
knownCostUsd?: number;
|
|
1806
|
+
}
|
|
1807
|
+
/**
|
|
1808
|
+
* Compute the stable finding_id from the identity-defining fields.
|
|
1809
|
+
* Default implementation hashes {analyst_id, area, subject, normalized claim}.
|
|
1810
|
+
* Analysts that emit findings whose claim text varies per run (timestamps,
|
|
1811
|
+
* counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
|
|
1812
|
+
* or (b) move the variable part into `rationale`/`metadata` and keep the
|
|
1813
|
+
* `claim` static.
|
|
1814
|
+
*/
|
|
1815
|
+
declare function computeFindingId(input: {
|
|
1816
|
+
analyst_id: string;
|
|
1817
|
+
area: string;
|
|
1818
|
+
subject?: string;
|
|
1819
|
+
claim: string;
|
|
1820
|
+
/** Override the claim for hashing — use when the displayed claim has run-specific bits. */
|
|
1821
|
+
id_basis?: string;
|
|
1822
|
+
}): string;
|
|
1823
|
+
/**
|
|
1824
|
+
* Convenience factory: produce a fully-formed AnalystFinding with the
|
|
1825
|
+
* id computed automatically. Analyst code stays terse.
|
|
1826
|
+
*/
|
|
1827
|
+
declare function makeFinding(init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {
|
|
1828
|
+
id_basis?: string;
|
|
1829
|
+
produced_at?: string;
|
|
1830
|
+
}): AnalystFinding;
|
|
1831
|
+
interface AnalystRunSummary {
|
|
1832
|
+
analyst_id: string;
|
|
1833
|
+
status: 'ok' | 'skipped' | 'failed';
|
|
1834
|
+
/** Why skipped — missing input, budget exceeded, capability unmet. */
|
|
1835
|
+
reason?: string;
|
|
1836
|
+
findings_count: number;
|
|
1837
|
+
latency_ms: number;
|
|
1838
|
+
cost_usd: number;
|
|
1839
|
+
/**
|
|
1840
|
+
* Additive receipt for model usage. Registry-produced summaries populate it
|
|
1841
|
+
* even when the analyst emits no findings. `cost_usd` remains the legacy
|
|
1842
|
+
* numeric field; inspect `usage.cost` before treating zero as observed.
|
|
1843
|
+
*/
|
|
1844
|
+
usage?: AnalystUsageReceipt;
|
|
1845
|
+
/** When `status='failed'`: the error class + message, never the full stack. */
|
|
1846
|
+
error?: {
|
|
1847
|
+
class: string;
|
|
1848
|
+
message: string;
|
|
1849
|
+
};
|
|
1850
|
+
}
|
|
1851
|
+
interface AnalystRunResult {
|
|
1852
|
+
run_id: string;
|
|
1853
|
+
correlation_id: string;
|
|
1854
|
+
started_at: string;
|
|
1855
|
+
ended_at: string;
|
|
1856
|
+
findings: AnalystFinding[];
|
|
1857
|
+
per_analyst: AnalystRunSummary[];
|
|
1858
|
+
/** Total LLM cost in USD across all analysts in this registry.run(). */
|
|
1859
|
+
total_cost_usd: number;
|
|
1860
|
+
/**
|
|
1861
|
+
* Provenance for `total_cost_usd`. When uncaptured, the numeric field is only
|
|
1862
|
+
* the known subtotal and must not be treated as the run's total spend.
|
|
1863
|
+
*/
|
|
1864
|
+
total_cost_provenance?: RunCostProvenance;
|
|
1865
|
+
}
|
|
1866
|
+
/**
|
|
1867
|
+
* Events emitted by `AnalystRegistry.runStream(...)` in real time as
|
|
1868
|
+
* the registry executes. UIs subscribe via `for await (const ev of
|
|
1869
|
+
* registry.runStream(...))`; `registry.run(...)` is a thin collector
|
|
1870
|
+
* over the same stream, so the two surfaces share their invariants.
|
|
1871
|
+
*
|
|
1872
|
+
* Per-finding events are intentionally omitted — analyzers are batch
|
|
1873
|
+
* operations (an Ax actor returns the full `findings:json[]` at the
|
|
1874
|
+
* end of the responder), so streaming inside one analyst would only
|
|
1875
|
+
* emit partial JSON consumers can't render. The kind-completion event
|
|
1876
|
+
* is the right granularity; subscribers wanting per-finding rendering
|
|
1877
|
+
* iterate `event.findings` themselves.
|
|
1878
|
+
*/
|
|
1879
|
+
type AnalystRunEvent = {
|
|
1880
|
+
type: 'run-started';
|
|
1881
|
+
run_id: string;
|
|
1882
|
+
correlation_id: string;
|
|
1883
|
+
started_at: string;
|
|
1884
|
+
/** The ordered list of analyst ids the registry will run. */
|
|
1885
|
+
analyst_ids: ReadonlyArray<string>;
|
|
1886
|
+
} | {
|
|
1887
|
+
type: 'analyst-skipped';
|
|
1888
|
+
summary: AnalystRunSummary;
|
|
1889
|
+
} | {
|
|
1890
|
+
type: 'analyst-started';
|
|
1891
|
+
analyst_id: string;
|
|
1892
|
+
started_at: string;
|
|
1893
|
+
} | {
|
|
1894
|
+
type: 'analyst-completed';
|
|
1895
|
+
/** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
|
|
1896
|
+
summary: AnalystRunSummary;
|
|
1897
|
+
findings: ReadonlyArray<AnalystFinding>;
|
|
1898
|
+
} | {
|
|
1899
|
+
type: 'run-completed';
|
|
1900
|
+
result: AnalystRunResult;
|
|
1901
|
+
};
|
|
1902
|
+
|
|
1903
|
+
/**
|
|
1904
|
+
* Adapter factories — lift each existing agent-eval primitive into the
|
|
1905
|
+
* Analyst contract without re-implementing it.
|
|
1906
|
+
*
|
|
1907
|
+
* Five primitives, five factories. Each one:
|
|
1908
|
+
* - Builds an Analyst with a stable id (caller chooses; defaults
|
|
1909
|
+
* given), a sensible default `inputKind`, a version derived from
|
|
1910
|
+
* the wrapped primitive's version + an adapter revision, and an
|
|
1911
|
+
* `analyze()` that calls the primitive and lifts its output to
|
|
1912
|
+
* AnalystFinding[] using `makeFinding()`.
|
|
1913
|
+
* - Maps severities: the existing `Severity` ('critical' | 'major' |
|
|
1914
|
+
* 'minor' | 'info') projects onto AnalystSeverity ('critical' |
|
|
1915
|
+
* 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →
|
|
1916
|
+
* 'medium'. Domain analysts that want finer-grained mapping override.
|
|
1917
|
+
*
|
|
1918
|
+
* Adapters never own state. Calling the same factory twice with the
|
|
1919
|
+
* same primitive instance is safe.
|
|
1920
|
+
*/
|
|
1921
|
+
|
|
1922
|
+
declare function liftSeverity(s: Severity): AnalystSeverity;
|
|
1923
|
+
interface VerifierAdapterOpts<Env> {
|
|
1924
|
+
id?: string;
|
|
1925
|
+
area?: string;
|
|
1926
|
+
verifier: MultiLayerVerifier<Env>;
|
|
1927
|
+
/**
|
|
1928
|
+
* The verifier expects an `env` per run. Adapters take it from
|
|
1929
|
+
* `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.
|
|
1930
|
+
*/
|
|
1931
|
+
options?: Omit<VerifyOptions<Env>, 'env'>;
|
|
1932
|
+
}
|
|
1933
|
+
declare function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env>;
|
|
1934
|
+
interface RunCriticAdapterOpts {
|
|
1935
|
+
id?: string;
|
|
1936
|
+
area?: string;
|
|
1937
|
+
critic?: RunCritic;
|
|
1938
|
+
/** Optional threshold below which a dimension is reported as a finding. Default 0.5. */
|
|
1939
|
+
threshold?: number;
|
|
1940
|
+
}
|
|
1941
|
+
declare function createRunCriticAdapter(opts?: RunCriticAdapterOpts): Analyst<RunTrace>;
|
|
1942
|
+
interface JudgeAdapterOpts {
|
|
1943
|
+
id?: string;
|
|
1944
|
+
area?: string;
|
|
1945
|
+
judge: JudgeFn;
|
|
1946
|
+
/** TCloud handle the JudgeFn calls. */
|
|
1947
|
+
tcloud: TCloud;
|
|
1948
|
+
/** Optional cost classification — most judges call an LLM. */
|
|
1949
|
+
cost?: Analyst['cost'];
|
|
1950
|
+
/** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */
|
|
1951
|
+
threshold?: number;
|
|
1952
|
+
}
|
|
1953
|
+
declare function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput>;
|
|
1954
|
+
interface SemanticConceptJudgeAdapterOpts {
|
|
1955
|
+
id?: string;
|
|
1956
|
+
area?: string;
|
|
1957
|
+
/** Registry context owns cancellation and the per-analyst cost ledger. */
|
|
1958
|
+
options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>;
|
|
1959
|
+
/** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
|
|
1960
|
+
settlementTimeoutMs?: number;
|
|
1961
|
+
}
|
|
1962
|
+
declare function createSemanticConceptJudgeAdapter(opts?: SemanticConceptJudgeAdapterOpts): Analyst<SemanticConceptJudgeInput>;
|
|
1963
|
+
|
|
1964
|
+
interface CreateAnalystAiConfig {
|
|
1965
|
+
/** OpenAI-compatible API key forwarded as `Authorization: Bearer`.
|
|
1966
|
+
* cli-bridge ignores the value on loopback but Ax requires a non-empty string. */
|
|
1967
|
+
apiKey: string;
|
|
1968
|
+
/** OpenAI-compatible base URL — e.g. `https://router.tangle.tools/v1` or a
|
|
1969
|
+
* cli-bridge loopback. */
|
|
1970
|
+
baseUrl?: string;
|
|
1971
|
+
/** Additional headers required by the gateway, such as tenant or execution policy. */
|
|
1972
|
+
headers?: Record<string, string>;
|
|
1973
|
+
/** Model id forwarded to analyst calls. */
|
|
1974
|
+
model: string;
|
|
1975
|
+
/** Ax provider name. Defaults to the OpenAI-compatible client. */
|
|
1976
|
+
provider?: AxAIArgs<unknown>['name'];
|
|
1977
|
+
}
|
|
1978
|
+
/**
|
|
1979
|
+
* Construct the `AxAIService` an analyst kind calls through
|
|
1980
|
+
* (`createTraceAnalystKind({ ai })`).
|
|
1981
|
+
*
|
|
1982
|
+
* Ax's `ai()` pins `config.model` to the OpenAI catalog enum, but every
|
|
1983
|
+
* OpenAI-compatible router an analyst points at (router.tangle.tools,
|
|
1984
|
+
* cli-bridge) accepts arbitrary model ids (claude-code/sonnet, openai/gpt-5.4,
|
|
1985
|
+
* …). Consumers were each re-rolling `ai({ name, apiKey, apiURL, config })`
|
|
1986
|
+
* behind an `as (a: any) => any` cast to dodge the enum; this is the one
|
|
1987
|
+
* canonical constructor so they don't have to — and don't take a direct
|
|
1988
|
+
* `@ax-llm/ax` dependency for it.
|
|
1989
|
+
*/
|
|
1990
|
+
declare function createAnalystAi(config: CreateAnalystAiConfig): AxAIService;
|
|
1991
|
+
|
|
1992
|
+
/**
|
|
1993
|
+
* Deterministic behavioral metrics over OTLP spans — pure arithmetic, no LLM.
|
|
1994
|
+
*
|
|
1995
|
+
* These are the model-independent multiplier: the four trace-quality signals a
|
|
1996
|
+
* tolerant analyzer (e.g. HALO) re-derives per run inside the model — token
|
|
1997
|
+
* growth, output decay, tool monoculture, missing self-verification — computed
|
|
1998
|
+
* here once, in TypeScript, with zero model judgment. A finding that falls out
|
|
1999
|
+
* of arithmetic is trivially model-agnostic and cannot hallucinate the trend.
|
|
2000
|
+
*
|
|
2001
|
+
* General, not trace-specific: the detectors key off token trajectories and
|
|
2002
|
+
* tool usage present in any agentic OTLP trace, not any one benchmark.
|
|
2003
|
+
*/
|
|
2004
|
+
|
|
2005
|
+
type SuboptimalCode = 'monotonic-input-growth' | 'output-length-decay' | 'single-tool-dependency' | 'no-self-verification';
|
|
2006
|
+
interface SuboptimalSignal {
|
|
2007
|
+
code: SuboptimalCode;
|
|
2008
|
+
severity: 'high' | 'medium' | 'low';
|
|
2009
|
+
/** Human-readable claim, with the backing numbers inlined. */
|
|
2010
|
+
detail: string;
|
|
2011
|
+
/** The exact figures the detector fired on — auditable, no model in the loop. */
|
|
2012
|
+
evidence: Record<string, number | string | boolean>;
|
|
2013
|
+
}
|
|
2014
|
+
interface BehavioralMetrics {
|
|
2015
|
+
/** The only trace represented by these metrics; null when spans are empty. */
|
|
2016
|
+
traceId: string | null;
|
|
2017
|
+
llmCallCount: number;
|
|
2018
|
+
/** Causally serial LLM timelines. Parallel branches are never joined. */
|
|
2019
|
+
tokenSequences: BehavioralTokenSequence[];
|
|
2020
|
+
/** Token values from the longest serial timeline, retained for convenience. */
|
|
2021
|
+
inputTokenTrajectory: number[];
|
|
2022
|
+
outputTokenTrajectory: number[];
|
|
2023
|
+
toolHistogram: Record<string, number>;
|
|
2024
|
+
totalToolCalls: number;
|
|
2025
|
+
distinctTools: number;
|
|
2026
|
+
/** distinct/total tool calls; 1.0 when there are no tool calls. */
|
|
2027
|
+
toolDiversityRatio: number;
|
|
2028
|
+
hasSelfVerification: boolean;
|
|
2029
|
+
signals: SuboptimalSignal[];
|
|
2030
|
+
}
|
|
2031
|
+
interface BehavioralTokenSequence {
|
|
2032
|
+
scopeId: string;
|
|
2033
|
+
spanIds: string[];
|
|
2034
|
+
inputTokenTrajectory: Array<number | null>;
|
|
2035
|
+
outputTokenTrajectory: Array<number | null>;
|
|
2036
|
+
}
|
|
2037
|
+
|
|
2038
|
+
/**
|
|
2039
|
+
* `behavioralAnalyst` — a DETERMINISTIC analyst (cost.kind = 'deterministic',
|
|
2040
|
+
* never calls the LLM). It produces the efficiency/behavioral findings a
|
|
2041
|
+
* tolerant agentic analyzer (HALO) re-derives per run inside the model —
|
|
2042
|
+
* context bloat, output decay, tool monoculture, missing self-verification —
|
|
2043
|
+
* directly from arithmetic over spans (`computeTraceMetrics`).
|
|
2044
|
+
*
|
|
2045
|
+
* Why it matters: these findings are model-agnostic BY CONSTRUCTION (no model
|
|
2046
|
+
* in the loop), so they cannot return 0 on a weak model the way the Ax-RLM
|
|
2047
|
+
* does — and they are strictly more reliable than HALO, which spends tokens
|
|
2048
|
+
* re-deriving the same numbers and can hallucinate the trend. The agentic
|
|
2049
|
+
* RLM kinds remain for SEMANTIC findings that genuinely need a model; this
|
|
2050
|
+
* analyst owns the behavioral class.
|
|
2051
|
+
*/
|
|
2052
|
+
|
|
2053
|
+
/**
|
|
2054
|
+
* Map computed signals → structured AnalystFindings. Pure: no LLM, no clock
|
|
2055
|
+
* dependence beyond `produced_at` (overridable for deterministic tests).
|
|
2056
|
+
*/
|
|
2057
|
+
declare function deriveEfficiencyFindings(metrics: BehavioralMetrics, opts?: {
|
|
2058
|
+
analystId?: string;
|
|
2059
|
+
producedAt?: string;
|
|
2060
|
+
}): AnalystFinding[];
|
|
2061
|
+
/** The deterministic behavioral/efficiency analyst (no LLM, any-model). */
|
|
2062
|
+
declare function behavioralAnalyst(): Analyst<TraceAnalysisStore>;
|
|
2063
|
+
|
|
2064
|
+
/**
|
|
2065
|
+
* Typed Ax output for analyst findings.
|
|
2066
|
+
*
|
|
2067
|
+
* Replaces the legacy `findings:string[]` pattern (where every bullet
|
|
2068
|
+
* became a flat-severity `AnalystFinding`) with a structured object
|
|
2069
|
+
* array. Ax binds the field as `findings:json[]` so the provider emits
|
|
2070
|
+
* native structured output; at the kind-factory boundary we Zod-validate
|
|
2071
|
+
* each emitted finding so malformed rows fail loud instead of being
|
|
2072
|
+
* silently lifted with default severity.
|
|
2073
|
+
*
|
|
2074
|
+
* Why not `f.object().array()` directly in the signature? The Ax
|
|
2075
|
+
* signature string `question:string -> findings:json[]` already lets
|
|
2076
|
+
* the provider emit JSON arrays. A Zod boundary is required either
|
|
2077
|
+
* way (the provider can return any JSON), and Zod gives us a single
|
|
2078
|
+
* validation surface independent of which Ax version is installed.
|
|
2079
|
+
*/
|
|
2080
|
+
|
|
2081
|
+
declare const ANALYST_SEVERITIES: readonly ["critical", "high", "medium", "low", "info"];
|
|
2082
|
+
declare const RawAnalystEvidenceSchema: z.ZodObject<{
|
|
2083
|
+
uri: z.ZodString;
|
|
2084
|
+
excerpt: z.ZodOptional<z.ZodString>;
|
|
2085
|
+
}, z.core.$strict>;
|
|
2086
|
+
type RawAnalystEvidence = z.infer<typeof RawAnalystEvidenceSchema>;
|
|
2087
|
+
/** Original public schema retained for stored rows and callback contracts. */
|
|
2088
|
+
declare const RawAnalystFindingSchema: z.ZodObject<{
|
|
2089
|
+
evidence_uri: z.ZodString;
|
|
2090
|
+
evidence_excerpt: z.ZodOptional<z.ZodString>;
|
|
2091
|
+
severity: z.ZodEnum<{
|
|
2092
|
+
critical: "critical";
|
|
2093
|
+
info: "info";
|
|
2094
|
+
low: "low";
|
|
2095
|
+
high: "high";
|
|
2096
|
+
medium: "medium";
|
|
2097
|
+
}>;
|
|
2098
|
+
claim: z.ZodString;
|
|
2099
|
+
subject: z.ZodOptional<z.ZodString>;
|
|
2100
|
+
confidence: z.ZodNumber;
|
|
2101
|
+
rationale: z.ZodOptional<z.ZodString>;
|
|
2102
|
+
recommended_action: z.ZodOptional<z.ZodString>;
|
|
2103
|
+
}, z.core.$strict>;
|
|
2104
|
+
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
2105
|
+
/**
|
|
2106
|
+
* Canonical plural-evidence contract. The preprocessor accepts the original
|
|
2107
|
+
* `evidence_uri` / `evidence_excerpt` pair and normalizes it into one evidence
|
|
2108
|
+
* item so persisted rows and older model fixtures remain readable. New output
|
|
2109
|
+
* always receives the plural shape.
|
|
2110
|
+
*/
|
|
2111
|
+
declare const CanonicalRawAnalystFindingSchema: z.ZodPipe<z.ZodTransform<unknown, unknown>, z.ZodObject<{
|
|
2112
|
+
evidence: z.ZodArray<z.ZodObject<{
|
|
2113
|
+
uri: z.ZodString;
|
|
2114
|
+
excerpt: z.ZodOptional<z.ZodString>;
|
|
2115
|
+
}, z.core.$strict>>;
|
|
2116
|
+
severity: z.ZodEnum<{
|
|
2117
|
+
critical: "critical";
|
|
2118
|
+
info: "info";
|
|
2119
|
+
low: "low";
|
|
2120
|
+
high: "high";
|
|
2121
|
+
medium: "medium";
|
|
2122
|
+
}>;
|
|
2123
|
+
claim: z.ZodString;
|
|
2124
|
+
subject: z.ZodOptional<z.ZodString>;
|
|
2125
|
+
confidence: z.ZodNumber;
|
|
2126
|
+
rationale: z.ZodOptional<z.ZodString>;
|
|
2127
|
+
recommended_action: z.ZodOptional<z.ZodString>;
|
|
2128
|
+
}, z.core.$strict>>;
|
|
2129
|
+
type CanonicalRawAnalystFinding = z.infer<typeof CanonicalRawAnalystFindingSchema>;
|
|
2130
|
+
/**
|
|
2131
|
+
* Description embedded into the actor prompt so the LLM knows what
|
|
2132
|
+
* shape to emit. Kept here so kinds share one source of truth rather
|
|
2133
|
+
* than restating the schema in every prompt.
|
|
2134
|
+
*/
|
|
2135
|
+
declare const RAW_FINDING_SCHEMA_PROMPT = "Each finding MUST be a strict JSON object with:\n - severity: \"critical\" | \"high\" | \"medium\" | \"low\" | \"info\"\n - claim: one-sentence statement (max 2000 chars)\n - subject?: one exact subject form listed by this kind; omit rather than guess\n - evidence: REQUIRED non-empty array of {\"uri\": string, \"excerpt\"?: string}. Use real identifiers with span://, event://, artifact://, metric://, or finding://. Include a short exact quote in excerpt when available. If nothing is citable, do not emit the finding.\n - confidence: number 0..1 (0.9+ exact evidence; 0.6-0.8 inferred pattern; <0.5 speculative)\n - rationale?: one or two reasoning sentences\n - recommended_action?: concrete imperative change; omit for descriptive findings\n\nUnknown fields are rejected. Do not emit area; the factory assigns it. Emit [] when there are no findings. Never fabricate evidence.";
|
|
2136
|
+
/** Convert canonical raw citations into the public finding evidence envelope. */
|
|
2137
|
+
declare function evidenceRefsFromRawFinding(finding: CanonicalRawAnalystFinding): EvidenceRef[];
|
|
2138
|
+
/**
|
|
2139
|
+
* Validate the original singular-evidence shape. This public parser retains
|
|
2140
|
+
* its pre-canonicalization result type so existing callback code and stored
|
|
2141
|
+
* rows continue to receive exactly the object accepted by
|
|
2142
|
+
* {@link RawAnalystFindingSchema}.
|
|
2143
|
+
*/
|
|
2144
|
+
declare function parseRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): RawAnalystFinding | null;
|
|
2145
|
+
/** Validate model output and normalize original singular citations. */
|
|
2146
|
+
declare function parseCanonicalRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): CanonicalRawAnalystFinding | null;
|
|
2147
|
+
|
|
2148
|
+
/**
|
|
2149
|
+
* Analyst-kind factory — the typed way to define trace analysts.
|
|
2150
|
+
*
|
|
2151
|
+
* A "kind" is a specialized analyst whose actor prompt, tool subset,
|
|
2152
|
+
* and bounded Ax subqueries target one failure-mode lens (failure-mode
|
|
2153
|
+
* classification, knowledge gap discovery, knowledge poisoning,
|
|
2154
|
+
* self-improvement, ...). Kinds emit findings in the typed
|
|
2155
|
+
* `CanonicalRawAnalystFinding` shape via a JSON-array Ax output; the factory
|
|
2156
|
+
* validates each row with Zod and lifts it into `AnalystFinding[]`.
|
|
2157
|
+
*
|
|
2158
|
+
* Composition rules:
|
|
2159
|
+
* - Each kind owns its actor description. No generic "answer this
|
|
2160
|
+
* question" prompt — the prompt names the failure lens.
|
|
2161
|
+
* - Each kind picks a narrow tool subset from `ANALYST_TOOL_GROUPS`.
|
|
2162
|
+
* A kind that never needs full-trace dumps can drop `viewTrace` /
|
|
2163
|
+
* `viewSpans` and stay cheap.
|
|
2164
|
+
* - Each kind declares its subquery + parallelism budget. Discovery-heavy
|
|
2165
|
+
* kinds can fan out more bounded semantic questions than narrow lenses.
|
|
2166
|
+
*
|
|
2167
|
+
* Optimizer hook: kinds may declare `goldens` — labeled examples used
|
|
2168
|
+
* by `AxBootstrapFewShot` / `AxGEPA` to fit the actor
|
|
2169
|
+
* description programmatically. Stored on the kind, not the registry,
|
|
2170
|
+
* because the right metric is kind-specific.
|
|
2171
|
+
*/
|
|
2172
|
+
|
|
2173
|
+
/**
|
|
2174
|
+
* Per-kind specification. The factory turns this into a regular
|
|
2175
|
+
* `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
|
|
2176
|
+
*/
|
|
2177
|
+
interface TraceAnalystKindSpec {
|
|
2178
|
+
/** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
|
|
2179
|
+
id: string;
|
|
2180
|
+
/** One-sentence description shown in `registry.list()`. */
|
|
2181
|
+
description: string;
|
|
2182
|
+
/** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
|
|
2183
|
+
area: string;
|
|
2184
|
+
/** Bump on any breaking change to the actor prompt or output schema. */
|
|
2185
|
+
version: string;
|
|
2186
|
+
/** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
|
|
2187
|
+
actorDescription: string;
|
|
2188
|
+
/** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
|
|
2189
|
+
buildTools: (store: TraceAnalysisStore) => AxFunction[];
|
|
2190
|
+
/** Bounded semantic subqueries. `maxCalls: 0` disables model fan-out. */
|
|
2191
|
+
subqueries?: {
|
|
2192
|
+
maxCalls: number;
|
|
2193
|
+
maxParallel?: number;
|
|
2194
|
+
};
|
|
2195
|
+
/** Actor turn cap. Default 12. */
|
|
2196
|
+
maxTurns?: number;
|
|
2197
|
+
/** Runtime char cap. Default 6000. */
|
|
2198
|
+
maxRuntimeChars?: number;
|
|
2199
|
+
/** Maximum output tokens for every actor and subquery model call. Default 4096. */
|
|
2200
|
+
maxOutputTokens?: number;
|
|
2201
|
+
/** Cost classification surfaced in `registry.list()` and budget enforcement. */
|
|
2202
|
+
cost: AnalystCost;
|
|
2203
|
+
/** Per-finding-row hook — kinds may reject / rewrite before lifting. */
|
|
2204
|
+
postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
|
|
2205
|
+
/** Minimum citations per finding. Default 1; rows below it are rejected. */
|
|
2206
|
+
minimumEvidenceCitations?: number;
|
|
2207
|
+
/** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
|
|
2208
|
+
goldens?: TraceAnalystGolden[];
|
|
2209
|
+
}
|
|
2210
|
+
/**
|
|
2211
|
+
* One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
|
|
2212
|
+
* Each input is the same `{question}` an analyst would receive; `expected`
|
|
2213
|
+
* is the ground-truth finding set a fitted prompt should produce on this
|
|
2214
|
+
* input. Metric: kind-specific (default: F1 on `finding_id` overlap).
|
|
2215
|
+
*/
|
|
2216
|
+
interface TraceAnalystGolden {
|
|
2217
|
+
question: string;
|
|
2218
|
+
expected: ReadonlyArray<Omit<CanonicalRawAnalystFinding, 'confidence'>>;
|
|
2219
|
+
}
|
|
2220
|
+
interface CreateTraceAnalystKindOpts {
|
|
2221
|
+
/** AxAIService bound at registration time. */
|
|
2222
|
+
ai: AxAIService;
|
|
2223
|
+
/** Required unless `ai` was created by {@link createAnalystAi}. */
|
|
2224
|
+
model?: string;
|
|
2225
|
+
/** Override the spec's `version` (e.g. when an optimizer has fitted a new prompt). */
|
|
2226
|
+
versionSuffix?: string;
|
|
2227
|
+
/**
|
|
2228
|
+
* Optional two-phase recovery: when the agentic harvest is empty but the
|
|
2229
|
+
* actor produced a substantive free-form `report`, extract findings from that
|
|
2230
|
+
* prose via a tolerant chat-completions pass (`structureFindings`) — no
|
|
2231
|
+
* strict-emission contract, so it works on weak models. Omit to leave the
|
|
2232
|
+
* actor's harvest as-is (the report is still surfaced fail-loud either way).
|
|
2233
|
+
*/
|
|
2234
|
+
recovery?: {
|
|
2235
|
+
baseUrl: string;
|
|
2236
|
+
apiKey?: string;
|
|
2237
|
+
model?: string;
|
|
2238
|
+
fetchImpl?: typeof fetch;
|
|
2239
|
+
};
|
|
2240
|
+
/** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
|
|
2241
|
+
settlementTimeoutMs?: number;
|
|
2242
|
+
}
|
|
2243
|
+
/**
|
|
2244
|
+
* Build an `Analyst<TraceAnalysisStore>` from a kind spec.
|
|
2245
|
+
*
|
|
2246
|
+
* Lifts the Ax pipeline once at registration time so the registry
|
|
2247
|
+
* gets a stateless analyst. The Ax agent is freshly constructed per
|
|
2248
|
+
* `analyze()` call (the agent carries chat-log + usage state we don't
|
|
2249
|
+
* want shared across analyst runs).
|
|
2250
|
+
*/
|
|
2251
|
+
declare function createTraceAnalystKind(spec: TraceAnalystKindSpec, opts: CreateTraceAnalystKindOpts): Analyst<TraceAnalysisStore>;
|
|
2252
|
+
/**
|
|
2253
|
+
* Render a compact prior-findings block the actor reads alongside its
|
|
2254
|
+
* brief. Each row is one line so the actor can scan dozens cheaply.
|
|
2255
|
+
* The kind's prompt instructs the actor to (a) check whether a new
|
|
2256
|
+
* cluster matches a prior `finding_id` (carry the id forward via
|
|
2257
|
+
* `id_basis` to keep diffs stable) and (b) raise severity / confidence
|
|
2258
|
+
* when a prior finding has reappeared without remediation.
|
|
2259
|
+
*
|
|
2260
|
+
* Returns the empty string when there are no prior findings — most
|
|
2261
|
+
* runs are "first-of-its-kind" and the prompt stays unchanged.
|
|
2262
|
+
*
|
|
2263
|
+
* Exported for tests + for consumers that build their own actor
|
|
2264
|
+
* prompts (e.g. specialized analysts living outside the default kinds).
|
|
2265
|
+
*/
|
|
2266
|
+
declare function renderPriorFindings(prior: AnalystContext['priorFindings']): string;
|
|
2267
|
+
/** Render findings produced earlier in this same registry run. */
|
|
2268
|
+
declare function renderUpstreamFindings(upstream: AnalystContext['upstreamFindings']): string;
|
|
2269
|
+
|
|
2270
|
+
/**
|
|
2271
|
+
* AnalystRegistry — orchestrate N analysts against one run.
|
|
2272
|
+
*
|
|
2273
|
+
* Owns three responsibilities and only three:
|
|
2274
|
+
* 1. Registration — ids must be unique; bad registrations fail loudly
|
|
2275
|
+
* at register-time, not run-time.
|
|
2276
|
+
* 2. Routing — each analyst declares its `inputKind`; the registry
|
|
2277
|
+
* picks the matching field from AnalystRunInputs and skips the
|
|
2278
|
+
* analyst with a logged reason if it's missing.
|
|
2279
|
+
* 3. Isolation — one analyst's exception MUST NOT stop other analysts.
|
|
2280
|
+
* Failed analysts produce zero findings + a 'failed' summary row.
|
|
2281
|
+
*
|
|
2282
|
+
* Cross-cutting concerns (telemetry, error → finding conversion, cost
|
|
2283
|
+
* ingestion, storage rotation) live in `AnalystHooks`. Budget shaping
|
|
2284
|
+
* (equal split vs weighted vs custom) lives in `BudgetPolicy`. Both
|
|
2285
|
+
* have sensible defaults; consumers override only what they need.
|
|
2286
|
+
*/
|
|
2287
|
+
|
|
2288
|
+
interface AnalystHooks {
|
|
2289
|
+
/** Before analyze() — last chance to mutate ctx (e.g. inject tags, override budget). */
|
|
2290
|
+
onBeforeAnalyze?(args: {
|
|
2291
|
+
analyst: Analyst;
|
|
2292
|
+
ctx: AnalystContext;
|
|
2293
|
+
runId: string;
|
|
2294
|
+
}): void | Promise<void>;
|
|
2295
|
+
/** After every analyst (ok | failed | skipped). Use for telemetry, ingestion, rotation. */
|
|
2296
|
+
onAfterAnalyze?(args: {
|
|
2297
|
+
analyst: Analyst;
|
|
2298
|
+
summary: AnalystRunSummary;
|
|
2299
|
+
findings: AnalystFinding[];
|
|
2300
|
+
runId: string;
|
|
2301
|
+
}): void | Promise<void>;
|
|
2302
|
+
/**
|
|
2303
|
+
* On analyst exception. Hook MAY return findings to convert the
|
|
2304
|
+
* error into structured findings; the summary still reports 'failed'.
|
|
2305
|
+
* Return void to keep the default empty-findings behavior.
|
|
2306
|
+
*/
|
|
2307
|
+
onError?(args: {
|
|
2308
|
+
analyst: Analyst;
|
|
2309
|
+
error: Error;
|
|
2310
|
+
runId: string;
|
|
2311
|
+
}): AnalystFinding[] | undefined | Promise<AnalystFinding[] | undefined>;
|
|
2312
|
+
/** Once after registry.run() completes. Use for final aggregation, persistence. */
|
|
2313
|
+
onComplete?(args: {
|
|
2314
|
+
result: AnalystRunResult;
|
|
2315
|
+
}): void | Promise<void>;
|
|
2316
|
+
}
|
|
2317
|
+
interface BudgetPolicy {
|
|
2318
|
+
/** Overall USD cap across the registry.run(). */
|
|
2319
|
+
totalUsd?: number;
|
|
2320
|
+
/** Per-analyst weight for the default allocator. Missing ids get weight 1. */
|
|
2321
|
+
weights?: Record<string, number>;
|
|
2322
|
+
/**
|
|
2323
|
+
* Custom allocator — receives the analyst, remaining/total budget, and
|
|
2324
|
+
* the count of analysts that will run. Returns the per-analyst budget
|
|
2325
|
+
* (or undefined only when the run has no overall cap). Overrides weights
|
|
2326
|
+
* when set.
|
|
2327
|
+
*/
|
|
2328
|
+
allocate?: (args: {
|
|
2329
|
+
analyst: Analyst;
|
|
2330
|
+
totalUsd: number | undefined;
|
|
2331
|
+
remainingUsd: number | undefined;
|
|
2332
|
+
runningCount: number;
|
|
2333
|
+
}) => number | undefined;
|
|
2334
|
+
}
|
|
2335
|
+
interface AnalystRegistryOptions {
|
|
2336
|
+
/** Shared chat client passed to every LLM analyst via AnalystContext. */
|
|
2337
|
+
chat?: ChatClient;
|
|
2338
|
+
/** Logger callback. Defaults to a no-op. */
|
|
2339
|
+
log?: (msg: string, fields?: Record<string, unknown>) => void;
|
|
2340
|
+
/** Hooks invoked around analyze() — observability + customization seam. */
|
|
2341
|
+
hooks?: AnalystHooks;
|
|
2342
|
+
/** Default budget when run() doesn't override. */
|
|
2343
|
+
defaultBudget?: BudgetPolicy;
|
|
2344
|
+
}
|
|
2345
|
+
interface RegistryRunOpts {
|
|
2346
|
+
/** Restrict to a subset of registered analysts by id. */
|
|
2347
|
+
only?: string[];
|
|
2348
|
+
/** Skip these analysts even if registered. Useful for cheap iteration. */
|
|
2349
|
+
skip?: string[];
|
|
2350
|
+
/** Budget policy — totalUsd + optional weights/allocator. Falls back to options.defaultBudget. */
|
|
2351
|
+
budget?: BudgetPolicy;
|
|
2352
|
+
/** Active-work cap for the complete registry run. Model receipt settlement may follow. */
|
|
2353
|
+
timeoutMs?: number;
|
|
2354
|
+
/** Abort signal — forwarded into every analyst's context. */
|
|
2355
|
+
signal?: AbortSignal;
|
|
2356
|
+
/** Shared paid-call account forwarded to every analyst. */
|
|
2357
|
+
costLedger?: CostLedgerHandle;
|
|
2358
|
+
/** Attribution phase for calls written to `costLedger`. */
|
|
2359
|
+
costPhase?: string;
|
|
2360
|
+
/** Tags echoed into AnalystContext.tags — useful for tracking environment/version in findings. */
|
|
2361
|
+
tags?: Record<string, string>;
|
|
2362
|
+
/**
|
|
2363
|
+
* Prior-run findings made available as retrieval context to every
|
|
2364
|
+
* analyst via `ctx.priorFindings`. The registry forwards the slice
|
|
2365
|
+
* whose `analyst_id` matches each registered analyst so a kind sees
|
|
2366
|
+
* only its own history. Pass `{ '*': findings }` to broadcast to
|
|
2367
|
+
* every analyst (useful when several kinds share the same historical
|
|
2368
|
+
* context). For findings from this run, use `chainFindings` instead.
|
|
2369
|
+
*/
|
|
2370
|
+
priorFindings?: ReadonlyArray<AnalystFinding> | Record<string, ReadonlyArray<AnalystFinding>>;
|
|
2371
|
+
/**
|
|
2372
|
+
* Pass findings produced earlier in this registry run to each later analyst
|
|
2373
|
+
* via `ctx.upstreamFindings`. Registration order is dependency order.
|
|
2374
|
+
* Disabled by default because independent analyst suites must opt in.
|
|
2375
|
+
*/
|
|
2376
|
+
chainFindings?: boolean;
|
|
2377
|
+
}
|
|
2378
|
+
declare class AnalystRegistry {
|
|
2379
|
+
private readonly analysts;
|
|
2380
|
+
private readonly options;
|
|
2381
|
+
constructor(options?: AnalystRegistryOptions);
|
|
2382
|
+
register(analyst: Analyst): void;
|
|
2383
|
+
list(): ReadonlyArray<{
|
|
2384
|
+
id: string;
|
|
2385
|
+
description: string;
|
|
2386
|
+
version: string;
|
|
2387
|
+
cost: Analyst['cost'];
|
|
2388
|
+
}>;
|
|
2389
|
+
run(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): Promise<AnalystRunResult>;
|
|
2390
|
+
/**
|
|
2391
|
+
* Streaming counterpart to `run()`. Emits `AnalystRunEvent` values
|
|
2392
|
+
* in real time — `run-started`, then per-analyst `skipped` /
|
|
2393
|
+
* `started` / `completed`, then a terminal `run-completed` whose
|
|
2394
|
+
* payload is the full `AnalystRunResult`. UIs use this to render
|
|
2395
|
+
* progress; persistence consumers use `run()` and read the result.
|
|
2396
|
+
*
|
|
2397
|
+
* Hooks (`onBeforeAnalyze` / `onAfterAnalyze` / `onError` /
|
|
2398
|
+
* `onComplete`) fire as before — streaming is additive, not a hook
|
|
2399
|
+
* replacement.
|
|
2400
|
+
*/
|
|
2401
|
+
runStream(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): AsyncGenerator<AnalystRunEvent, void, void>;
|
|
2402
|
+
private selectAnalysts;
|
|
2403
|
+
private routeInput;
|
|
2404
|
+
}
|
|
2405
|
+
|
|
2406
|
+
/**
|
|
2407
|
+
* `buildDefaultAnalystRegistry` — the canonical analyst suite, so consumers
|
|
2408
|
+
* stop hand-wiring `new AnalystRegistry()` + per-kind `createTraceAnalystKind`.
|
|
2409
|
+
*
|
|
2410
|
+
* The deterministic `behavioralAnalyst` is ALWAYS registered (it needs no
|
|
2411
|
+
* model and is model-agnostic by construction). The agentic RLM kinds are
|
|
2412
|
+
* registered only when an `ai` service is supplied — so a caller with no LLM
|
|
2413
|
+
* still gets the full behavioral/efficiency diagnosis, and the substrate's
|
|
2414
|
+
* "any model (including no model)" guarantee holds at the suite level.
|
|
2415
|
+
*/
|
|
2416
|
+
|
|
2417
|
+
interface DefaultAnalystRegistryOptions {
|
|
2418
|
+
/** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
|
|
2419
|
+
ai?: AxAIService;
|
|
2420
|
+
/** Required unless `ai` was created by `createAnalystAi`. */
|
|
2421
|
+
model?: string;
|
|
2422
|
+
/** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
|
|
2423
|
+
kinds?: readonly TraceAnalystKindSpec[];
|
|
2424
|
+
/** Set false to omit the deterministic behavioral analyst (default: include). */
|
|
2425
|
+
includeBehavioral?: boolean;
|
|
2426
|
+
/** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
|
|
2427
|
+
registry?: AnalystRegistryOptions;
|
|
2428
|
+
}
|
|
2429
|
+
declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
|
|
2430
|
+
|
|
2431
|
+
/**
|
|
2432
|
+
* Typed `FindingSubject` — the canonical grammar every analyst kind emits.
|
|
2433
|
+
*
|
|
2434
|
+
* Background: kind actor prompts have always documented a subject grammar
|
|
2435
|
+
* (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`) but the
|
|
2436
|
+
* LLM was unconstrained — it could emit `subject: "fix the prompt"`
|
|
2437
|
+
* (prose) and downstream adapters routed on `startsWith(...)` would
|
|
2438
|
+
* silently skip it. Every per-vertical `ImprovementAdapter` had a
|
|
2439
|
+
* routing table that mostly caught nothing.
|
|
2440
|
+
*
|
|
2441
|
+
* This module fixes that:
|
|
2442
|
+
* - `parseFindingSubject(raw)` — returns the typed `FindingSubject`
|
|
2443
|
+
* when `raw` matches the grammar, else `null`. Used at the
|
|
2444
|
+
* `RawAnalystFindingSchema` boundary so malformed subjects are
|
|
2445
|
+
* rejected loudly instead of silently lifted into the registry.
|
|
2446
|
+
* - `FindingSubjectKind` — the union of valid locus categories. Each
|
|
2447
|
+
* variant carries the typed components downstream adapters resolve
|
|
2448
|
+
* against the agent's surface manifest (no string parsing in the
|
|
2449
|
+
* adapter).
|
|
2450
|
+
* - `FINDING_SUBJECT_GRAMMAR_PROMPT` — single source of truth for the
|
|
2451
|
+
* grammar string embedded in kind actor prompts. Drift between
|
|
2452
|
+
* prompt and parser is impossible if every kind imports this.
|
|
2453
|
+
*
|
|
2454
|
+
* The grammar is intentionally NARROW — only loci the substrate's
|
|
2455
|
+
* default `ImprovementAdapter` / `KnowledgeAdapter` can act on. A
|
|
2456
|
+
* finding with a subject outside this set fails the parser; the kind
|
|
2457
|
+
* author either extends the grammar here (and adds adapter routing)
|
|
2458
|
+
* or rephrases the prompt to map onto an existing variant.
|
|
2459
|
+
*
|
|
2460
|
+
* `failure-mode` is the one exception — its subjects are free-form
|
|
2461
|
+
* cluster labels, not loci. The schema preserves them as
|
|
2462
|
+
* `{ kind: 'cluster', label }` and the adapters skip them (cluster
|
|
2463
|
+
* findings are evidence, not actionable mutations).
|
|
2464
|
+
*/
|
|
2465
|
+
|
|
2466
|
+
/**
|
|
2467
|
+
* Discriminated union of every locus the substrate can route findings to.
|
|
2468
|
+
*
|
|
2469
|
+
* Adapters narrow on `kind` and use the typed components (no string
|
|
2470
|
+
* parsing). Adding a variant here REQUIRES updating the parser, the
|
|
2471
|
+
* grammar prompt, and at least one adapter — by design.
|
|
2472
|
+
*/
|
|
2473
|
+
type FindingSubject = {
|
|
2474
|
+
kind: 'knowledge.wiki';
|
|
2475
|
+
slug: string;
|
|
2476
|
+
heading?: string;
|
|
2477
|
+
} | {
|
|
2478
|
+
kind: 'knowledge.claim';
|
|
2479
|
+
topic: string;
|
|
2480
|
+
} | {
|
|
2481
|
+
kind: 'knowledge.raw';
|
|
2482
|
+
sourceId: string;
|
|
2483
|
+
} | {
|
|
2484
|
+
kind: 'knowledge.stale';
|
|
2485
|
+
slug: string;
|
|
2486
|
+
} | {
|
|
2487
|
+
kind: 'system-prompt';
|
|
2488
|
+
section: string;
|
|
2489
|
+
} | {
|
|
2490
|
+
kind: 'skill';
|
|
2491
|
+
name: string;
|
|
2492
|
+
} | {
|
|
2493
|
+
kind: 'tool-doc';
|
|
2494
|
+
tool: string;
|
|
2495
|
+
aspect?: string;
|
|
2496
|
+
} | {
|
|
2497
|
+
kind: 'new-tool';
|
|
2498
|
+
name: string;
|
|
2499
|
+
} | {
|
|
2500
|
+
kind: 'mcp';
|
|
2501
|
+
server: string;
|
|
2502
|
+
tool?: string;
|
|
2503
|
+
} | {
|
|
2504
|
+
kind: 'hook';
|
|
2505
|
+
name: string;
|
|
2506
|
+
} | {
|
|
2507
|
+
kind: 'subagent';
|
|
2508
|
+
name: string;
|
|
2509
|
+
} | {
|
|
2510
|
+
kind: 'workflow';
|
|
2511
|
+
name: string;
|
|
2512
|
+
} | {
|
|
2513
|
+
kind: 'rollout-policy';
|
|
2514
|
+
field: string;
|
|
2515
|
+
} | {
|
|
2516
|
+
kind: 'agent-profile';
|
|
2517
|
+
field: string;
|
|
2518
|
+
} | {
|
|
2519
|
+
kind: 'code';
|
|
2520
|
+
path: string;
|
|
2521
|
+
} | {
|
|
2522
|
+
kind: 'rag';
|
|
2523
|
+
corpus: string;
|
|
2524
|
+
docId: string;
|
|
2525
|
+
} | {
|
|
2526
|
+
kind: 'memory';
|
|
2527
|
+
key: string;
|
|
2528
|
+
} | {
|
|
2529
|
+
kind: 'scaffolding';
|
|
2530
|
+
concern: string;
|
|
2531
|
+
} | {
|
|
2532
|
+
kind: 'output-schema';
|
|
2533
|
+
field: string;
|
|
2534
|
+
} | {
|
|
2535
|
+
kind: 'websearch.outdated';
|
|
2536
|
+
topic: string;
|
|
2537
|
+
} | {
|
|
2538
|
+
kind: 'prior-run-summary';
|
|
2539
|
+
topic: string;
|
|
2540
|
+
} | {
|
|
2541
|
+
kind: 'cluster';
|
|
2542
|
+
label: string;
|
|
2543
|
+
};
|
|
2544
|
+
type FindingSubjectKind = FindingSubject['kind'];
|
|
2545
|
+
declare const FINDING_SUBJECT_KINDS: ReadonlyArray<FindingSubjectKind>;
|
|
2546
|
+
/**
|
|
2547
|
+
* Parse a raw subject string emitted by an analyst kind's actor.
|
|
2548
|
+
*
|
|
2549
|
+
* Returns the typed `FindingSubject` when `raw` matches the grammar,
|
|
2550
|
+
* else `null`. Callers use the `null` return as a signal to either
|
|
2551
|
+
* (a) reject the finding at parse time (kinds that emit typed loci —
|
|
2552
|
+
* knowledge-gap, improvement, knowledge-poisoning) or (b) lift it as
|
|
2553
|
+
* a cluster label (failure-mode).
|
|
2554
|
+
*
|
|
2555
|
+
* Slugs are constrained to `[a-z0-9-]+` (lowercase kebab) to keep file
|
|
2556
|
+
* paths sane downstream. Topics / keys / sections allow any non-empty
|
|
2557
|
+
* string (free-form for the LLM's voice) but get trimmed.
|
|
2558
|
+
*
|
|
2559
|
+
* Empty / whitespace-only inputs return `null`. `undefined` returns
|
|
2560
|
+
* `null`. Both are surfaced by the caller as a rejected subject.
|
|
2561
|
+
*/
|
|
2562
|
+
declare function parseFindingSubject(raw: string | null | undefined): FindingSubject | null;
|
|
2563
|
+
/**
|
|
2564
|
+
* Render the parsed subject back to its canonical string form. Inverse
|
|
2565
|
+
* of `parseFindingSubject`; useful when the substrate constructs new
|
|
2566
|
+
* findings programmatically (e.g. for tests, replays, or
|
|
2567
|
+
* `id_basis` carry-forward).
|
|
2568
|
+
*/
|
|
2569
|
+
declare function renderFindingSubject(s: FindingSubject): string;
|
|
2570
|
+
/**
|
|
2571
|
+
* The grammar text embedded into kind actor prompts. Kinds opt into
|
|
2572
|
+
* the subset of variants they emit (e.g. `improvement` excludes the
|
|
2573
|
+
* cluster variant; `failure-mode` includes ONLY the cluster variant).
|
|
2574
|
+
*
|
|
2575
|
+
* Drift between prompt and parser is impossible: every kind imports
|
|
2576
|
+
* this constant + the matching `expects` set, and the unit tests below
|
|
2577
|
+
* lock the table to the parser.
|
|
2578
|
+
*/
|
|
2579
|
+
declare const FINDING_SUBJECT_SYNTAX: Readonly<Record<FindingSubjectKind, string>>;
|
|
2580
|
+
declare const FINDING_SUBJECT_GRAMMAR_PROMPT: string;
|
|
2581
|
+
/**
|
|
2582
|
+
* The variants each kind is allowed to emit. Used at the kind factory
|
|
2583
|
+
* boundary so a knowledge-gap finding can't sneak in a `system-prompt:*`
|
|
2584
|
+
* subject (the improvement-analyst's job) and vice versa.
|
|
2585
|
+
*
|
|
2586
|
+
* `failure-mode` is restricted to `cluster` — the only kind that emits
|
|
2587
|
+
* a non-locus subject.
|
|
2588
|
+
*/
|
|
2589
|
+
declare const KIND_EXPECTED_SUBJECTS: Record<string, ReadonlyArray<FindingSubjectKind>>;
|
|
2590
|
+
/** Render only the subject forms one analyst kind is permitted to emit. */
|
|
2591
|
+
declare function findingSubjectGrammarPromptFor(kindId: string): string;
|
|
2592
|
+
/**
|
|
2593
|
+
* Zod schema that validates a raw subject string and returns the parsed
|
|
2594
|
+
* `FindingSubject`. Embedded in `RawAnalystFindingSchema` via
|
|
2595
|
+
* `transform`, so `subject` arrives at the kind factory either as a
|
|
2596
|
+
* typed locus or as a parse error attached to a single Zod issue.
|
|
2597
|
+
*
|
|
2598
|
+
* Optionality is preserved: subjects ARE optional on the wire (some
|
|
2599
|
+
* findings are descriptive, not actionable). When present, they MUST
|
|
2600
|
+
* parse — emitting a malformed subject is a contract violation, not a
|
|
2601
|
+
* soft signal.
|
|
2602
|
+
*/
|
|
2603
|
+
declare const FindingSubjectStringSchema: z.ZodString;
|
|
2604
|
+
|
|
2605
|
+
/**
|
|
2606
|
+
* FindingsStore — durable persistence for AnalystFinding rows + a diff
|
|
2607
|
+
* helper so we can answer "what changed since the last run?" without
|
|
2608
|
+
* recomputing analysts.
|
|
2609
|
+
*
|
|
2610
|
+
* On-disk shape is JSONL: one finding per line, append-only, locked via
|
|
2611
|
+
* LockedJsonlAppender. Operators get crash-safety (no partial JSON),
|
|
2612
|
+
* cheap reads (sequential parse), and trivial backup (rsync the file).
|
|
2613
|
+
*
|
|
2614
|
+
* Reads are non-locking: a reader sees a consistent snapshot of all
|
|
2615
|
+
* fully-written lines and skips an incomplete trailing line if the
|
|
2616
|
+
* writer is mid-append. Cross-process locking is intentionally out of
|
|
2617
|
+
* scope (see locked-jsonl-appender.ts).
|
|
2618
|
+
*
|
|
2619
|
+
* The store is run-scoped: callers pass `runId` on append and on load,
|
|
2620
|
+
* which keeps multi-run files cleanly partitioned. The `diffFindings`
|
|
2621
|
+
* helper compares two run-id sets using stable `finding_id` semantics —
|
|
2622
|
+
* the diff is the cross-run signal the regression dashboard renders.
|
|
2623
|
+
*/
|
|
2624
|
+
|
|
2625
|
+
/**
|
|
2626
|
+
* One persisted row. We attach `run_id` on disk so a single file can
|
|
2627
|
+
* hold multiple runs and the diff helper can query without re-walking
|
|
2628
|
+
* separate files.
|
|
2629
|
+
*/
|
|
2630
|
+
interface PersistedFinding extends AnalystFinding {
|
|
2631
|
+
run_id: string;
|
|
2632
|
+
}
|
|
2633
|
+
declare class FindingsStore {
|
|
2634
|
+
readonly path: string;
|
|
2635
|
+
private readonly appender;
|
|
2636
|
+
constructor(path: string);
|
|
2637
|
+
append(runId: string, findings: AnalystFinding[]): Promise<void>;
|
|
2638
|
+
/** Load every persisted finding. Discards malformed trailing lines silently. */
|
|
2639
|
+
loadAll(): PersistedFinding[];
|
|
2640
|
+
/** Filter to a single run. */
|
|
2641
|
+
loadRun(runId: string): PersistedFinding[];
|
|
2642
|
+
}
|
|
2643
|
+
interface FindingsDiff {
|
|
2644
|
+
/** New finding ids in `current` that weren't in `previous`. */
|
|
2645
|
+
appeared: PersistedFinding[];
|
|
2646
|
+
/** Finding ids in `previous` that aren't in `current`. */
|
|
2647
|
+
disappeared: PersistedFinding[];
|
|
2648
|
+
/** Same finding id present in both runs and unchanged per the materiality test. */
|
|
2649
|
+
persisted: PersistedFinding[];
|
|
2650
|
+
/**
|
|
2651
|
+
* Same finding id in both runs but at least one non-identity field
|
|
2652
|
+
* shifted per `DiffPolicy.isMaterial`. Reported as [previous, current].
|
|
2653
|
+
*/
|
|
2654
|
+
changed: Array<{
|
|
2655
|
+
previous: PersistedFinding;
|
|
2656
|
+
current: PersistedFinding;
|
|
2657
|
+
}>;
|
|
2658
|
+
}
|
|
2659
|
+
interface DiffPolicy {
|
|
2660
|
+
/**
|
|
2661
|
+
* Predicate that decides whether two findings (same finding_id) count
|
|
2662
|
+
* as a material change. Defaults to {@link defaultIsMaterial}: severity
|
|
2663
|
+
* shift, confidence Δ > 0.05, or evidence count change. Compliance /
|
|
2664
|
+
* perf consumers MAY supply a stricter predicate (e.g. rationale text
|
|
2665
|
+
* diff, metric Δ thresholds).
|
|
2666
|
+
*/
|
|
2667
|
+
isMaterial?: (previous: AnalystFinding, current: AnalystFinding) => boolean;
|
|
2668
|
+
}
|
|
2669
|
+
/**
|
|
2670
|
+
* Default materiality test. Deliberately narrow so LLM-reword churn
|
|
2671
|
+
* doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.
|
|
2672
|
+
*/
|
|
2673
|
+
declare function defaultIsMaterial(a: AnalystFinding, b: AnalystFinding): boolean;
|
|
2674
|
+
/**
|
|
2675
|
+
* Diff two findings sets by stable finding_id. Callers typically load
|
|
2676
|
+
* the two run-id slices from the same store and pass them in.
|
|
2677
|
+
*/
|
|
2678
|
+
declare function diffFindings(previous: PersistedFinding[], current: PersistedFinding[], policy?: DiffPolicy): FindingsDiff;
|
|
2679
|
+
|
|
2680
|
+
/**
|
|
2681
|
+
* Failure-mode analyst — classifies what went wrong and why.
|
|
2682
|
+
*
|
|
2683
|
+
* Brief: read the trace dataset, identify the top failure modes across
|
|
2684
|
+
* runs, classify each with severity + evidence, and surface them as
|
|
2685
|
+
* findings. The actor's job is *taxonomy + evidence*, not fix-design —
|
|
2686
|
+
* that's the improvement-analyst's job.
|
|
2687
|
+
*
|
|
2688
|
+
* Eight bounded model subqueries let the actor compare candidate
|
|
2689
|
+
* clusters in parallel after it has loaded representative evidence.
|
|
2690
|
+
*/
|
|
2691
|
+
|
|
2692
|
+
declare const FAILURE_MODE_KIND_SPEC: TraceAnalystKindSpec;
|
|
2693
|
+
|
|
2694
|
+
/**
|
|
2695
|
+
* Improvement analyst — actionable self-improvement findings.
|
|
2696
|
+
*
|
|
2697
|
+
* Brief: read findings from upstream analysts (failure-mode,
|
|
2698
|
+
* knowledge-gap, knowledge-poisoning) AND the trace dataset itself,
|
|
2699
|
+
* then propose **concrete edits** to the agent's runtime: prompt
|
|
2700
|
+
* additions, RAG documents to ingest, tool descriptions to rewrite,
|
|
2701
|
+
* scaffolding changes to make, memory entries to invalidate. Each
|
|
2702
|
+
* finding is one proposed edit with the locus, the diff, and the
|
|
2703
|
+
* expected effect.
|
|
2704
|
+
*
|
|
2705
|
+
* This is the self-improvement loop's last mile: the prior
|
|
2706
|
+
* kinds describe *what's wrong*; this kind describes *what to change*.
|
|
2707
|
+
*
|
|
2708
|
+
* Eight bounded model subqueries let the actor compare competing fix
|
|
2709
|
+
* directions over the same cited evidence before recommending one.
|
|
2710
|
+
*/
|
|
2711
|
+
|
|
2712
|
+
declare const IMPROVEMENT_KIND_SPEC: TraceAnalystKindSpec;
|
|
2713
|
+
|
|
2714
|
+
/**
|
|
2715
|
+
* Knowledge-gap analyst — what did the agent NOT know that it needed?
|
|
2716
|
+
*
|
|
2717
|
+
* Brief: find moments in the trace where the agent had to guess, ask
|
|
2718
|
+
* the user to fill in context, recover from a wrong assumption, or
|
|
2719
|
+
* loop on a retrieval. Each finding names a *missing or outdated piece
|
|
2720
|
+
* of knowledge* the agent's curated knowledge base should have held —
|
|
2721
|
+
* or a downstream lookup (web, docs, tool description) that surfaced
|
|
2722
|
+
* stale or outdated information.
|
|
2723
|
+
*
|
|
2724
|
+
* The primary expected store is `@tangle-network/agent-knowledge`: a
|
|
2725
|
+
* Karpathy-style wiki the agent maintains with raw ↔ curated pages,
|
|
2726
|
+
* source anchors, and claim/relation triples. A gap is anything the
|
|
2727
|
+
* agent had to discover at run-time that should already have lived
|
|
2728
|
+
* there. Secondary loci: web-search results that returned outdated
|
|
2729
|
+
* pages, tool descriptions that omitted critical behavior, system-
|
|
2730
|
+
* prompt sections that didn't cover the case.
|
|
2731
|
+
*
|
|
2732
|
+
* Distinct from failure-mode: failure-mode classifies *how* it broke;
|
|
2733
|
+
* knowledge-gap names the *information* whose absence (or staleness)
|
|
2734
|
+
* caused the break. One failure-mode often maps to several gaps.
|
|
2735
|
+
*
|
|
2736
|
+
* Five bounded model subqueries let the actor compare candidate gaps
|
|
2737
|
+
* across source layers after it has loaded the relevant excerpts.
|
|
2738
|
+
*/
|
|
2739
|
+
|
|
2740
|
+
declare const KNOWLEDGE_GAP_KIND_SPEC: TraceAnalystKindSpec;
|
|
2741
|
+
|
|
2742
|
+
/**
|
|
2743
|
+
* Knowledge-poisoning analyst — what FALSE information misled the agent?
|
|
2744
|
+
*
|
|
2745
|
+
* Brief: find moments where the agent acted on information that was
|
|
2746
|
+
* *wrong* — stale memory, RAG documents that contradicted ground truth,
|
|
2747
|
+
* tool descriptions that lied about return shapes, system-prompt
|
|
2748
|
+
* instructions that no longer matched reality, prior-run summaries that
|
|
2749
|
+
* cached a wrong decision.
|
|
2750
|
+
*
|
|
2751
|
+
* Distinct from knowledge-gap: a gap is "the agent didn't know X"; a
|
|
2752
|
+
* poisoning is "the agent confidently used X, but X was wrong." Gaps
|
|
2753
|
+
* surface as questions / self-correction; poisonings surface as
|
|
2754
|
+
* confident-but-wrong actions that downstream evidence contradicts.
|
|
2755
|
+
*
|
|
2756
|
+
* Eight bounded model subqueries let the actor independently assess
|
|
2757
|
+
* the action and contradiction excerpts for candidate poisonings.
|
|
2758
|
+
*/
|
|
2759
|
+
|
|
2760
|
+
declare const KNOWLEDGE_POISONING_KIND_SPEC: TraceAnalystKindSpec;
|
|
2761
|
+
|
|
2762
|
+
/**
|
|
2763
|
+
* Default analyst kinds focused on agent failure + recursive
|
|
2764
|
+
* self-improvement.
|
|
2765
|
+
*
|
|
2766
|
+
* The four kinds chain: failure-mode classifies; knowledge-gap and
|
|
2767
|
+
* knowledge-poisoning explain *why* in two orthogonal ways; improvement
|
|
2768
|
+
* proposes concrete edits. Register all four against the same trace
|
|
2769
|
+
* store in this order and run the registry with `chainFindings: true`
|
|
2770
|
+
* to pass each completed kind's findings to the kinds that follow it.
|
|
2771
|
+
*/
|
|
2772
|
+
|
|
2773
|
+
/**
|
|
2774
|
+
* The default kind suite. Order is the run order operators should
|
|
2775
|
+
* use: failure-mode first (no upstream deps), gap + poisoning next
|
|
2776
|
+
* (both depend on failures), improvement last (chains all three).
|
|
2777
|
+
*/
|
|
2778
|
+
declare const DEFAULT_TRACE_ANALYST_KINDS: readonly TraceAnalystKindSpec[];
|
|
2779
|
+
|
|
2780
|
+
/**
|
|
2781
|
+
* Skill-usage analyst — a DETERMINISTIC `Analyst` over a Claude/Codex skill
|
|
2782
|
+
* library + its trace corpus. Unlike the trace-store kinds (failure-mode,
|
|
2783
|
+
* improvement, ...) this kind calls no LLM: it mines real usage and skill
|
|
2784
|
+
* structure and emits findings by rule.
|
|
2785
|
+
*
|
|
2786
|
+
* It exists because the naive "Skill-tool invocation count" lies low — it
|
|
2787
|
+
* misses orchestrated sub-dispatch (a leaf skill run BY /pursue or /governor
|
|
2788
|
+
* logs under the parent), slash-command entry, local-script bypass, and
|
|
2789
|
+
* on-disk artifacts. The 2026-05-30 skill audit found 39/53 skills at zero
|
|
2790
|
+
* direct invocations, yet only one was a genuine cut: the rest were
|
|
2791
|
+
* measurement-invisible or discovery-limited. This analyst encodes that
|
|
2792
|
+
* lesson as a multi-signal usage model so a cheap repeatable pass can keep
|
|
2793
|
+
* the library honest, and so the expensive audit workflow's verdicts can
|
|
2794
|
+
* GEPA-distill it toward agreement (see `gold/skill-verdicts.gold.jsonl`).
|
|
2795
|
+
*
|
|
2796
|
+
* Report-building (`buildSkillUsageReport`, an fs scan) is separated from
|
|
2797
|
+
* finding emission (`SkillUsageAnalyst.analyze`, pure) so the slow scan runs
|
|
2798
|
+
* once at the registry boundary and the rule logic stays unit-testable.
|
|
2799
|
+
*/
|
|
2800
|
+
|
|
2801
|
+
type SkillKind = 'public' | 'private';
|
|
2802
|
+
/** One skill's multi-signal usage + structure. All counts are deterministic. */
|
|
2803
|
+
interface SkillUsageRecord {
|
|
2804
|
+
name: string;
|
|
2805
|
+
kind: SkillKind;
|
|
2806
|
+
/** Absolute path to the skill's SKILL.md. */
|
|
2807
|
+
path: string;
|
|
2808
|
+
lines: number;
|
|
2809
|
+
/** `"skill":"<name>"` Skill-tool invocations across the trace corpus. */
|
|
2810
|
+
directInvocations: number;
|
|
2811
|
+
/** `<command-name>/<name>` slash invocations across the trace corpus. */
|
|
2812
|
+
slashInvocations: number;
|
|
2813
|
+
/** Sibling skills whose SKILL.md dispatches to this one (`/<name>`). Proxy
|
|
2814
|
+
* for orchestrated sub-dispatch the per-skill counter cannot see. */
|
|
2815
|
+
inboundRefs: number;
|
|
2816
|
+
/** On-disk artifacts attributable to the skill (e.g. `.evolve/<name>/**`). */
|
|
2817
|
+
artifactCount: number;
|
|
2818
|
+
/** Tangle-private reference count in the body (leak signal for public skills). */
|
|
2819
|
+
tanglePrivateRefs: number;
|
|
2820
|
+
hasReferencesDir: boolean;
|
|
2821
|
+
hasEvalsDir: boolean;
|
|
2822
|
+
/** Body mentions `skill-runs.jsonl` (visible to /reflect + /governor). */
|
|
2823
|
+
logsRuns: boolean;
|
|
2824
|
+
/** Description carries an explicit `Triggers:` clause / trigger phrases. */
|
|
2825
|
+
hasTriggerPhrases: boolean;
|
|
2826
|
+
}
|
|
2827
|
+
interface SkillUsageReport {
|
|
2828
|
+
generatedFromTraces: number;
|
|
2829
|
+
records: SkillUsageRecord[];
|
|
2830
|
+
}
|
|
2831
|
+
interface SkillUsageScanConfig {
|
|
2832
|
+
/** Dirs holding `*.jsonl` transcripts (Claude `~/.claude/projects`, Codex sessions). */
|
|
2833
|
+
transcriptDirs: string[];
|
|
2834
|
+
/** Skill roots to scan; each dir directly under `root` with a `SKILL.md` is a skill. */
|
|
2835
|
+
skillRoots: {
|
|
2836
|
+
root: string;
|
|
2837
|
+
kind: SkillKind;
|
|
2838
|
+
}[];
|
|
2839
|
+
/** Roots scanned for `<root>/.evolve/<skill>` artifact dirs. */
|
|
2840
|
+
artifactRoots?: string[];
|
|
2841
|
+
/** Token-prefixed mappings: skill name → extra artifact subpaths under an artifactRoot
|
|
2842
|
+
* (e.g. reflect → `.evolve/reflections`). Catches non-eponymous artifact dirs. */
|
|
2843
|
+
artifactAliases?: Record<string, string[]>;
|
|
2844
|
+
/** Cap files read per transcript dir (bounds a huge corpus); 0 = unbounded. */
|
|
2845
|
+
maxTranscriptsPerDir?: number;
|
|
2846
|
+
}
|
|
2847
|
+
/** Scan the corpus + skill roots into a {@link SkillUsageReport}. Deterministic. */
|
|
2848
|
+
declare function buildSkillUsageReport(config: SkillUsageScanConfig): SkillUsageReport;
|
|
2849
|
+
/** Pure rule pass over a report → findings. Exported for direct/unit use. */
|
|
2850
|
+
declare function emitSkillUsageFindings(report: SkillUsageReport, producedAt: string): AnalystFinding[];
|
|
2851
|
+
declare class SkillUsageAnalyst implements Analyst<SkillUsageReport> {
|
|
2852
|
+
readonly id = "skill-usage";
|
|
2853
|
+
readonly description = "Deterministic multi-signal skill-usage analysis: flags dead skills, measurement-invisible (orchestrated) usage, discovery gaps, public-repo leaks, bloat, missing evals, and missing run-logging.";
|
|
2854
|
+
readonly inputKind: "custom";
|
|
2855
|
+
readonly cost: {
|
|
2856
|
+
kind: "deterministic";
|
|
2857
|
+
est_usd_per_run: number;
|
|
2858
|
+
};
|
|
2859
|
+
readonly version = "1.0.0";
|
|
2860
|
+
analyze(input: SkillUsageReport, ctx: AnalystContext): Promise<AnalystFinding[]>;
|
|
2861
|
+
}
|
|
2862
|
+
declare const SKILL_USAGE_ANALYST: SkillUsageAnalyst;
|
|
2863
|
+
|
|
2864
|
+
/**
|
|
2865
|
+
* Forgiving pre-parse for analyst findings. Weak models routinely emit
|
|
2866
|
+
* schema-correct content in an unusable wrapper — fenced ```json blocks, a
|
|
2867
|
+
* single object where an array is expected, trailing commas. Measured: GPT-4o
|
|
2868
|
+
* drops to 0% usable output purely from markdown-fence wrapping
|
|
2869
|
+
* (arXiv:2605.02363). A five-line de-fence recovers most of it. This module is
|
|
2870
|
+
* the de-fence/coerce step that runs BEFORE Zod, so a recoverable finding is
|
|
2871
|
+
* repaired, not dropped.
|
|
2872
|
+
*
|
|
2873
|
+
* Pure + deterministic. No model, no network.
|
|
2874
|
+
*/
|
|
2875
|
+
/** Strip a ```lang ... ``` (or bare ``` ... ```) code fence, if the string is one. */
|
|
2876
|
+
declare function stripCodeFences(text: string): string;
|
|
2877
|
+
/**
|
|
2878
|
+
* Best-effort parse of a string into JSON. De-fences, drops trailing commas,
|
|
2879
|
+
* then `JSON.parse`. Returns `undefined` (never throws) when unrecoverable.
|
|
2880
|
+
*/
|
|
2881
|
+
declare function coerceJson(text: string): unknown;
|
|
2882
|
+
/**
|
|
2883
|
+
* Coerce arbitrary actor/structurer output into an array of candidate finding
|
|
2884
|
+
* rows: a JSON string → parse; a single object → 1-element array; an array →
|
|
2885
|
+
* as-is; anything else → []. Callers still run each row through Zod
|
|
2886
|
+
* (`parseCanonicalRawFinding`) — this only fixes the SHAPE, never invents fields.
|
|
2887
|
+
*/
|
|
2888
|
+
declare function coerceToFindingRows(raw: unknown): unknown[];
|
|
2889
|
+
|
|
2890
|
+
type PolicyEditSchemaVersion = 'policy-edit/v1';
|
|
2891
|
+
declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
|
|
2892
|
+
type PolicyEditAxis = (typeof POLICY_EDIT_AXES)[number];
|
|
2893
|
+
declare const POLICY_EDIT_TARGET_SURFACES: readonly ["prompt", "tool-contract", "runtime-config", "memory", "agent-profile", "code", "deployment"];
|
|
2894
|
+
type PolicyEditTargetSurface = (typeof POLICY_EDIT_TARGET_SURFACES)[number];
|
|
2895
|
+
type PolicyEditRisk = 'low' | 'medium' | 'high' | 'unknown';
|
|
2896
|
+
type PolicyEditGainDirection = 'increase' | 'decrease';
|
|
2897
|
+
type PolicyEditGainUnit = 'absolute' | 'relative' | 'percent' | 'score';
|
|
2898
|
+
interface PolicyEditTarget {
|
|
2899
|
+
surface: PolicyEditTargetSurface;
|
|
2900
|
+
/** Stable path inside the target surface, for example `system-prompt:tools`
|
|
2901
|
+
* or `budget.maxTurns`. */
|
|
2902
|
+
path?: string;
|
|
2903
|
+
/** Optional canonical deployment identity. Store the existing cell, not a
|
|
2904
|
+
* local profile shape. */
|
|
2905
|
+
agentProfileCell?: AgentProfileCell;
|
|
2906
|
+
/** Human label when the path is not enough for a readable audit trail. */
|
|
2907
|
+
label?: string;
|
|
2908
|
+
}
|
|
2909
|
+
type PolicyEditChange = {
|
|
2910
|
+
kind: 'text';
|
|
2911
|
+
mode: 'append' | 'prepend' | 'replace';
|
|
2912
|
+
value: string;
|
|
2913
|
+
/** Required when `mode === 'replace'`; exact match only. */
|
|
2914
|
+
find?: string;
|
|
2915
|
+
} | {
|
|
2916
|
+
kind: 'json';
|
|
2917
|
+
mode: 'set' | 'merge' | 'remove';
|
|
2918
|
+
path: string;
|
|
2919
|
+
value?: AgentProfileJson;
|
|
2920
|
+
};
|
|
2921
|
+
interface PolicyEditExpectedGain {
|
|
2922
|
+
/** Metric this edit is expected to move, e.g. `holdout.composite`. */
|
|
2923
|
+
metric: string;
|
|
2924
|
+
direction: PolicyEditGainDirection;
|
|
2925
|
+
/** Positive magnitude in the metric's native units. */
|
|
2926
|
+
amount: number;
|
|
2927
|
+
unit?: PolicyEditGainUnit;
|
|
2928
|
+
rationale?: string;
|
|
2929
|
+
}
|
|
2930
|
+
interface PolicyEditSource {
|
|
2931
|
+
findingIds: string[];
|
|
2932
|
+
analystIds: string[];
|
|
2933
|
+
evidenceRefs: EvidenceRef[];
|
|
2934
|
+
/** Mirrors `AnalystFinding.derived_from_judge`; admission rejects it. */
|
|
2935
|
+
derivedFromJudge?: boolean;
|
|
2936
|
+
}
|
|
2937
|
+
interface PolicyEdit {
|
|
2938
|
+
schemaVersion: PolicyEditSchemaVersion;
|
|
2939
|
+
editId: string;
|
|
2940
|
+
axis: PolicyEditAxis;
|
|
2941
|
+
target: PolicyEditTarget;
|
|
2942
|
+
change: PolicyEditChange;
|
|
2943
|
+
claim: string;
|
|
2944
|
+
expectedGain: PolicyEditExpectedGain;
|
|
2945
|
+
confidence: number;
|
|
2946
|
+
risk: PolicyEditRisk;
|
|
2947
|
+
source: PolicyEditSource;
|
|
2948
|
+
rationale?: string;
|
|
2949
|
+
validationPlan?: string;
|
|
2950
|
+
metadata?: Record<string, unknown>;
|
|
2951
|
+
}
|
|
2952
|
+
declare const POLICY_EDIT_CANDIDATE_RECORD_SCHEMA: "tangle.policy-edit-candidate.v1";
|
|
2953
|
+
/** JSON-safe attribution carried with a measured candidate and its scores. */
|
|
2954
|
+
interface PolicyEditCandidateRecord {
|
|
2955
|
+
schema: typeof POLICY_EDIT_CANDIDATE_RECORD_SCHEMA;
|
|
2956
|
+
policyEdit: PolicyEdit;
|
|
2957
|
+
}
|
|
2958
|
+
type PolicyEditInit = Omit<PolicyEdit, 'schemaVersion' | 'editId'> & {
|
|
2959
|
+
schemaVersion?: PolicyEditSchemaVersion;
|
|
2960
|
+
editId?: string;
|
|
2961
|
+
};
|
|
2962
|
+
declare class PolicyEditValidationError extends ValidationError {
|
|
2963
|
+
readonly path: string;
|
|
2964
|
+
constructor(message: string, path?: string);
|
|
2965
|
+
}
|
|
2966
|
+
interface FindingToPolicyEditOptions {
|
|
2967
|
+
expectedGain?: PolicyEditExpectedGain | ((finding: AnalystFinding) => PolicyEditExpectedGain | null | undefined);
|
|
2968
|
+
risk?: PolicyEditRisk | ((finding: AnalystFinding) => PolicyEditRisk);
|
|
2969
|
+
defaultAxis?: PolicyEditAxis;
|
|
2970
|
+
defaultTargetSurface?: PolicyEditTargetSurface;
|
|
2971
|
+
}
|
|
2972
|
+
interface PolicyEditAdmissionOptions {
|
|
2973
|
+
minScore?: number;
|
|
2974
|
+
minExpectedGain?: number;
|
|
2975
|
+
allowHighRisk?: boolean;
|
|
2976
|
+
requireEvidence?: boolean;
|
|
2977
|
+
}
|
|
2978
|
+
interface PolicyEditAdmission {
|
|
2979
|
+
edit: PolicyEdit;
|
|
2980
|
+
decision: 'admit' | 'reject';
|
|
2981
|
+
score: number;
|
|
2982
|
+
reasons: string[];
|
|
2983
|
+
}
|
|
2984
|
+
declare function makePolicyEdit(init: PolicyEditInit): PolicyEdit;
|
|
2985
|
+
declare function computePolicyEditId(edit: Omit<PolicyEdit, 'editId'> | PolicyEdit): string;
|
|
2986
|
+
declare function validatePolicyEdit(input: unknown): PolicyEdit;
|
|
2987
|
+
declare function makePolicyEditCandidateRecord(edit: PolicyEdit): PolicyEditCandidateRecord;
|
|
2988
|
+
declare function validatePolicyEditCandidateRecord(input: unknown): PolicyEditCandidateRecord;
|
|
2989
|
+
declare function isPolicyEdit(input: unknown): input is PolicyEdit;
|
|
2990
|
+
declare function policyEditsFromFindings(findings: ReadonlyArray<AnalystFinding>, opts?: FindingToPolicyEditOptions): PolicyEdit[];
|
|
2991
|
+
declare function policyEditFromFinding(finding: AnalystFinding, opts?: FindingToPolicyEditOptions): PolicyEdit | null;
|
|
2992
|
+
declare function scorePolicyEditReadiness(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): number;
|
|
2993
|
+
declare function admitPolicyEdit(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): PolicyEditAdmission;
|
|
2994
|
+
declare function applyPolicyEditToSurface(surface: unknown, edit: PolicyEdit): unknown;
|
|
2995
|
+
|
|
2996
|
+
/** DESCRIPTIVE predicate: does the finding cite at least one observable
|
|
2997
|
+
* (span/event/artifact) evidence ref. Useful for ranking evidence quality or
|
|
2998
|
+
* rendering — it is NOT the steer gate. Evidence presence is the WRONG
|
|
2999
|
+
* discriminator for steering: a legitimate trace-analyst observation may cite
|
|
3000
|
+
* nothing (it would be wrongly rejected), and a judge verdict may cite an
|
|
3001
|
+
* artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate
|
|
3002
|
+
* steering; use this only where "is this grounded in observable evidence" is the
|
|
3003
|
+
* literal question. */
|
|
3004
|
+
declare function isTraceObservable(finding: AnalystFinding): boolean;
|
|
3005
|
+
/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a
|
|
3006
|
+
* finding), identified by provenance set at the lift site — independent of
|
|
3007
|
+
* whatever evidence it cites. */
|
|
3008
|
+
declare function isJudgeVerdict(finding: AnalystFinding): boolean;
|
|
3009
|
+
/**
|
|
3010
|
+
* THE steer firewall. Fail-loud guard for any path that admits analyst findings
|
|
3011
|
+
* as STEERING input (the `f(trace)` role): rejects — naming the offenders — any
|
|
3012
|
+
* finding whose provenance is a judge verdict, rather than let `J` leak into the
|
|
3013
|
+
* loop. Returns the findings unchanged for chaining.
|
|
3014
|
+
*
|
|
3015
|
+
* Call this at the chokepoint where a detector that ALSO scores/gates has its
|
|
3016
|
+
* findings turned into a steer (the judge-and-steer dual-role case). It keys on
|
|
3017
|
+
* provenance, so it correctly admits evidence-less trace-analyst observations and
|
|
3018
|
+
* correctly rejects an artifact-citing judge verdict — the cases an evidence
|
|
3019
|
+
* check gets backwards.
|
|
3020
|
+
*
|
|
3021
|
+
* It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge
|
|
3022
|
+
* whose output is laundered through a hand-built finding with no provenance flag
|
|
3023
|
+
* is out of its reach — provenance must be honestly set at every judge→finding
|
|
3024
|
+
* lift (today: createJudgeAdapter). That is why the integrity rule lives at the
|
|
3025
|
+
* lift site, and why ProposeContext.judgeScores?: never is the complementary
|
|
3026
|
+
* compile-time tripwire on the obvious direct channel.
|
|
3027
|
+
*/
|
|
3028
|
+
declare function assertNoJudgeVerdict(findings: ReadonlyArray<AnalystFinding>, context?: string): ReadonlyArray<AnalystFinding>;
|
|
3029
|
+
|
|
3030
|
+
/**
|
|
3031
|
+
* `structureFindings` — the deferred structuring pass (DSPy TwoStepAdapter /
|
|
3032
|
+
* HALO `synthesize_traces` analog). The agentic actor reasons FREE-FORM and
|
|
3033
|
+
* emits a prose `report` (which any model does reliably); this separate, cheap
|
|
3034
|
+
* call's ONLY job is to turn that report into `AnalystFinding[]`. Decoupling
|
|
3035
|
+
* reasoning from structuring is what makes the SEMANTIC findings model-agnostic
|
|
3036
|
+
* — the reasoning model never has to satisfy a strict typed-array contract
|
|
3037
|
+
* while it diagnoses.
|
|
3038
|
+
*
|
|
3039
|
+
* Forgiving: the response runs through `coerceToFindingRows` (de-fence, lift
|
|
3040
|
+
* single→array) before Zod, and on a zero-finding extraction from a substantive
|
|
3041
|
+
* report it reasks ONCE with the schema restated. Returns a typed outcome so a
|
|
3042
|
+
* legitimate "nothing to report" is distinguishable from a failed extraction
|
|
3043
|
+
* (no silent empty).
|
|
3044
|
+
*/
|
|
3045
|
+
|
|
3046
|
+
interface StructureFindingsOptions {
|
|
3047
|
+
/** The actor's free-form diagnosis prose. */
|
|
3048
|
+
report: string;
|
|
3049
|
+
analystId: string;
|
|
3050
|
+
/** Coarse classification stamped on every extracted finding. */
|
|
3051
|
+
area: string;
|
|
3052
|
+
model: string;
|
|
3053
|
+
baseUrl: string;
|
|
3054
|
+
apiKey?: string;
|
|
3055
|
+
/** Optional ledger for direct use. */
|
|
3056
|
+
costLedger?: CostLedgerHandle;
|
|
3057
|
+
costPhase?: string;
|
|
3058
|
+
costTags?: Record<string, string>;
|
|
3059
|
+
maxTokens?: number;
|
|
3060
|
+
signal?: AbortSignal;
|
|
3061
|
+
/** Max reask attempts after a zero/invalid extraction. Default 1. */
|
|
3062
|
+
maxReasks?: number;
|
|
3063
|
+
/** Apply the caller's normal finding rules before a recovered row is lifted. */
|
|
3064
|
+
processRow?: (row: RawAnalystFinding) => RawAnalystFinding | null;
|
|
3065
|
+
/** Apply canonical multi-citation rules after any original callback. */
|
|
3066
|
+
processCanonicalRow?: (row: CanonicalRawAnalystFinding) => CanonicalRawAnalystFinding | null;
|
|
3067
|
+
/** Provenance copied onto every recovered finding. */
|
|
3068
|
+
findingMetadata?: Record<string, unknown>;
|
|
3069
|
+
/** Test seam: inject a fetch (no network in unit tests). */
|
|
3070
|
+
fetchImpl?: LlmClientOptions['fetch'];
|
|
3071
|
+
}
|
|
3072
|
+
interface StructureFindingsResult {
|
|
3073
|
+
findings: AnalystFinding[];
|
|
3074
|
+
outcome: 'ok' | 'extraction_failed';
|
|
3075
|
+
}
|
|
3076
|
+
declare function structureFindings(opts: StructureFindingsOptions): Promise<StructureFindingsResult>;
|
|
3077
|
+
|
|
3078
|
+
/**
|
|
3079
|
+
* Pre-curated tool subsets for analyst kinds.
|
|
3080
|
+
*
|
|
3081
|
+
* The full trace-analyst tool set is seven functions. Most kinds only
|
|
3082
|
+
* need three or four. Picking from named groups instead of importing
|
|
3083
|
+
* the whole bundle keeps every kind's actor-context budget tight and
|
|
3084
|
+
* makes "what can this analyst see?" obvious at registration time.
|
|
3085
|
+
*
|
|
3086
|
+
* Each function in the group keeps its full `name`/`description` from
|
|
3087
|
+
* `buildTraceAnalystTools` — we filter, we don't re-implement.
|
|
3088
|
+
*/
|
|
3089
|
+
|
|
3090
|
+
/** Named tool sets. Kinds pass `tools: TRACE_TOOL_GROUPS.failureForensics` etc. */
|
|
3091
|
+
type TraceToolGroupName =
|
|
3092
|
+
/** All seven tools. Use for open-ended discovery kinds. */
|
|
3093
|
+
'all'
|
|
3094
|
+
/** Overview + paginated query + count. No deep reads. Cheap. */
|
|
3095
|
+
| 'discovery'
|
|
3096
|
+
/** Discovery + viewTrace + viewSpans. Deep-read but no regex search. */
|
|
3097
|
+
| 'discoveryAndRead'
|
|
3098
|
+
/** Discovery + search tools. For pattern-matching across many traces. */
|
|
3099
|
+
| 'discoveryAndSearch'
|
|
3100
|
+
/** Discovery + viewSpans + searchSpan. Targeted-span work after another kind narrows down. */
|
|
3101
|
+
| 'targeted';
|
|
3102
|
+
/**
|
|
3103
|
+
* Build the tool set for a named group bound to a specific trace store.
|
|
3104
|
+
*
|
|
3105
|
+
* `all` returns every tool. Other groups filter `buildTraceAnalystTools`
|
|
3106
|
+
* by name to the documented subset. An unrecognised group name throws —
|
|
3107
|
+
* silently returning all tools would defeat the cost-control point.
|
|
3108
|
+
*/
|
|
3109
|
+
declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): AxFunction[];
|
|
3110
|
+
|
|
3111
|
+
export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type CanonicalRawAnalystFinding, CanonicalRawAnalystFindingSchema, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingToPolicyEditOptions, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, POLICY_EDIT_AXES, POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, POLICY_EDIT_TARGET_SURFACES, type PersistedFinding, type PolicyEdit, type PolicyEditAdmission, type PolicyEditAdmissionOptions, type PolicyEditAxis, type PolicyEditCandidateRecord, type PolicyEditChange, type PolicyEditExpectedGain, type PolicyEditGainDirection, type PolicyEditGainUnit, type PolicyEditInit, type PolicyEditRisk, type PolicyEditSchemaVersion, type PolicyEditSource, type PolicyEditTarget, type PolicyEditTargetSurface, PolicyEditValidationError, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts, admitPolicyEdit, applyPolicyEditToSurface, assertNoJudgeVerdict, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, computePolicyEditId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isJudgeVerdict, isPolicyEdit, isTraceObservable, liftSeverity, makeFinding, makePolicyEdit, makePolicyEditCandidateRecord, parseCanonicalRawFinding, parseFindingSubject, parseRawFinding, policyEditFromFinding, policyEditsFromFindings, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, scorePolicyEditReadiness, stripCodeFences, structureFindings, validatePolicyEdit, validatePolicyEditCandidateRecord };
|