@tangle-network/agent-eval 0.120.0 → 0.120.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/package.json +1 -1
  3. package/dist/analyst/index.d.ts +0 -3111
  4. package/dist/analyst/index.js +0 -403
  5. package/dist/analyst/index.js.map +0 -1
  6. package/dist/authenticity/index.d.ts +0 -161
  7. package/dist/authenticity/index.js +0 -215
  8. package/dist/authenticity/index.js.map +0 -1
  9. package/dist/belief-state/index.d.ts +0 -1301
  10. package/dist/belief-state/index.js +0 -2152
  11. package/dist/belief-state/index.js.map +0 -1
  12. package/dist/benchmarks/index.d.ts +0 -974
  13. package/dist/benchmarks/index.js +0 -60
  14. package/dist/benchmarks/index.js.map +0 -1
  15. package/dist/builder-eval/index.d.ts +0 -695
  16. package/dist/builder-eval/index.js +0 -366
  17. package/dist/builder-eval/index.js.map +0 -1
  18. package/dist/campaign/index.d.ts +0 -7454
  19. package/dist/campaign/index.js +0 -272
  20. package/dist/campaign/index.js.map +0 -1
  21. package/dist/chunk-32BZXMSO.js +0 -3878
  22. package/dist/chunk-32BZXMSO.js.map +0 -1
  23. package/dist/chunk-3A246TSA.js +0 -998
  24. package/dist/chunk-3A246TSA.js.map +0 -1
  25. package/dist/chunk-3RF76KTD.js +0 -84
  26. package/dist/chunk-3RF76KTD.js.map +0 -1
  27. package/dist/chunk-3YYRZDON.js +0 -45
  28. package/dist/chunk-3YYRZDON.js.map +0 -1
  29. package/dist/chunk-4I2E3LLO.js +0 -1030
  30. package/dist/chunk-4I2E3LLO.js.map +0 -1
  31. package/dist/chunk-ARU2PZFM.js +0 -312
  32. package/dist/chunk-ARU2PZFM.js.map +0 -1
  33. package/dist/chunk-BOD4O7OF.js +0 -40
  34. package/dist/chunk-BOD4O7OF.js.map +0 -1
  35. package/dist/chunk-DPZAEKA6.js +0 -880
  36. package/dist/chunk-DPZAEKA6.js.map +0 -1
  37. package/dist/chunk-DTJ6QUQB.js +0 -131
  38. package/dist/chunk-DTJ6QUQB.js.map +0 -1
  39. package/dist/chunk-GGE4NNQT.js +0 -65
  40. package/dist/chunk-GGE4NNQT.js.map +0 -1
  41. package/dist/chunk-H5UD2323.js +0 -286
  42. package/dist/chunk-H5UD2323.js.map +0 -1
  43. package/dist/chunk-HHWE3POT.js +0 -94
  44. package/dist/chunk-HHWE3POT.js.map +0 -1
  45. package/dist/chunk-HKUCJ437.js +0 -787
  46. package/dist/chunk-HKUCJ437.js.map +0 -1
  47. package/dist/chunk-JHCHEVET.js +0 -274
  48. package/dist/chunk-JHCHEVET.js.map +0 -1
  49. package/dist/chunk-JHOJHHU7.js +0 -867
  50. package/dist/chunk-JHOJHHU7.js.map +0 -1
  51. package/dist/chunk-JM2SKQMS.js +0 -750
  52. package/dist/chunk-JM2SKQMS.js.map +0 -1
  53. package/dist/chunk-JN2FCO5W.js +0 -7958
  54. package/dist/chunk-JN2FCO5W.js.map +0 -1
  55. package/dist/chunk-K4DBDHLK.js +0 -158
  56. package/dist/chunk-K4DBDHLK.js.map +0 -1
  57. package/dist/chunk-K6N6XJJX.js +0 -306
  58. package/dist/chunk-K6N6XJJX.js.map +0 -1
  59. package/dist/chunk-MA6HLL3S.js +0 -65
  60. package/dist/chunk-MA6HLL3S.js.map +0 -1
  61. package/dist/chunk-MAZ26DC7.js +0 -99
  62. package/dist/chunk-MAZ26DC7.js.map +0 -1
  63. package/dist/chunk-MOXWMGPC.js +0 -577
  64. package/dist/chunk-MOXWMGPC.js.map +0 -1
  65. package/dist/chunk-NJC7U437.js +0 -626
  66. package/dist/chunk-NJC7U437.js.map +0 -1
  67. package/dist/chunk-NPCTHQIO.js +0 -91
  68. package/dist/chunk-NPCTHQIO.js.map +0 -1
  69. package/dist/chunk-ONWEPEDO.js +0 -57
  70. package/dist/chunk-ONWEPEDO.js.map +0 -1
  71. package/dist/chunk-OYZAPX5G.js +0 -1526
  72. package/dist/chunk-OYZAPX5G.js.map +0 -1
  73. package/dist/chunk-PC4UYEBM.js +0 -166
  74. package/dist/chunk-PC4UYEBM.js.map +0 -1
  75. package/dist/chunk-PICTDURQ.js +0 -766
  76. package/dist/chunk-PICTDURQ.js.map +0 -1
  77. package/dist/chunk-PJQFMIOX.js +0 -1182
  78. package/dist/chunk-PJQFMIOX.js.map +0 -1
  79. package/dist/chunk-PXD6ZFNY.js +0 -1107
  80. package/dist/chunk-PXD6ZFNY.js.map +0 -1
  81. package/dist/chunk-PXE2VKMX.js +0 -140
  82. package/dist/chunk-PXE2VKMX.js.map +0 -1
  83. package/dist/chunk-PZ5AY32C.js +0 -10
  84. package/dist/chunk-PZ5AY32C.js.map +0 -1
  85. package/dist/chunk-QBRSJK47.js +0 -622
  86. package/dist/chunk-QBRSJK47.js.map +0 -1
  87. package/dist/chunk-QWMPPZ3X.js +0 -550
  88. package/dist/chunk-QWMPPZ3X.js.map +0 -1
  89. package/dist/chunk-S3UZOQ5Y.js +0 -328
  90. package/dist/chunk-S3UZOQ5Y.js.map +0 -1
  91. package/dist/chunk-S5TT5R3L.js +0 -2668
  92. package/dist/chunk-S5TT5R3L.js.map +0 -1
  93. package/dist/chunk-T4SQEITX.js +0 -95
  94. package/dist/chunk-T4SQEITX.js.map +0 -1
  95. package/dist/chunk-TT4KNT67.js +0 -124
  96. package/dist/chunk-TT4KNT67.js.map +0 -1
  97. package/dist/chunk-U5CHZ5M3.js +0 -357
  98. package/dist/chunk-U5CHZ5M3.js.map +0 -1
  99. package/dist/chunk-ULOKLHIQ.js +0 -1937
  100. package/dist/chunk-ULOKLHIQ.js.map +0 -1
  101. package/dist/chunk-VI2UW6B6.js +0 -162
  102. package/dist/chunk-VI2UW6B6.js.map +0 -1
  103. package/dist/chunk-VQMK5FMP.js +0 -247
  104. package/dist/chunk-VQMK5FMP.js.map +0 -1
  105. package/dist/chunk-VSMTAMNK.js +0 -53
  106. package/dist/chunk-VSMTAMNK.js.map +0 -1
  107. package/dist/chunk-VZSRQ272.js +0 -149
  108. package/dist/chunk-VZSRQ272.js.map +0 -1
  109. package/dist/chunk-WW2A73HW.js +0 -159
  110. package/dist/chunk-WW2A73HW.js.map +0 -1
  111. package/dist/chunk-X4UCIOTZ.js +0 -136
  112. package/dist/chunk-X4UCIOTZ.js.map +0 -1
  113. package/dist/chunk-XDIRG3TO.js +0 -1266
  114. package/dist/chunk-XDIRG3TO.js.map +0 -1
  115. package/dist/chunk-XJYR7XFV.js +0 -317
  116. package/dist/chunk-XJYR7XFV.js.map +0 -1
  117. package/dist/chunk-ZET2UAYW.js +0 -89
  118. package/dist/chunk-ZET2UAYW.js.map +0 -1
  119. package/dist/chunk-ZZUXHH3R.js +0 -99
  120. package/dist/chunk-ZZUXHH3R.js.map +0 -1
  121. package/dist/cli.d.ts +0 -1
  122. package/dist/cli.js +0 -112
  123. package/dist/cli.js.map +0 -1
  124. package/dist/contract/index.d.ts +0 -4969
  125. package/dist/contract/index.js +0 -1653
  126. package/dist/contract/index.js.map +0 -1
  127. package/dist/control.d.ts +0 -1013
  128. package/dist/control.js +0 -34
  129. package/dist/control.js.map +0 -1
  130. package/dist/fuzz.d.ts +0 -759
  131. package/dist/fuzz.js +0 -714
  132. package/dist/fuzz.js.map +0 -1
  133. package/dist/hosted/index.d.ts +0 -730
  134. package/dist/hosted/index.js +0 -14
  135. package/dist/hosted/index.js.map +0 -1
  136. package/dist/index.d.ts +0 -16780
  137. package/dist/index.js +0 -12168
  138. package/dist/index.js.map +0 -1
  139. package/dist/matrix/index.d.ts +0 -155
  140. package/dist/matrix/index.js +0 -8
  141. package/dist/matrix/index.js.map +0 -1
  142. package/dist/meta-eval/index.d.ts +0 -1030
  143. package/dist/meta-eval/index.js +0 -417
  144. package/dist/meta-eval/index.js.map +0 -1
  145. package/dist/multishot/index.d.ts +0 -579
  146. package/dist/multishot/index.js +0 -589
  147. package/dist/multishot/index.js.map +0 -1
  148. package/dist/openapi.json +0 -992
  149. package/dist/pipelines/index.d.ts +0 -567
  150. package/dist/pipelines/index.js +0 -515
  151. package/dist/pipelines/index.js.map +0 -1
  152. package/dist/reporting.d.ts +0 -1277
  153. package/dist/reporting.js +0 -48
  154. package/dist/reporting.js.map +0 -1
  155. package/dist/rl.d.ts +0 -4092
  156. package/dist/rl.js +0 -1724
  157. package/dist/rl.js.map +0 -1
  158. package/dist/run-campaign-HNFPJET4.js +0 -14
  159. package/dist/run-campaign-HNFPJET4.js.map +0 -1
  160. package/dist/storyboard/index.d.ts +0 -279
  161. package/dist/storyboard/index.js +0 -767
  162. package/dist/storyboard/index.js.map +0 -1
  163. package/dist/trace-attributes.d.ts +0 -52
  164. package/dist/trace-attributes.js +0 -62
  165. package/dist/trace-attributes.js.map +0 -1
  166. package/dist/traces.d.ts +0 -2343
  167. package/dist/traces.js +0 -249
  168. package/dist/traces.js.map +0 -1
  169. package/dist/wire/index.d.ts +0 -1252
  170. package/dist/wire/index.js +0 -81
  171. package/dist/wire/index.js.map +0 -1
@@ -1,3111 +0,0 @@
1
- import { TCloud } from '@tangle-network/tcloud';
2
- import { AxAIArgs, AxAIService, AxFunction } from '@ax-llm/ax';
3
- import { z } from 'zod';
4
-
5
- /**
6
- * Validator-output verdict — substrate primitive for "did this output pass,
7
- * and how well?"
8
- *
9
- * Used by:
10
- * - `@tangle-network/agent-eval/matrix` — verdict per cell in the cartesian.
11
- * - `@tangle-network/agent-runtime` — Validator<Output, Verdict = DefaultVerdict>.
12
- * Runtime keeps `Validator` because it's coupled to runtime-shaped
13
- * `ValidationCtx` (iteration, signal, traceEmitter); the verdict TYPE
14
- * itself is a substrate concept and lives here.
15
- *
16
- * Repo layering: agent-eval is the substrate (no upward deps). Both
17
- * agent-runtime and agent-knowledge consume this type FROM agent-eval —
18
- * never the other way around. See CLAUDE.md "Repo layering" for the rule.
19
- */
20
- /**
21
- * Minimal verdict shape — `valid` + `score` are required; `scores` +
22
- * `notes` are optional surface. Validators that need richer shapes
23
- * parameterise `Validator<Output, MyVerdict>` with their own type.
24
- *
25
- * Need structured extras? Extend DefaultVerdict with typed fields — never
26
- * serialize extras into `notes`.
27
- */
28
- interface DefaultVerdict {
29
- /** Whether the output meets the validator's pass criteria. */
30
- valid: boolean;
31
- /** Aggregate score in [0, 1]. Drivers use this for winner selection. */
32
- score: number;
33
- /** Per-dimension scores. Free-form; weighted into `score` by the validator. */
34
- scores?: Record<string, number>;
35
- /** Human-readable rationale; surfaces in trace + final-result `winner.verdict`. */
36
- notes?: string;
37
- }
38
-
39
- /**
40
- * Multi-layer verifier — ordered pipeline of verification layers.
41
- *
42
- * Different contract from {@link JudgeRunner} (which runs parallel
43
- * specs against a sandbox). MultiLayerVerifier is a DAG of layers
44
- * (install → typecheck → build → lint → serve → semantic → …) with
45
- * dependency-based skip, per-layer findings, soft-fail semantics, and
46
- * an aggregated `blendedScore` across all passed layers.
47
- *
48
- * Use when you want:
49
- * - ordered stages where a failing upstream stage skips downstream ones
50
- * - each stage produces rich `findings` (severity + message + evidence)
51
- * - a single composite score across stages with per-stage weights
52
- * - soft-fail stages whose failure doesn't abort the pipeline
53
- *
54
- * Use {@link JudgeRunner} when you want:
55
- * - N independent judges running in parallel against the same artifact
56
- * - no inter-judge dependencies
57
- * - boolean `passed` per judge + overall
58
- *
59
- * Both primitives compose — JudgeRunner can be invoked as a single
60
- * layer inside a MultiLayerVerifier if that suits the caller.
61
- */
62
-
63
- type LayerStatus = 'pass' | 'fail' | 'skipped' | 'error' | 'timeout';
64
- type Severity = 'critical' | 'major' | 'minor' | 'info';
65
- interface Finding {
66
- severity: Severity;
67
- message: string;
68
- evidence?: string;
69
- /** Optional layer name the finding belongs to (set by the verifier if omitted). */
70
- layer?: string;
71
- /**
72
- * Free-form structured payload — used by `multiToolchainLayer` to attach
73
- * `{ adapter: 'pnpm' }`, by judges to attach evidence pointers, etc.
74
- * Renderers MAY interrogate; agent-eval primitives never assume shape.
75
- */
76
- detail?: Record<string, unknown>;
77
- }
78
- interface LayerResult {
79
- layer: string;
80
- status: LayerStatus;
81
- /** 0..1 score, optional — layers that don't produce a numeric score omit. */
82
- score?: number;
83
- durationMs: number;
84
- findings: Finding[];
85
- /** Short human-readable summary (one line). */
86
- reason?: string;
87
- /**
88
- * Numeric layer-level diagnostics: error counts, warning counts,
89
- * cyclomatic complexity, total adapter wall-time, etc. Keyed by
90
- * diagnostic name; null = "diagnostic not applicable / not measured."
91
- * Renderers that know the keys can display them; ones that don't,
92
- * ignore. Free-form on purpose — consumers type the value shape in
93
- * their own namespace.
94
- */
95
- diagnostics?: Record<string, number | null>;
96
- /** Any rich per-layer detail — rendered as-is by consumers that know the layer. */
97
- detail?: Record<string, unknown>;
98
- }
99
- interface VerifyContext<Env = unknown> {
100
- /** Per-run opaque context the caller provides. Layers destructure what they need. */
101
- env: Env;
102
- /** Previously-computed results from layers that already ran. */
103
- prior: Record<string, LayerResult>;
104
- /** Signal — if aborted, layers MUST bail within reasonable wall. */
105
- signal: AbortSignal;
106
- }
107
- interface Layer<Env = unknown> {
108
- name: string;
109
- /** Stages that must have `status: 'pass'` before this layer runs. */
110
- dependsOn?: string[];
111
- /**
112
- * Weight in the composite `blendedScore`. Default 1.0. Layers with weight 0
113
- * contribute findings but not score.
114
- */
115
- weight?: number;
116
- /**
117
- * If true, a `fail` status contributes to `blendedScore` (as 0) instead of
118
- * being dropped — use for layers whose failure is a real signal. Default:
119
- * fail drops from numerator + denominator, matching VB's existing semantics.
120
- */
121
- failContributesToScore?: boolean;
122
- /** Optional per-layer wall-cap in ms. Honored by the verifier (AbortSignal). */
123
- capMs?: number;
124
- run: (ctx: VerifyContext<Env>) => Promise<LayerResult> | LayerResult;
125
- }
126
- interface VerifyOptions<Env = unknown> {
127
- env: Env;
128
- /**
129
- * Overall wall cap. Default: sum of layer capMs, or Infinity if any layer
130
- * omits a cap. The verifier short-circuits remaining layers on overall cap.
131
- */
132
- overallCapMs?: number;
133
- /** Called with each layer result as it completes. */
134
- onLayer?: (result: LayerResult) => void;
135
- }
136
- /** Extends the substrate verdict spine: `valid` = `allPass` and `score` =
137
- * `blendedScore` — derived where the report is aggregated, so spine
138
- * consumers (drivers, gates) read this report without an adapter. */
139
- interface VerificationReport extends DefaultVerdict {
140
- layers: LayerResult[];
141
- passCount: number;
142
- failCount: number;
143
- skippedCount: number;
144
- errorCount: number;
145
- /** True iff at least one scored layer ran AND every scored layer passed. */
146
- allPass: boolean;
147
- /**
148
- * Weighted mean of `score` across contributing layers. 0 when no layers
149
- * contributed. See {@link Layer.failContributesToScore} for fail semantics.
150
- */
151
- blendedScore: number;
152
- durationMs: number;
153
- startedAt: string;
154
- finishedAt: string;
155
- }
156
- /**
157
- * Ordered DAG of verification layers with dependency-based skipping, per-layer findings, soft-fail semantics, and a blended composite score across all passed layers.
158
- */
159
- declare class MultiLayerVerifier<Env = unknown> {
160
- private readonly layers;
161
- constructor(layers: Layer<Env>[]);
162
- run(opts: VerifyOptions<Env>): Promise<VerificationReport>;
163
- }
164
-
165
- interface RunScore {
166
- success: number;
167
- goalProgress: number;
168
- repoGroundedness: number;
169
- driftPenalty: number;
170
- toolUseQuality: number;
171
- patchQuality: number;
172
- testReality: number;
173
- finalGate: number;
174
- reviewerBlockers: number;
175
- costUsd: number;
176
- wallSeconds: number;
177
- notes?: string[];
178
- }
179
- interface RunScoreWeights {
180
- success: number;
181
- goalProgress: number;
182
- repoGroundedness: number;
183
- driftPenalty: number;
184
- toolUseQuality: number;
185
- patchQuality: number;
186
- testReality: number;
187
- finalGate: number;
188
- reviewerBlockers: number;
189
- costUsd: number;
190
- wallSeconds: number;
191
- }
192
-
193
- /**
194
- * Error taxonomy for `@tangle-network/agent-eval`.
195
- *
196
- * Every error this package throws as part of its *public contract* extends
197
- * `AgentEvalError`. Consumers can pattern-match by `instanceof <Subclass>` or
198
- * by the stable string `code` carried on the base class.
199
- *
200
- * The codes are stable across minor versions; new codes can be added, but
201
- * existing codes never change meaning. New subclasses are non-breaking.
202
- *
203
- * Internal invariant guards (`throw new Error('this should never happen')`)
204
- * remain plain `Error`s on purpose — they're programmer-mistake assertions,
205
- * not consumer-catchable contract failures.
206
- */
207
- type AgentEvalErrorCode = 'validation' | 'not_found' | 'config' | 'capture_integrity' | 'judge' | 'verification' | 'replay' | 'backend_integrity' | 'profile_matrix';
208
- /**
209
- * Base class for every contract error this package throws — carries the stable
210
- * string `code` taxonomy so consumers can `instanceof`-match or switch on `code`.
211
- */
212
- declare class AgentEvalError extends Error {
213
- /** Stable string code. Survives minification; safe to switch on. */
214
- readonly code: AgentEvalErrorCode;
215
- constructor(code: AgentEvalErrorCode, message: string, options?: {
216
- cause?: unknown;
217
- });
218
- }
219
- /** Caller passed invalid arguments (out of range, mutually-exclusive options, bad shape). */
220
- declare class ValidationError extends AgentEvalError {
221
- constructor(message: string, options?: {
222
- cause?: unknown;
223
- });
224
- }
225
-
226
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
227
- type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | {
228
- [key: string]: AgentProfileJson;
229
- };
230
- type AgentProfileDimensionValue = string | number | boolean | null;
231
- interface AgentProfileSource {
232
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
233
- kind: string;
234
- /** sha256 over the canonical source profile object. */
235
- hash: string;
236
- }
237
- interface AgentProfileHarness {
238
- id: string;
239
- version?: string;
240
- hash?: string;
241
- }
242
- interface AgentProfileCell {
243
- schemaVersion: AgentProfileCellSchemaVersion;
244
- cellId: string;
245
- profileId: string;
246
- sourceProfile: AgentProfileSource;
247
- harness?: AgentProfileHarness;
248
- model?: string;
249
- promptHash?: string;
250
- dimensions?: Record<string, AgentProfileDimensionValue>;
251
- }
252
-
253
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
254
- interface BudgetSpec {
255
- tokens?: number;
256
- wallMs?: number;
257
- calls?: number;
258
- usd?: number;
259
- }
260
- interface RunOutcome$1 {
261
- score?: number;
262
- pass?: boolean;
263
- failureClass?: FailureClass;
264
- notes?: string;
265
- }
266
- /**
267
- * Layer — optional classification in a nested build workflow.
268
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
269
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
270
- * `app-runtime`: a run of the generated agent against a domain scenario.
271
- * `meta`: any meta-eval (judge replay, correlation analysis).
272
- */
273
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
274
- interface Run {
275
- runId: string;
276
- /**
277
- * Stable identifier of the scenario being executed.
278
- *
279
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
280
- * input WITHOUT this field, substituting a sensible default
281
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
282
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
283
- * keeps the persisted shape unambiguous for downstream filters + aggregations
284
- * while removing the boilerplate of inventing placeholder ids at the call site.
285
- */
286
- scenarioId: string;
287
- variantId?: string;
288
- datasetVersion?: string;
289
- /** Git SHA of agent code at run time. */
290
- codeSha?: string;
291
- /** Hash of the prompt template + any system prompt. */
292
- promptSha?: string;
293
- /** Model id + date + system-prompt hash, concatenated. */
294
- modelFingerprint?: string;
295
- seed?: number;
296
- /** Arbitrary environment markers (shell, docker version, tz). */
297
- envFingerprint?: Record<string, string>;
298
- /** Version of the redaction rules applied to this run. */
299
- redactionVersion?: string;
300
- /** Parent run in a nested build workflow. A builder run's children are
301
- * app-build runs; those children are app-runtime runs. */
302
- parentRunId?: string;
303
- /** Stable project identifier — groups runs across chats + sessions. */
304
- projectId?: string;
305
- /** Chat/conversation identifier within a project. */
306
- chatId?: string;
307
- /** Layer classification — hint for aggregation; not enforced. */
308
- layer?: RunLayer;
309
- startedAt: number;
310
- endedAt?: number;
311
- status: RunStatus;
312
- outcome?: RunOutcome$1;
313
- budget?: BudgetSpec;
314
- /** Free-form labels for downstream grouping. */
315
- tags?: Record<string, string>;
316
- }
317
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
318
- type SpanStatus = 'ok' | 'error';
319
- interface SpanBase {
320
- spanId: string;
321
- parentSpanId?: string;
322
- runId: string;
323
- kind: SpanKind;
324
- name: string;
325
- startedAt: number;
326
- endedAt?: number;
327
- status?: SpanStatus;
328
- error?: string;
329
- /** Anything not covered by typed fields. Kept deliberately free-form. */
330
- attributes?: Record<string, unknown>;
331
- }
332
- interface Message {
333
- role: 'system' | 'user' | 'assistant' | 'tool';
334
- content: string;
335
- tokens?: number;
336
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
337
- images?: Array<{
338
- artifactId?: string;
339
- url?: string;
340
- mime?: string;
341
- }>;
342
- }
343
- interface LlmSpan extends SpanBase {
344
- kind: 'llm';
345
- model: string;
346
- messages: Message[];
347
- output?: string;
348
- inputTokens?: number;
349
- /** All generated tokens, including the reasoning subset when present. */
350
- outputTokens?: number;
351
- cachedTokens?: number;
352
- cacheWriteTokens?: number;
353
- /** Reasoning-token subset of `outputTokens`. */
354
- reasoningTokens?: number;
355
- costUsd?: number;
356
- finishReason?: string;
357
- }
358
- interface ToolSpan extends SpanBase {
359
- kind: 'tool';
360
- toolName: string;
361
- args: unknown;
362
- /** False when the source observed the call but did not capture its arguments. */
363
- argsCaptured?: boolean;
364
- result?: unknown;
365
- latencyMs?: number;
366
- }
367
- interface RetrievalSpan extends SpanBase {
368
- kind: 'retrieval';
369
- query: string;
370
- hits: Array<{
371
- docId: string;
372
- score: number;
373
- content?: string;
374
- }>;
375
- }
376
- interface JudgeSpan extends SpanBase {
377
- kind: 'judge';
378
- judgeId: string;
379
- /** Span this judgment applies to. */
380
- targetSpanId: string;
381
- dimension: string;
382
- /** Numeric score (free-range; interpretation up to the judge). */
383
- score: number;
384
- rationale?: string;
385
- evidence?: string;
386
- }
387
- interface SandboxSpan extends SpanBase {
388
- kind: 'sandbox';
389
- image?: string;
390
- command?: string;
391
- exitCode?: number;
392
- testsTotal?: number;
393
- testsPassed?: number;
394
- stdoutHash?: string;
395
- stderrHash?: string;
396
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
397
- wallMs?: number;
398
- }
399
- interface GenericSpan extends SpanBase {
400
- kind: 'agent' | 'custom';
401
- }
402
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
403
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
404
- interface TraceEvent {
405
- eventId: string;
406
- runId: string;
407
- spanId?: string;
408
- kind: EventKind;
409
- timestamp: number;
410
- payload: Record<string, unknown>;
411
- }
412
- interface BudgetLedgerEntry {
413
- runId: string;
414
- dimension: keyof BudgetSpec;
415
- limit: number;
416
- consumed: number;
417
- remaining: number;
418
- timestamp: number;
419
- breached: boolean;
420
- /** Span that triggered this entry, if any. */
421
- spanId?: string;
422
- }
423
- interface Artifact {
424
- artifactId: string;
425
- runId: string;
426
- spanId?: string;
427
- contentType: string;
428
- sizeBytes: number;
429
- /** sha256 in hex. */
430
- hash: string;
431
- /** External storage URL (R2, S3, filesystem path). */
432
- storageUrl?: string;
433
- /** Inline content for small blobs — keep under ~64KB. */
434
- inlineContent?: string;
435
- }
436
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
437
-
438
- /**
439
- * Paper-grade RunRecord schema + runtime validator.
440
- *
441
- * Every run that participates in a promotion gate, paper table, or
442
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
443
- * fields are exactly those the paper "Two Loops, Three Roles" requires
444
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
445
- * holdout split tag and either a `searchScore` or a `holdoutScore`.
446
- *
447
- * This is intentionally NOT a replacement for the rich `Run` /
448
- * `ProposeReviewReport` / `ScenarioResult` types already in the
449
- * package. Those are runtime structures with full provenance. A
450
- * `RunRecord` is the analysis-time projection — the JSON-friendly
451
- * row you'd put in a parquet file or paste into a notebook.
452
- *
453
- * Validate at the boundary:
454
- *
455
- * const rec = validateRunRecord(rawJson) // throws on missing
456
- * const ok = isRunRecord(rawJson) // boolean check
457
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
458
- *
459
- * The validator runs in pure TS — zod is intentionally NOT a
460
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
461
- */
462
-
463
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
464
- * combined train+test pool that the optimizer is allowed to read. */
465
- type RunSplitTag = 'search' | 'dev' | 'holdout';
466
- interface RunTokenUsage {
467
- input: number;
468
- /** All generated tokens charged as output, including reasoning tokens. */
469
- output: number;
470
- /** Reasoning-token subset of `output`, when the provider reports it. */
471
- reasoning?: number;
472
- /** Prompt tokens served from a provider cache. */
473
- cached?: number;
474
- /** Prompt tokens written into a provider cache. */
475
- cacheWrite?: number;
476
- }
477
- /**
478
- * How a run's USD amount was obtained.
479
- *
480
- * `costUsd` remains mandatory for wire compatibility. New producers should
481
- * always populate this discriminated union so a missing bill is never
482
- * mistaken for an observed zero-dollar run. For `uncaptured`, `costUsd` uses
483
- * the legacy `0` sentinel while this field carries the truthful null.
484
- */
485
- type RunCostProvenance = {
486
- kind: 'observed';
487
- usd: number;
488
- } | {
489
- kind: 'estimated';
490
- usd: number;
491
- } | {
492
- kind: 'uncaptured';
493
- usd: null;
494
- };
495
- interface RunJudgeMetadata {
496
- model: string;
497
- promptVersion: string;
498
- /** [0,1] confidence the judge declared. Constant judge confidence
499
- * across many runs is a fallback signal (see `canary.ts`). */
500
- confidence: number;
501
- /** True if the judge degraded to a fallback path (rules-only,
502
- * prior-call cache, etc.). The canary uses this to alert. */
503
- fallback: boolean;
504
- }
505
- /**
506
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
507
- * judges over a multi-dimensional rubric.
508
- *
509
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
510
- * composite the gate uses. The full breakdown belongs here so consumers
511
- * can answer "which judge disagreed?", "which dimension dragged the
512
- * composite down?", and "did half the panel fail?" without re-running.
513
- *
514
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
515
- * `composite` are convenience projections — derivable but precomputed so
516
- * downstream IRR primitives (`interRaterReliability`,
517
- * `corpusInterRaterAgreement`) and reporters don't pay the same
518
- * aggregation twice.
519
- *
520
- * Fail-loud discipline: judges that errored out land in `failedJudges`
521
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
522
- * run); the explicit list makes a partial-failure recorded as such.
523
- */
524
- interface JudgeScoresRecord {
525
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
526
- perJudge: Record<string, Record<string, number>>;
527
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
528
- perDimMean: Record<string, number>;
529
- /** Composite mean across all dims and judges. Mirrors the score
530
- * the gate sees on `outcome.searchScore` / `holdoutScore`. */
531
- composite: number;
532
- /** Judges that errored or returned an unparseable verdict. Recorded
533
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
534
- * not inferred from missing keys in `perJudge`. */
535
- failedJudges?: string[];
536
- /** Free-form notes the judges emitted (joined across judges or
537
- * first-judge only — consumer's choice). */
538
- notes?: string;
539
- }
540
- interface RunOutcome {
541
- /** Score on the search/optimization split. Optional because a
542
- * holdout-only evaluation only fills `holdoutScore`. */
543
- searchScore?: number;
544
- /** Score on the held-out split. Optional because a search-only run
545
- * only fills `searchScore`. At least one must be present. */
546
- holdoutScore?: number;
547
- /** Bag of any other metric the run produced — judge dimensions,
548
- * pass/fail counters, latency stats, etc. Numeric only — keeps
549
- * reporters honest. */
550
- raw: Record<string, number>;
551
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
552
- * judgements populate this; substrate primitives like
553
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
554
- * these records as input. Optional — single-judge or scalar-only
555
- * runs leave it unset. */
556
- judgeScores?: JudgeScoresRecord;
557
- /** Authenticity / realness verdict — did the run build the REAL thing on the
558
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
559
- * with an authenticity config populate it. Carried in the corpus so the
560
- * flywheel / off-policy learning can optimize for real completion, not gamed
561
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
562
- * must not count as a real success regardless of `score`. */
563
- realness?: {
564
- score: number;
565
- gated: boolean;
566
- reason?: string;
567
- };
568
- }
569
- /**
570
- * Mandatory paper-grade fields for a single evaluation run. Optional
571
- * fields are extension points; mandatory fields throw if missing.
572
- *
573
- * Hash discipline:
574
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
575
- * model (after any steering bundle merge).
576
- * - `configHash` is the sha256 of the effective run config (model,
577
- * temperature, tools, judges, splits). The pair (promptHash,
578
- * configHash) uniquely identifies an experiment cell.
579
- *
580
- * Model snapshot discipline:
581
- * - `model` MUST encode a snapshot version. Bare aliases like
582
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
583
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
584
- */
585
- interface RunRecord {
586
- /** UUID for the run. */
587
- runId: string;
588
- /** Logical experiment grouping (a treatment vs a baseline within
589
- * the same sweep should share `experimentId`). */
590
- experimentId: string;
591
- /** Stable identifier for the candidate (variant) being run. The
592
- * promotion gate compares two `candidateId`s on matched items. */
593
- candidateId: string;
594
- /** RNG seed for the run. Always recorded — silent re-seeding is
595
- * the most common cause of non-reproducible numbers. */
596
- seed: number;
597
- /** Model identifier WITH snapshot version. */
598
- model: string;
599
- /** sha256 of the effective prompt (post-steering). */
600
- promptHash: string;
601
- /** sha256 of the effective config. */
602
- configHash: string;
603
- /** Git SHA the harness was run from. */
604
- commitSha: string;
605
- /** End-to-end wall-clock duration in milliseconds. */
606
- wallMs: number;
607
- /** Time spent queued before execution started, if known. */
608
- queueMs?: number;
609
- /** Total USD cost. Mandatory — runs without a cost number are
610
- * unbounded by definition and must not be admitted into the gate.
611
- * `0` is retained as the compatibility sentinel for an uncaptured amount;
612
- * inspect `costProvenance` before treating it as observed. */
613
- costUsd: number;
614
- /** Observed, model-priced estimate, or genuinely uncaptured USD amount.
615
- * Optional only so existing serialized RunRecords remain valid. */
616
- costProvenance?: RunCostProvenance;
617
- /** Token usage breakdown. */
618
- tokenUsage: RunTokenUsage;
619
- /** Judge-side metadata, if a judge was used. */
620
- judgeMetadata?: RunJudgeMetadata;
621
- /** Per-split scores + raw bag. */
622
- outcome: RunOutcome;
623
- /** Canonical, cross-agent failure class drawn from the shared
624
- * `FAILURE_CLASSES` taxonomy. This is the aggregation key that makes
625
- * "which failure dominates across the whole fleet" answerable in ONE
626
- * vocabulary — every agent classifies against the same enum. Producers
627
- * set it via the substrate classifier; leave unset only when the failure
628
- * genuinely can't be classified. */
629
- failureClass?: FailureClass;
630
- /** Free-form domain-specific failure detail, scoped UNDER `failureClass`
631
- * (e.g. failureClass='tool_recovery_failure', failureMode='forge_build_unsatisfied').
632
- * The within-agent drill-down; `failureClass` is the cross-agent key. */
633
- failureMode?: string;
634
- /** Which split this run was drawn from. */
635
- splitTag: RunSplitTag;
636
- /**
637
- * Stable scenario identifier the run was scored against. Optional for
638
- * backwards compatibility, but **strongly recommended**: every primitive
639
- * that pairs runs by scenario (preferences, paired stats, BT tournament)
640
- * keys on this. The campaign artifact populates it canonically; legacy
641
- * runs without it fall back to inference from `outcome.raw.scenario_id`
642
- * or `experimentId`.
643
- */
644
- scenarioId?: string;
645
- /**
646
- * Canonical identity for the agent profile cell that produced this row:
647
- * profile artifact hash plus optional harness/model/prompt/reporting
648
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
649
- * longitudinal reports by the complete source profile, not by a loose
650
- * candidate label or opaque config hash.
651
- */
652
- agentProfile?: AgentProfileCell;
653
- }
654
-
655
- /**
656
- * RawProviderSink — first-class persistence for the actual HTTP-level
657
- * request/response bodies of every LLM provider call.
658
- *
659
- * Why this is a separate sink from the structured `LlmSpan`:
660
- *
661
- * - `LlmSpan` records the *intent* — model name, messages, output text,
662
- * usage. It's what dashboards read; it's NOT enough for forensics.
663
- * - When a downstream consumer reports "the verifier used the wrong route"
664
- * or "tokens look right but reasoning was missing," the only way to
665
- * answer is the raw HTTP body. Span fields can lie (a proxy can echo
666
- * a different `model` value than what actually answered); the raw
667
- * response is ground truth.
668
- *
669
- * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
670
- * matrix runner / BuilderSession sets it up automatically) and every
671
- * request, response, and error is recorded — including retries, with the
672
- * attempt index attached so a flaky call's full event chain is recoverable.
673
- *
674
- * Redaction is enforced at sink time. The default redactor strips
675
- * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
676
- * payload field whose key matches `apiKey | api_key | bearer | password |
677
- * secret | token` (case-insensitive). Override via the sink constructor or
678
- * the per-call `redactor`. The `redactedFields` array on the persisted
679
- * event lets a reviewer see what was stripped without exposing the values.
680
- */
681
- type RawProviderDirection = 'request' | 'response' | 'error';
682
- interface RawProviderEvent {
683
- /** Stable id. Generated by the sink if omitted. */
684
- eventId: string;
685
- /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
686
- runId?: string;
687
- spanId?: string;
688
- /**
689
- * Logical provider name. Free-form so callers can use whatever id matches
690
- * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
691
- * omitted, derived from `baseUrl` in `LlmClientOptions`.
692
- */
693
- provider: string;
694
- model: string;
695
- /** Endpoint path, e.g. `'/v1/chat/completions'`. */
696
- endpoint: string;
697
- /** Base URL used for the call (already-normalised — no trailing slash). */
698
- baseUrl: string;
699
- /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
700
- attemptIndex: number;
701
- direction: RawProviderDirection;
702
- /** Unix ms. */
703
- timestamp: number;
704
- /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
705
- durationMs?: number;
706
- statusCode?: number;
707
- requestHeaders?: Record<string, string>;
708
- requestBody?: unknown;
709
- responseHeaders?: Record<string, string>;
710
- responseBody?: unknown;
711
- /** Set on `direction: 'error'` events. */
712
- errorMessage?: string;
713
- /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
714
- redactedFields: string[];
715
- }
716
- interface RawProviderSinkFilter {
717
- runId?: string;
718
- spanId?: string;
719
- direction?: RawProviderDirection;
720
- attemptIndex?: number;
721
- }
722
- interface RawProviderSink {
723
- record(event: RawProviderEvent): Promise<void>;
724
- /** Optional listing — implementations that durably persist (file, db) should support this. */
725
- list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
726
- /** Optional teardown for backed implementations. */
727
- close?(): Promise<void>;
728
- }
729
- type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
730
-
731
- interface RunFilter {
732
- scenarioId?: string;
733
- variantId?: string;
734
- status?: RunStatus;
735
- since?: number;
736
- until?: number;
737
- tag?: {
738
- key: string;
739
- value: string;
740
- };
741
- parentRunId?: string;
742
- projectId?: string;
743
- chatId?: string;
744
- layer?: RunLayer;
745
- }
746
- interface SpanFilter {
747
- runId?: string;
748
- parentSpanId?: string;
749
- kind?: SpanKind;
750
- name?: string;
751
- toolName?: string;
752
- judgeId?: string;
753
- since?: number;
754
- until?: number;
755
- }
756
- interface EventFilter {
757
- runId?: string;
758
- spanId?: string;
759
- kind?: EventKind;
760
- since?: number;
761
- until?: number;
762
- }
763
- interface TraceStore {
764
- appendRun(run: Run): Promise<void>;
765
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
766
- appendSpan(span: Span): Promise<void>;
767
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
768
- appendEvent(event: TraceEvent): Promise<void>;
769
- appendArtifact(artifact: Artifact): Promise<void>;
770
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
771
- getRun(runId: string): Promise<Run | undefined>;
772
- listRuns(filter?: RunFilter): Promise<Run[]>;
773
- spans(filter?: SpanFilter): Promise<Span[]>;
774
- events(filter?: EventFilter): Promise<TraceEvent[]>;
775
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
776
- artifacts(runId: string): Promise<Artifact[]>;
777
- }
778
-
779
- interface RunTrace {
780
- run: Run;
781
- spans: Span[];
782
- events: TraceEvent[];
783
- artifacts: Artifact[];
784
- budget: BudgetLedgerEntry[];
785
- }
786
- interface RunCriticOptions {
787
- weights?: Partial<RunScoreWeights>;
788
- driftPatterns?: RegExp[];
789
- }
790
- declare class RunCritic {
791
- private readonly weights?;
792
- private readonly driftPatterns;
793
- constructor(options?: RunCriticOptions);
794
- score(store: TraceStore, runId: string): Promise<RunScore>;
795
- scoreTrace(trace: RunTrace): RunScore;
796
- rank(score: RunScore): number;
797
- private isDrift;
798
- }
799
-
800
- type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
801
- interface CostUsage {
802
- inputTokens: number;
803
- /** Includes reasoning tokens when the provider bills them as output. */
804
- outputTokens: number;
805
- /** Reasoning-token subset of outputTokens, when reported. */
806
- reasoningTokens?: number;
807
- /** Prompt tokens served from a provider cache. */
808
- cachedTokens?: number;
809
- /** Prompt tokens written into a provider cache. */
810
- cacheWriteTokens?: number;
811
- }
812
- interface CostCallBase {
813
- callId: string;
814
- channel: CostChannel;
815
- phase: string;
816
- actor: string;
817
- model: string;
818
- maximumCostUsd?: number;
819
- tags?: Record<string, string>;
820
- timestamp: number;
821
- }
822
- interface CostReceipt extends CostCallBase, CostUsage {
823
- status: 'settled';
824
- costUsd: number;
825
- costUnknown: boolean;
826
- usageUnknown?: boolean;
827
- pricing?: {
828
- inputUsdPerThousand: number;
829
- outputUsdPerThousand: number;
830
- };
831
- actualCostUsd?: number;
832
- error?: string;
833
- }
834
- interface CostReceiptInput extends CostUsage {
835
- model: string;
836
- actualCostUsd?: number;
837
- costUnknown?: boolean;
838
- usageUnknown?: boolean;
839
- }
840
- type MaximumCharge = {
841
- externallyEnforcedMaximumUsd: number;
842
- } | ({
843
- model: string;
844
- } & CostUsage);
845
- interface RunPaidCallInput<T> {
846
- callId?: string;
847
- channel: CostChannel;
848
- phase: string;
849
- actor: string;
850
- /** Used before a provider receipt exists and on failures without one. */
851
- model?: string;
852
- tags?: Record<string, string>;
853
- signal?: AbortSignal;
854
- /** Provider-enforced dollar maximum, or maximum priced token usage. Required when capped. */
855
- maximumCharge?: MaximumCharge;
856
- /** `callId` can be forwarded as the provider's idempotency key. */
857
- execute(signal: AbortSignal, callId: string): Promise<T>;
858
- receipt(value: T): CostReceiptInput;
859
- receiptFromError?(error: Error): CostReceiptInput | undefined;
860
- }
861
- type PaidCallResult<T> = {
862
- succeeded: true;
863
- callId: string;
864
- value: T;
865
- receipt: CostReceipt;
866
- } | {
867
- succeeded: false;
868
- callId?: string;
869
- error: Error;
870
- receipt?: CostReceipt;
871
- };
872
- interface ChannelRollup {
873
- channel: CostChannel;
874
- calls: number;
875
- inputTokens: number;
876
- outputTokens: number;
877
- reasoningTokens?: number;
878
- cachedTokens: number;
879
- cacheWriteTokens?: number;
880
- costUsd: number;
881
- unpricedCalls: number;
882
- unknownUsageCalls: number;
883
- }
884
- interface CostLedgerSummary {
885
- totalCalls: number;
886
- pendingCalls: number;
887
- unresolvedCalls: number;
888
- reservedCostUsd: number;
889
- inputTokens: number;
890
- outputTokens: number;
891
- reasoningTokens?: number;
892
- cachedTokens: number;
893
- cacheWriteTokens?: number;
894
- totalCostUsd: number;
895
- byChannel: ChannelRollup[];
896
- unpricedModels: string[];
897
- fullyPriced: boolean;
898
- usageComplete: boolean;
899
- accountingComplete: boolean;
900
- incompleteReasons: string[];
901
- }
902
- interface CostLedgerFilter {
903
- channel?: CostChannel;
904
- phase?: string;
905
- tags?: Record<string, string>;
906
- }
907
- interface CostLedgerWaitOptions {
908
- /** Maximum time to wait for active provider calls. Default 5 seconds. */
909
- timeoutMs?: number;
910
- }
911
- /** Append-only storage. `append` must atomically reject stale revisions. */
912
- interface CostLedgerPersistence {
913
- read(): {
914
- revision: string;
915
- events: string;
916
- };
917
- append(expectedRevision: string, event: string): string | undefined;
918
- }
919
- interface CostLedgerOptions {
920
- costCeilingUsd?: number;
921
- persistence?: CostLedgerPersistence;
922
- /** Import already-settled receipts without admitting new paid work. */
923
- receipts?: readonly CostReceipt[];
924
- }
925
- /** Run-wide paid-call admission, durable call state, receipts, and summaries. */
926
- declare class CostLedger {
927
- private readonly records;
928
- private readonly activeCallIds;
929
- private readonly lateCallIds;
930
- private readonly idleWaiters;
931
- private completedTasks;
932
- private revision;
933
- private costLimitPersisted;
934
- readonly costCeilingUsd?: number;
935
- private readonly persistence?;
936
- constructor(input?: number | CostLedgerOptions);
937
- runPaidCall<T>(input: RunPaidCallInput<T>): Promise<PaidCallResult<T>>;
938
- /** Wait until every call started by this ledger has produced a durable outcome. */
939
- waitForIdle(options?: CostLedgerWaitOptions): Promise<boolean>;
940
- /** Settle a call left pending by a crashed process after reconciling with the provider. */
941
- reconcile(callId: string, observed: CostReceiptInput, options?: {
942
- error?: string;
943
- }): CostReceipt;
944
- list(filter?: CostLedgerFilter): CostReceipt[];
945
- summary(filter?: CostLedgerFilter): CostLedgerSummary;
946
- markCompleted(count?: number): void;
947
- costPerCompletedTask(): number | null;
948
- private execute;
949
- private captureLateOutcome;
950
- private releaseActiveCall;
951
- private commitOutcome;
952
- private captureFailure;
953
- private commitReceipt;
954
- private resolveMaximum;
955
- private hasIncompleteSettledCall;
956
- private appendRecord;
957
- private ensureCostLimitPersisted;
958
- private appendEvent;
959
- }
960
- /** Public callback surface for a shared cost ledger.
961
- *
962
- * Declaration bundles may expose this type through multiple package subpaths.
963
- * Keeping callback contracts structural lets those subpaths compose while the
964
- * concrete {@link CostLedger} retains its private durable state.
965
- */
966
- type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'waitForIdle'>> & Partial<Pick<CostLedger, 'waitForIdle'>>;
967
-
968
- /**
969
- * LLM client with graceful degrade.
970
- *
971
- * OpenAI-compatible `/v1/chat/completions` client with:
972
- * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
973
- * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
974
- * - Graceful json_schema → json_object degrade on 400 with schema-reject body.
975
- * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
976
- * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
977
- * directly, cli-bridge subscriptions, and any router that speaks the spec.
978
- *
979
- * Usage:
980
- * const { value, result } = await callLlmJson<MyType>(
981
- * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
982
- * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
983
- * )
984
- *
985
- * This is THE llm-calling seam for agent-eval primitives that need structured
986
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
987
- * that need free-form text use `callLlm` and parse output themselves.
988
- */
989
-
990
- interface LlmMessage {
991
- role: 'system' | 'user' | 'assistant';
992
- /**
993
- * Either a plain text content string OR a multimodal content array
994
- * (text + image_url parts) for vision-capable models.
995
- */
996
- content: string | Array<{
997
- type: 'text';
998
- text: string;
999
- } | {
1000
- type: 'image_url';
1001
- image_url: {
1002
- url: string;
1003
- detail?: 'auto' | 'low' | 'high';
1004
- };
1005
- }>;
1006
- }
1007
- interface LlmCallRequest {
1008
- model: string;
1009
- messages: LlmMessage[];
1010
- /** Optional JSON-mode response format (response_format: json_object). */
1011
- jsonMode?: boolean;
1012
- /** Optional structured output via JSON Schema. Falls back to json_object on 400. */
1013
- jsonSchema?: {
1014
- name: string;
1015
- schema: Record<string, unknown>;
1016
- };
1017
- temperature?: number;
1018
- maxTokens?: number;
1019
- /** Per-call timeout, default 300s. */
1020
- timeoutMs?: number;
1021
- }
1022
- interface LlmUsage {
1023
- promptTokens: number;
1024
- completionTokens: number;
1025
- totalTokens: number;
1026
- /** False when the provider omitted or malformed prompt/completion usage. */
1027
- captured?: boolean;
1028
- /** Proxies populate this when prompt caching is on. */
1029
- cachedPromptTokens?: number;
1030
- }
1031
- interface LlmCallResult {
1032
- /** The text content of the first choice. Empty string if none. */
1033
- content: string;
1034
- usage: LlmUsage;
1035
- /**
1036
- * Cost in USD. Pulled from proxy's `_response_cost` field when present;
1037
- * `null` when neither the proxy nor the caller can derive it.
1038
- */
1039
- costUsd: number | null;
1040
- /** Model name actually used (echoed from response). */
1041
- model: string;
1042
- /** Wall-clock duration of the HTTP call (last attempt, if retried). */
1043
- durationMs: number;
1044
- /**
1045
- * `finish_reason` echoed from the first choice (`stop`, `length`,
1046
- * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
1047
- * Exposed so a free-form `callLlm` caller CAN detect a truncated answer
1048
- * (`length`) instead of treating a cut-off completion as complete. Note:
1049
- * `callLlm` does not itself reject on it — acting on this signal is the
1050
- * caller's responsibility (in-repo free-form drivers do not yet enforce it).
1051
- */
1052
- finishReason?: string | null;
1053
- /**
1054
- * True when `content.trim()` is empty. An empty completion is a silent zero
1055
- * for free-form `callLlm` callers; this flag is the signal a caller can
1056
- * inspect to fail loud rather than proceed on an empty string. `callLlm`
1057
- * surfaces it but does not throw on it.
1058
- */
1059
- contentEmpty?: boolean;
1060
- /** Raw response body. */
1061
- raw: Record<string, unknown>;
1062
- }
1063
- interface LlmClientOptions {
1064
- /** Base URL (without trailing slash). Must end at the `/v1` prefix. */
1065
- baseUrl?: string;
1066
- /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
1067
- apiKey?: string;
1068
- bearer?: string;
1069
- /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
1070
- authHeader?: {
1071
- name: string;
1072
- value: string;
1073
- };
1074
- /** Stable provider idempotency key, reused across retries of this logical call. */
1075
- idempotencyKey?: string;
1076
- /** Default timeout in ms. Per-call can override. */
1077
- defaultTimeoutMs?: number;
1078
- /**
1079
- * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
1080
- * each attempt's per-attempt timeout controller, so aborting it cancels
1081
- * the in-flight fetch. A caller abort is FATAL: it is not retried even
1082
- * though an AbortError otherwise matches the transient patterns.
1083
- */
1084
- signal?: AbortSignal;
1085
- /**
1086
- * Cross-attempt wall-clock budget in ms, measured from the first attempt.
1087
- * Before launching each attempt the loop checks the remaining budget and
1088
- * stops retrying once it is exhausted, rather than waiting the full
1089
- * per-attempt timeout on every retry. Bounds total time independent of
1090
- * total attempts × `timeoutMs`.
1091
- */
1092
- deadlineMs?: number;
1093
- /** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
1094
- maxRetries?: number;
1095
- /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
1096
- fetch?: typeof fetch;
1097
- /**
1098
- * Optional raw HTTP capture sink. When provided, every request, response,
1099
- * and error (across all retry attempts) is recorded to the sink, with auth
1100
- * headers and credential-shaped body fields redacted by default. This is
1101
- * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
1102
- * raw events record what actually crossed the wire.
1103
- */
1104
- rawSink?: RawProviderSink;
1105
- /**
1106
- * Logical provider id attached to raw events. When omitted, derived from
1107
- * `baseUrl` via `providerFromBaseUrl`.
1108
- */
1109
- provider?: string;
1110
- /** Trace context attached to raw events; populated by emitter-aware callers. */
1111
- traceContext?: {
1112
- runId?: string;
1113
- spanId?: string;
1114
- };
1115
- /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
1116
- redactor?: ProviderRedactor;
1117
- }
1118
-
1119
- /**
1120
- * Semantic concept judge — "does the built artifact actually implement
1121
- * the features the user asked for?"
1122
- *
1123
- * Distinct from the domain/code/coherence judges in `judges.ts`:
1124
- * - those judges score free-form conversational agent outputs along
1125
- * quality dimensions (accuracy, depth, etc.)
1126
- * - this judge scores a *built artifact* (served HTML + source files)
1127
- * against an explicit list of expected concepts, returning per-concept
1128
- * {present, score 0-10, evidence, severity}.
1129
- *
1130
- * The judge is strict about distinguishing (a) a working implementation
1131
- * from (b) a keyword-present stub. "// TODO: mint button" is NOT present.
1132
- * Only real, functional, wired-up code counts.
1133
- *
1134
- * Use via {@link createSemanticConceptJudge} or directly via
1135
- * {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM
1136
- * or JSON-parse errors so the caller can treat that as "layer skipped"
1137
- * rather than "layer failed" in a multi-layer pipeline.
1138
- */
1139
-
1140
- /**
1141
- * Implementation complexity class for weighted scoring.
1142
- *
1143
- * - `render` (default): the concept is a UI surface that displays static
1144
- * data — render a list, show a counter, lay out a button. Single-file
1145
- * work, no external integration.
1146
- * - `integrate`: the concept requires wiring a real external system —
1147
- * wallet connect (wagmi + RainbowKit + chain config), payment provider
1148
- * (Stripe Elements + intent + webhook), an API client with auth.
1149
- * Multi-file, library-knowledge, runtime correctness matters.
1150
- * - `compute`: the concept requires algorithmic work — solver, simulator,
1151
- * constraint propagation, ML inference. Correctness > UI polish.
1152
- *
1153
- * Default weights (when applied via `weightConcepts: 'complexity'`):
1154
- * render=1.0, integrate=2.0, compute=2.5
1155
- *
1156
- * Cross-vertical scoring without complexity weighting silently inflates
1157
- * the rate of UI-heavy verticals (healthcare, fintech dashboards) vs
1158
- * integration-heavy verticals (DeFi, wallets) — all concepts treated
1159
- * equally even though the agent does 2-3x the work for `integrate`.
1160
- */
1161
- type ConceptComplexity = 'render' | 'integrate' | 'compute';
1162
- interface ConceptSpec {
1163
- name: string;
1164
- /** Short hints that help the judge; not used for matching. */
1165
- keywords?: string[];
1166
- /** Optional explicit weight; default 1.0. Overrides complexity-derived weight. */
1167
- weight?: number;
1168
- /** Implementation complexity class. Default `render`. */
1169
- complexity?: ConceptComplexity;
1170
- }
1171
- interface SemanticConceptJudgeInput {
1172
- /** Full natural-language prompt the agent was handed. */
1173
- userRequest: string;
1174
- /** Rendered HTML the preview returns (UI artifacts). Optional. */
1175
- servedHtml?: string;
1176
- /** Top-level source files from the agent's workdir. */
1177
- sourceFiles: Array<{
1178
- path: string;
1179
- content: string;
1180
- }>;
1181
- /** The expected concept list. */
1182
- expectedConcepts: ConceptSpec[];
1183
- /** Free-form metadata (id, difficulty) to inject into the prompt. */
1184
- artifactLabel?: string;
1185
- artifactDescription?: string;
1186
- }
1187
- /**
1188
- * Score-aggregation strategy. `mean` averages 0-10 scores uniformly.
1189
- * `complexity` applies the default weight table (render=1, integrate=2,
1190
- * compute=2.5) unless a concept has an explicit `weight`. `explicit`
1191
- * honors only `weight` (defaulting to 1 for unspecified).
1192
- */
1193
- type ConceptWeightStrategy = 'mean' | 'complexity' | 'explicit';
1194
- interface SemanticConceptJudgeOptions {
1195
- /** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */
1196
- model?: string;
1197
- /** Per-call timeout. Default 300s. */
1198
- timeoutMs?: number;
1199
- /** Provider-enforced output limit. Default 16000. */
1200
- maxTokens?: number;
1201
- /** Pipeline budget for the prompt (source blob truncation). Default 45000. */
1202
- maxSourceChars?: number;
1203
- /** Per-file cap before inclusion. Default 20000. */
1204
- maxPerFileChars?: number;
1205
- /** HTML cap. Default 30000. */
1206
- maxHtmlChars?: number;
1207
- /** LlmClient config (baseUrl, apiKey, authHeader, …). */
1208
- llm?: LlmClientOptions;
1209
- costLedger?: CostLedgerHandle;
1210
- costPhase?: string;
1211
- signal?: AbortSignal;
1212
- /**
1213
- * Score aggregation strategy. Default `mean` — uniform average across
1214
- * concepts. Cross-vertical comparisons should use `complexity` to
1215
- * neutralize the integrate-vs-render asymmetry.
1216
- */
1217
- weightConcepts?: ConceptWeightStrategy;
1218
- /** Override the default complexity → weight table. */
1219
- complexityWeights?: Partial<Record<ConceptComplexity, number>>;
1220
- }
1221
-
1222
- interface Scenario {
1223
- id: string;
1224
- persona: string;
1225
- label: string;
1226
- thesis: string;
1227
- dimensions: string[];
1228
- turns: Turn[];
1229
- artifactChecks: ArtifactCheck[];
1230
- systemPromptAppend?: string;
1231
- }
1232
- interface Turn {
1233
- user: string;
1234
- expectedBehaviors: string[];
1235
- adversarial?: boolean;
1236
- feedbackType?: 'correction' | 'rejection' | 'vague' | 'contradictory' | 'escalation';
1237
- }
1238
- interface ArtifactCheck {
1239
- type: 'vault_file_exists' | 'vault_file_contains' | 'block_extracted' | 'code_valid' | 'generation_produced' | 'tool_created' | string;
1240
- target: string;
1241
- contains?: string;
1242
- minCount?: number;
1243
- description: string;
1244
- }
1245
- interface TurnResult {
1246
- turnIndex: number;
1247
- userMessage: string;
1248
- agentResponse: string;
1249
- durationMs: number;
1250
- blocksExtracted: {
1251
- type: string;
1252
- title: string;
1253
- }[];
1254
- containsCode: boolean;
1255
- containsToolCall: boolean;
1256
- }
1257
- interface JudgeScore {
1258
- judgeName: string;
1259
- dimension: string;
1260
- score: number;
1261
- reasoning: string;
1262
- evidence?: string;
1263
- }
1264
- interface CollectedArtifacts {
1265
- vaultFiles: {
1266
- path: string;
1267
- content: string;
1268
- }[];
1269
- blocksExtracted: {
1270
- type: string;
1271
- fields: Record<string, string>;
1272
- }[];
1273
- codeBlocks: {
1274
- language: string;
1275
- code: string;
1276
- }[];
1277
- toolCalls: string[];
1278
- }
1279
- interface JudgeInput {
1280
- scenario: Scenario;
1281
- turns: TurnResult[];
1282
- artifacts: CollectedArtifacts;
1283
- /** Shared ledger for paid built-in judges. Direct calls default to an uncapped ledger. */
1284
- costLedger?: CostLedgerHandle;
1285
- costPhase?: string;
1286
- costTags?: Record<string, string>;
1287
- signal?: AbortSignal;
1288
- /** Exact maximum provider attempts configured on the supplied TCloud client. */
1289
- tcloudMaximumAttempts?: number;
1290
- }
1291
- type JudgeFn = (tc: TCloud, input: JudgeInput) => Promise<JudgeScore[]>;
1292
-
1293
- /**
1294
- * Shared types for the trace-analyst module.
1295
- *
1296
- * Wire format. The store interface speaks `OtlpSpanLike` rows — one JSONL
1297
- * line per span, OTLP-shaped. We do NOT depend on a specific tracing
1298
- * vendor at the type level. Adapter
1299
- * layers map upstream shapes onto this interface.
1300
- *
1301
- * Design constraint. Every read operation that can return arbitrary
1302
- * payload must carry a byte budget so the agent's tool result stays
1303
- * bounded regardless of input trace size. Oversized responses
1304
- * substitute a deterministic summary instead of bytes — see
1305
- * `ViewTraceOversized`.
1306
- */
1307
- /** OTLP span kind (subset we actually use). */
1308
- type TraceAnalystSpanKind = 'AGENT' | 'LLM' | 'TOOL' | 'CHAIN' | 'GUARDRAIL' | 'SPAN' | 'UNKNOWN';
1309
- type TraceAnalystSpanStatus = 'OK' | 'ERROR' | 'UNSET';
1310
- /** Subset of OTLP span fields the analyst exposes to the agent. The
1311
- * store's job is to project upstream's full span shape down to this
1312
- * view — the analyst never sees vendor extensions directly. */
1313
- interface TraceAnalystSpan {
1314
- trace_id: string;
1315
- span_id: string;
1316
- parent_span_id: string | null;
1317
- name: string;
1318
- kind: TraceAnalystSpanKind;
1319
- start_time: string;
1320
- end_time: string;
1321
- duration_ms: number;
1322
- status: TraceAnalystSpanStatus;
1323
- status_message?: string;
1324
- service_name: string | null;
1325
- agent_name: string | null;
1326
- model_name: string | null;
1327
- tool_name: string | null;
1328
- /** Raw JSON-serialisable attribute map. May contain large strings;
1329
- * callers must respect the per-attribute byte cap. */
1330
- attributes: Record<string, unknown>;
1331
- }
1332
- interface TraceAnalystTraceSummary {
1333
- trace_id: string;
1334
- service_name: string | null;
1335
- agent_name: string | null;
1336
- span_count: number;
1337
- has_errors: boolean;
1338
- start_time: string;
1339
- end_time: string;
1340
- duration_ms: number;
1341
- raw_jsonl_bytes: number;
1342
- models: string[];
1343
- tools: string[];
1344
- }
1345
- interface TraceAnalystFilters {
1346
- /** Restrict to traces that contain at least one error span. */
1347
- has_errors?: boolean;
1348
- /** Match if any span's `service.name` is in this list. */
1349
- service_names?: string[];
1350
- /** Match if any span's `agent.name` is in this list. */
1351
- agent_names?: string[];
1352
- /** Match if any LLM span's `llm.model_name` is in this list. */
1353
- model_names?: string[];
1354
- /** Match if any tool span's `tool.name` is in this list. */
1355
- tool_names?: string[];
1356
- /** ISO-8601 lower bound on the trace's earliest start time. */
1357
- start_time_after?: string;
1358
- /** ISO-8601 upper bound on the trace's earliest start time. */
1359
- start_time_before?: string;
1360
- /** Single regex applied to raw JSONL bytes for the trace. Opt-in;
1361
- * expensive on large datasets. Use the indexed filters above first. */
1362
- regex_pattern?: string;
1363
- }
1364
- /** One distinct error signature across the dataset — the deterministic unit of
1365
- * failure coverage. Signatures normalize volatile tokens (digits, hex/uuids,
1366
- * paths, durations) out of the span `status_message` so semantically identical
1367
- * failures collapse into one cluster. An analyst that accounts for every
1368
- * cluster has, by construction, covered every distinct failure mode. */
1369
- interface ErrorCluster {
1370
- /** Normalized status_message — the cluster key. */
1371
- signature: string;
1372
- /** A verbatim, un-normalized exemplar message (for exact-string citation). */
1373
- status_message_sample: string;
1374
- /** The span name that most often carries this signature, if any. */
1375
- span_name: string | null;
1376
- /** The tool that most often carries this signature, if any. */
1377
- tool_name: string | null;
1378
- trace_count: number;
1379
- span_count: number;
1380
- /** trace_count / total error traces in the matched set (0..1). */
1381
- prevalence: number;
1382
- /** Real trace ids carrying this signature (capped), passable to view/search. */
1383
- exemplar_trace_ids: string[];
1384
- /** Real span ids carrying this signature (capped). */
1385
- exemplar_span_ids: string[];
1386
- }
1387
- interface DatasetOverview {
1388
- total_traces: number;
1389
- raw_jsonl_bytes: number;
1390
- services: string[];
1391
- agents: string[];
1392
- models: string[];
1393
- tool_names: string[];
1394
- /** Up to 20 real trace ids the agent may pass to view/search tools. */
1395
- sample_trace_ids: string[];
1396
- errors: {
1397
- trace_count: number;
1398
- span_count: number;
1399
- };
1400
- /** The COMPLETE deterministic error-signature population, sorted by
1401
- * trace_count desc. This is the failure-coverage checklist: an analysis is
1402
- * complete only when every cluster here is accounted for. Empty when the
1403
- * matched set has no error spans. */
1404
- error_clusters: ErrorCluster[];
1405
- time_range: {
1406
- earliest: string;
1407
- latest: string;
1408
- } | null;
1409
- }
1410
- interface QueryTracesPage {
1411
- traces: TraceAnalystTraceSummary[];
1412
- total: number;
1413
- has_more: boolean;
1414
- }
1415
- /** Full-trace view. When the response would exceed the per-call byte
1416
- * budget, `oversized` is populated INSTEAD of `spans` so the agent
1417
- * knows to switch to `searchTrace` / `viewSpans`. */
1418
- interface ViewTraceResult {
1419
- trace_id: string;
1420
- spans?: TraceAnalystSpan[];
1421
- oversized?: ViewTraceOversized;
1422
- }
1423
- interface ViewTraceOversized {
1424
- span_count: number;
1425
- /** Names with their counts, sorted desc. Capped at 20 entries. */
1426
- top_span_names: Array<[string, number]>;
1427
- /** Largest single span body (bytes after attribute-cap projection). */
1428
- span_response_bytes_max: number;
1429
- error_span_count: number;
1430
- }
1431
- interface ViewSpansResult {
1432
- trace_id: string;
1433
- spans: TraceAnalystSpan[];
1434
- /** Number of requested span ids that were not found in the trace. */
1435
- missing_span_ids: string[];
1436
- /** Number of attribute fields truncated to fit the per-attribute cap. */
1437
- truncated_attribute_count: number;
1438
- }
1439
- interface SpanMatchRecord {
1440
- trace_id: string;
1441
- span_id: string;
1442
- span_name: string;
1443
- span_kind: TraceAnalystSpanKind;
1444
- /** JSON pointer-style path to the matched value, e.g.
1445
- * `attributes."llm.input_messages"[2].content`. */
1446
- attribute_path: string;
1447
- matched_text: string;
1448
- context_before: string;
1449
- context_after: string;
1450
- match_offset: number;
1451
- }
1452
- interface SearchTraceResult {
1453
- trace_id: string;
1454
- hits: SpanMatchRecord[];
1455
- total_matches: number;
1456
- has_more: boolean;
1457
- }
1458
- interface SearchSpanResult {
1459
- trace_id: string;
1460
- span_id: string;
1461
- hits: SpanMatchRecord[];
1462
- total_matches: number;
1463
- has_more: boolean;
1464
- }
1465
-
1466
- /**
1467
- * `TraceAnalysisStore` — read-side interface the trace-analyst calls
1468
- * through. Six operations, all bounded:
1469
- *
1470
- * - `getOverview(filters?)` — dataset rollup + sample trace ids.
1471
- * - `queryTraces(filters?, limit, offset)` — paginated summaries.
1472
- * - `countTraces(filters?)` — cheap count without materialisation.
1473
- * - `viewTrace(trace_id, perAttrCap)` — full span list, oversized → summary.
1474
- * - `viewSpans(trace_id, span_ids, perAttrCap)` — surgical span fetch.
1475
- * - `searchTrace(trace_id, regex, max_matches)` — bounded regex hits.
1476
- * - `searchSpan(trace_id, span_id, regex, max_matches)` — single-span search.
1477
- *
1478
- * Multiple implementations ship in the core (`OtlpFileTraceStore`).
1479
- * Downstream callers can supply their own — e.g. a DuckDB-backed
1480
- * adapter or an in-memory adapter for tests — by implementing this
1481
- * interface.
1482
- *
1483
- * Filters compose with AND semantics. Empty/undefined fields impose
1484
- * no constraint. `regex_pattern` is the only opt-in raw-bytes scan —
1485
- * implementations may skip it via `count`/`overview` when not set.
1486
- */
1487
-
1488
- interface TraceAnalysisStore {
1489
- getOverview(filters?: TraceAnalystFilters): Promise<DatasetOverview>;
1490
- queryTraces(opts: {
1491
- filters?: TraceAnalystFilters;
1492
- limit: number;
1493
- offset?: number;
1494
- }): Promise<QueryTracesPage>;
1495
- countTraces(filters?: TraceAnalystFilters): Promise<number>;
1496
- viewTrace(opts: {
1497
- trace_id: string;
1498
- /** Override per-attribute byte cap. Defaults to discovery budget. */
1499
- per_attribute_byte_cap?: number;
1500
- }): Promise<ViewTraceResult>;
1501
- viewSpans(opts: {
1502
- trace_id: string;
1503
- span_ids: readonly string[];
1504
- /** Override per-attribute byte cap. Defaults to surgical budget. */
1505
- per_attribute_byte_cap?: number;
1506
- }): Promise<ViewSpansResult>;
1507
- searchTrace(opts: {
1508
- trace_id: string;
1509
- regex_pattern: string;
1510
- /** Hard cap on matches returned. Default 50. */
1511
- max_matches?: number;
1512
- }): Promise<SearchTraceResult>;
1513
- searchSpan(opts: {
1514
- trace_id: string;
1515
- span_id: string;
1516
- regex_pattern: string;
1517
- max_matches?: number;
1518
- }): Promise<SearchSpanResult>;
1519
- }
1520
-
1521
- /**
1522
- * ChatClient — the single LLM abstraction analysts call.
1523
- *
1524
- * agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
1525
- * graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
1526
- * mixed patterns force every analyst author to pick a transport, which
1527
- * couples analyst code to runtime concerns (cli-bridge vs router vs
1528
- * sandbox-sdk) it shouldn't know about.
1529
- *
1530
- * `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
1531
- * The operator decides at the registry boundary which transport binds
1532
- * to it. Analyst code stays transport-agnostic; swapping production
1533
- * (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
1534
- * line factory call.
1535
- *
1536
- * Designed to coexist: existing `LlmClient` callers and existing
1537
- * `TCloud`-based judges keep working untouched. New analyst code uses
1538
- * `ChatClient`. When old call sites migrate, they pick up budgeting,
1539
- * cancellation, and unified telemetry for free.
1540
- */
1541
-
1542
- /**
1543
- * Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
1544
- * compatible mental model stays. Two methods: a one-shot `chat()` and
1545
- * an `streamChat()` for future agentic loops (not yet exposed).
1546
- */
1547
- interface ChatClient {
1548
- /** Display name of the bound transport — included in telemetry. */
1549
- readonly transport: ChatTransport;
1550
- /** Default model when caller omits — operators bind this per environment. */
1551
- readonly defaultModel?: string;
1552
- /** Total provider attempts this transport can make for one chat call. */
1553
- readonly maximumAttempts?: number;
1554
- /** Implementations must enforce `req.maxTokens` when it is present. */
1555
- chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
1556
- }
1557
- type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
1558
- interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
1559
- /** Optional — falls back to ChatClient.defaultModel. */
1560
- model?: string;
1561
- }
1562
- type ChatResponse = LlmCallResult;
1563
- interface ChatCallOpts {
1564
- /** Cancel the in-flight request. */
1565
- signal?: AbortSignal;
1566
- /** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
1567
- maxCostUsd?: number;
1568
- /** Correlation tag carried into request headers when the transport allows. */
1569
- correlationId?: string;
1570
- /** Stable provider idempotency key for retries/redrives of one paid call. */
1571
- idempotencyKey?: string;
1572
- }
1573
- type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
1574
- interface BaseTransportOpts {
1575
- defaultModel?: string;
1576
- /** Total provider attempts. Required for opaque transports used in capped runs. */
1577
- maximumAttempts?: number;
1578
- }
1579
- interface RouterTransportOpts extends BaseTransportOpts {
1580
- transport: 'router';
1581
- baseUrl?: string;
1582
- apiKey: string;
1583
- }
1584
- interface CliBridgeTransportOpts extends BaseTransportOpts {
1585
- transport: 'cli-bridge';
1586
- baseUrl?: string;
1587
- bearer?: string;
1588
- }
1589
- interface DirectProviderTransportOpts extends BaseTransportOpts {
1590
- transport: 'direct-provider';
1591
- baseUrl: string;
1592
- apiKey: string;
1593
- }
1594
- /**
1595
- * Sandbox-SDK transport. Provided as a thin pass-through: the caller
1596
- * supplies a callable that mimics LlmClient.chat() against an already-
1597
- * configured Sandbox handle. We don't import the SDK here to keep
1598
- * agent-eval dep-free of @tangle-network/sandbox.
1599
- */
1600
- interface SandboxSdkTransportOpts extends BaseTransportOpts {
1601
- transport: 'sandbox-sdk';
1602
- chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1603
- }
1604
- /**
1605
- * Mock transport for tests. The handler receives the request and returns
1606
- * whatever the test wants. No retries, no JSON-schema degrade.
1607
- */
1608
- interface MockTransportOpts extends BaseTransportOpts {
1609
- transport: 'mock';
1610
- handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1611
- }
1612
- /**
1613
- * Build a ChatClient bound to a specific transport. The returned client
1614
- * is safe to share across analysts in a single registry run.
1615
- */
1616
- declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
1617
-
1618
- /**
1619
- * Analyst contract — the missing orchestration layer over agent-eval's
1620
- * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
1621
- * SemanticConceptJudge, JudgeFn, ...).
1622
- *
1623
- * Each existing primitive returns its own output shape. The Analyst
1624
- * contract is the single envelope every primitive lifts into, so a
1625
- * registry can run N analysts against a run and a single renderer can
1626
- * compose findings without knowing which analyzer produced them.
1627
- *
1628
- * The contract is intentionally domain-agnostic: nothing here knows
1629
- * about code, voice, RAG, or any particular agent stack. Analysts
1630
- * declare what INPUT KIND they need (a trace store, an artifact dir,
1631
- * a RunRecord, a JudgeInput, or `custom`), and the registry routes
1632
- * the matching input from `AnalystRunInputs`.
1633
- */
1634
-
1635
- /**
1636
- * Unified envelope every analyst emits. Schema-versioned so renderers
1637
- * and time-series diffs survive future field additions.
1638
- */
1639
- interface AnalystFinding {
1640
- schema_version: '1.0.0';
1641
- /**
1642
- * Stable hash over identity-defining fields (analyst_id + canonical
1643
- * claim + area + optional subject). Two findings from two runs that
1644
- * "are the same finding" share this id — that's what `diffFindings`
1645
- * uses to compute appeared/disappeared sets across runs.
1646
- */
1647
- finding_id: string;
1648
- analyst_id: string;
1649
- produced_at: string;
1650
- severity: AnalystSeverity;
1651
- /**
1652
- * Coarse classification. Renderers group by this. Free-form so
1653
- * domain-specific analysts can introduce categories without a
1654
- * schema change ('agent-reasoning', 'verification', 'cost',
1655
- * 'tool-use', 'safety', 'latency', 'data-quality', ...).
1656
- */
1657
- area: string;
1658
- claim: string;
1659
- rationale?: string;
1660
- evidence_refs: EvidenceRef[];
1661
- recommended_action?: string;
1662
- validation_plan?: string;
1663
- /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
1664
- confidence: number;
1665
- /**
1666
- * Optional subject the finding is about — leaf id, agent id, request
1667
- * id. Included in finding_id when present so per-subject findings
1668
- * diff cleanly across runs.
1669
- */
1670
- subject?: string;
1671
- /** FIREWALL provenance (docs/learning-flywheel.md): true iff this finding was
1672
- * lifted from a JUDGE verdict (an acceptance score), not OBSERVED from the
1673
- * agent's behavior. A judge-derived finding must NEVER be admitted as a
1674
- * steering input — that is the held-out judge leaking into the loop. Set at
1675
- * the lift site (createJudgeAdapter); checked by `assertNoJudgeVerdict`.
1676
- * Provenance, not evidence presence, is the correct discriminator: an
1677
- * evidence-less trace-analyst observation legitimately steers, while a judge
1678
- * verdict that happens to cite an artifact must not. */
1679
- derived_from_judge?: boolean;
1680
- /** Analyst-private extras; renderers ignore unless they know the analyst. */
1681
- metadata?: Record<string, unknown>;
1682
- }
1683
- type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
1684
- interface EvidenceRef {
1685
- /**
1686
- * Where the evidence lives. `span` and `event` refer to OTLP trace
1687
- * elements; `artifact` to a file inside the run's artifact tree;
1688
- * `finding` to another AnalystFinding (cross-analyst chaining);
1689
- * `metric` to a named scalar reading the renderer knows how to read.
1690
- */
1691
- kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
1692
- uri: string;
1693
- excerpt?: string;
1694
- }
1695
- /**
1696
- * The discriminator the registry uses to pass the right input.
1697
- * `custom` is the escape hatch — analysts that need something else
1698
- * (e.g. an embedding cache, a partner SDK handle) read it from
1699
- * `AnalystRunInputs.custom[<analyst id>]`.
1700
- */
1701
- type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
1702
- interface AnalystCost {
1703
- /** `deterministic` analysts MUST NOT call the LLM. */
1704
- kind: 'deterministic' | 'llm';
1705
- /** Optional declared upper bound; the registry can enforce a budget. */
1706
- est_usd_per_run?: number;
1707
- /** Models the analyst expects to use (informational). */
1708
- models?: string[];
1709
- /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */
1710
- settlement_timeout_ms?: number;
1711
- }
1712
- interface AnalystRequirements {
1713
- /** Min number of shots / samples the analyst needs to produce signal. */
1714
- min_shots?: number;
1715
- /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
1716
- capabilities?: string[];
1717
- }
1718
- /**
1719
- * What's passed to every analyst call. The registry resolves which
1720
- * field the analyst's `inputKind` selects and asserts it's present.
1721
- */
1722
- interface AnalystRunInputs {
1723
- traceStore?: TraceAnalysisStore;
1724
- artifactDir?: string;
1725
- runRecord?: RunRecord;
1726
- judgeInput?: JudgeInput;
1727
- /** Keyed by analyst id; populated by callers that registered custom analysts. */
1728
- custom?: Record<string, unknown>;
1729
- }
1730
- interface AnalystContext {
1731
- runId: string;
1732
- /** Stable correlation id so logs from a single registry.run() share a tag. */
1733
- correlationId: string;
1734
- /** Enforced wall-clock deadline (epoch ms). */
1735
- deadlineMs?: number;
1736
- /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
1737
- budgetUsd?: number;
1738
- /** Shared paid-call account when the analyst runs inside a larger campaign. */
1739
- costLedger?: CostLedgerHandle;
1740
- /** Attribution phase used when writing to the shared paid-call account. */
1741
- costPhase?: string;
1742
- /**
1743
- * Shared chat client. Analysts that call an LLM go through this so
1744
- * the operator picks transport (sandbox-sdk | router | cli-bridge |
1745
- * direct-provider | mock) at the registry boundary without touching
1746
- * analyst code.
1747
- */
1748
- chat?: ChatClient;
1749
- /**
1750
- * Findings from a prior run the operator wants the analyst to see as
1751
- * retrieval context. Kinds that take advantage of cross-run memory
1752
- * (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
1753
- * page I asked for is still missing") render these into the actor's
1754
- * working set. Filtering is the operator's job: pass the slice that
1755
- * matches the analyst's id, or pass everything and let the kind
1756
- * filter. Empty / absent means no cross-run context.
1757
- */
1758
- priorFindings?: ReadonlyArray<AnalystFinding>;
1759
- /**
1760
- * Findings emitted by analysts that completed earlier in this registry run.
1761
- * This is separate from `priorFindings`: upstream findings are dependency
1762
- * context for the current pass, while prior findings are cross-run memory.
1763
- * The registry populates this only when `RegistryRunOpts.chainFindings` is on.
1764
- */
1765
- upstreamFindings?: ReadonlyArray<AnalystFinding>;
1766
- /**
1767
- * Report metered work independently of findings. This keeps an empty finding
1768
- * set from erasing token/cost telemetry. Multiple receipts are accumulated.
1769
- */
1770
- recordUsage?: (receipt: AnalystUsageReceipt) => void;
1771
- /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
1772
- tags?: Record<string, string>;
1773
- /** Logger callback — analysts SHOULD prefer this over console.* for testability. */
1774
- log?: (msg: string, fields?: Record<string, unknown>) => void;
1775
- /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
1776
- signal?: AbortSignal;
1777
- }
1778
- /**
1779
- * The minimal contract. Concrete analysts can refine `TInput` so
1780
- * implementations stay type-safe (e.g. a trace analyst's `TInput` is
1781
- * `TraceAnalysisStore`); the registry passes the right field from
1782
- * `AnalystRunInputs` based on `inputKind`.
1783
- */
1784
- interface Analyst<TInput = unknown> {
1785
- /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
1786
- readonly id: string;
1787
- /** Human-readable. One sentence. */
1788
- readonly description: string;
1789
- readonly inputKind: AnalystInputKind;
1790
- readonly cost: AnalystCost;
1791
- readonly requires?: AnalystRequirements;
1792
- /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
1793
- readonly version: string;
1794
- analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
1795
- }
1796
- /** Metered work performed by one analyst call. */
1797
- interface AnalystUsageReceipt {
1798
- /** Number of model-usage records observed at the provider boundary. */
1799
- calls: number | null;
1800
- /** Null when the provider did not return token accounting. */
1801
- tokens: RunTokenUsage | null;
1802
- /** Observed, estimated, or explicitly uncaptured dollar cost. */
1803
- cost: RunCostProvenance;
1804
- /** Known lower bound when one or more calls have uncaptured cost. */
1805
- knownCostUsd?: number;
1806
- }
1807
- /**
1808
- * Compute the stable finding_id from the identity-defining fields.
1809
- * Default implementation hashes {analyst_id, area, subject, normalized claim}.
1810
- * Analysts that emit findings whose claim text varies per run (timestamps,
1811
- * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
1812
- * or (b) move the variable part into `rationale`/`metadata` and keep the
1813
- * `claim` static.
1814
- */
1815
- declare function computeFindingId(input: {
1816
- analyst_id: string;
1817
- area: string;
1818
- subject?: string;
1819
- claim: string;
1820
- /** Override the claim for hashing — use when the displayed claim has run-specific bits. */
1821
- id_basis?: string;
1822
- }): string;
1823
- /**
1824
- * Convenience factory: produce a fully-formed AnalystFinding with the
1825
- * id computed automatically. Analyst code stays terse.
1826
- */
1827
- declare function makeFinding(init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {
1828
- id_basis?: string;
1829
- produced_at?: string;
1830
- }): AnalystFinding;
1831
- interface AnalystRunSummary {
1832
- analyst_id: string;
1833
- status: 'ok' | 'skipped' | 'failed';
1834
- /** Why skipped — missing input, budget exceeded, capability unmet. */
1835
- reason?: string;
1836
- findings_count: number;
1837
- latency_ms: number;
1838
- cost_usd: number;
1839
- /**
1840
- * Additive receipt for model usage. Registry-produced summaries populate it
1841
- * even when the analyst emits no findings. `cost_usd` remains the legacy
1842
- * numeric field; inspect `usage.cost` before treating zero as observed.
1843
- */
1844
- usage?: AnalystUsageReceipt;
1845
- /** When `status='failed'`: the error class + message, never the full stack. */
1846
- error?: {
1847
- class: string;
1848
- message: string;
1849
- };
1850
- }
1851
- interface AnalystRunResult {
1852
- run_id: string;
1853
- correlation_id: string;
1854
- started_at: string;
1855
- ended_at: string;
1856
- findings: AnalystFinding[];
1857
- per_analyst: AnalystRunSummary[];
1858
- /** Total LLM cost in USD across all analysts in this registry.run(). */
1859
- total_cost_usd: number;
1860
- /**
1861
- * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only
1862
- * the known subtotal and must not be treated as the run's total spend.
1863
- */
1864
- total_cost_provenance?: RunCostProvenance;
1865
- }
1866
- /**
1867
- * Events emitted by `AnalystRegistry.runStream(...)` in real time as
1868
- * the registry executes. UIs subscribe via `for await (const ev of
1869
- * registry.runStream(...))`; `registry.run(...)` is a thin collector
1870
- * over the same stream, so the two surfaces share their invariants.
1871
- *
1872
- * Per-finding events are intentionally omitted — analyzers are batch
1873
- * operations (an Ax actor returns the full `findings:json[]` at the
1874
- * end of the responder), so streaming inside one analyst would only
1875
- * emit partial JSON consumers can't render. The kind-completion event
1876
- * is the right granularity; subscribers wanting per-finding rendering
1877
- * iterate `event.findings` themselves.
1878
- */
1879
- type AnalystRunEvent = {
1880
- type: 'run-started';
1881
- run_id: string;
1882
- correlation_id: string;
1883
- started_at: string;
1884
- /** The ordered list of analyst ids the registry will run. */
1885
- analyst_ids: ReadonlyArray<string>;
1886
- } | {
1887
- type: 'analyst-skipped';
1888
- summary: AnalystRunSummary;
1889
- } | {
1890
- type: 'analyst-started';
1891
- analyst_id: string;
1892
- started_at: string;
1893
- } | {
1894
- type: 'analyst-completed';
1895
- /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
1896
- summary: AnalystRunSummary;
1897
- findings: ReadonlyArray<AnalystFinding>;
1898
- } | {
1899
- type: 'run-completed';
1900
- result: AnalystRunResult;
1901
- };
1902
-
1903
- /**
1904
- * Adapter factories — lift each existing agent-eval primitive into the
1905
- * Analyst contract without re-implementing it.
1906
- *
1907
- * Five primitives, five factories. Each one:
1908
- * - Builds an Analyst with a stable id (caller chooses; defaults
1909
- * given), a sensible default `inputKind`, a version derived from
1910
- * the wrapped primitive's version + an adapter revision, and an
1911
- * `analyze()` that calls the primitive and lifts its output to
1912
- * AnalystFinding[] using `makeFinding()`.
1913
- * - Maps severities: the existing `Severity` ('critical' | 'major' |
1914
- * 'minor' | 'info') projects onto AnalystSeverity ('critical' |
1915
- * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →
1916
- * 'medium'. Domain analysts that want finer-grained mapping override.
1917
- *
1918
- * Adapters never own state. Calling the same factory twice with the
1919
- * same primitive instance is safe.
1920
- */
1921
-
1922
- declare function liftSeverity(s: Severity): AnalystSeverity;
1923
- interface VerifierAdapterOpts<Env> {
1924
- id?: string;
1925
- area?: string;
1926
- verifier: MultiLayerVerifier<Env>;
1927
- /**
1928
- * The verifier expects an `env` per run. Adapters take it from
1929
- * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.
1930
- */
1931
- options?: Omit<VerifyOptions<Env>, 'env'>;
1932
- }
1933
- declare function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env>;
1934
- interface RunCriticAdapterOpts {
1935
- id?: string;
1936
- area?: string;
1937
- critic?: RunCritic;
1938
- /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */
1939
- threshold?: number;
1940
- }
1941
- declare function createRunCriticAdapter(opts?: RunCriticAdapterOpts): Analyst<RunTrace>;
1942
- interface JudgeAdapterOpts {
1943
- id?: string;
1944
- area?: string;
1945
- judge: JudgeFn;
1946
- /** TCloud handle the JudgeFn calls. */
1947
- tcloud: TCloud;
1948
- /** Optional cost classification — most judges call an LLM. */
1949
- cost?: Analyst['cost'];
1950
- /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */
1951
- threshold?: number;
1952
- }
1953
- declare function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput>;
1954
- interface SemanticConceptJudgeAdapterOpts {
1955
- id?: string;
1956
- area?: string;
1957
- /** Registry context owns cancellation and the per-analyst cost ledger. */
1958
- options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>;
1959
- /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
1960
- settlementTimeoutMs?: number;
1961
- }
1962
- declare function createSemanticConceptJudgeAdapter(opts?: SemanticConceptJudgeAdapterOpts): Analyst<SemanticConceptJudgeInput>;
1963
-
1964
- interface CreateAnalystAiConfig {
1965
- /** OpenAI-compatible API key forwarded as `Authorization: Bearer`.
1966
- * cli-bridge ignores the value on loopback but Ax requires a non-empty string. */
1967
- apiKey: string;
1968
- /** OpenAI-compatible base URL — e.g. `https://router.tangle.tools/v1` or a
1969
- * cli-bridge loopback. */
1970
- baseUrl?: string;
1971
- /** Additional headers required by the gateway, such as tenant or execution policy. */
1972
- headers?: Record<string, string>;
1973
- /** Model id forwarded to analyst calls. */
1974
- model: string;
1975
- /** Ax provider name. Defaults to the OpenAI-compatible client. */
1976
- provider?: AxAIArgs<unknown>['name'];
1977
- }
1978
- /**
1979
- * Construct the `AxAIService` an analyst kind calls through
1980
- * (`createTraceAnalystKind({ ai })`).
1981
- *
1982
- * Ax's `ai()` pins `config.model` to the OpenAI catalog enum, but every
1983
- * OpenAI-compatible router an analyst points at (router.tangle.tools,
1984
- * cli-bridge) accepts arbitrary model ids (claude-code/sonnet, openai/gpt-5.4,
1985
- * …). Consumers were each re-rolling `ai({ name, apiKey, apiURL, config })`
1986
- * behind an `as (a: any) => any` cast to dodge the enum; this is the one
1987
- * canonical constructor so they don't have to — and don't take a direct
1988
- * `@ax-llm/ax` dependency for it.
1989
- */
1990
- declare function createAnalystAi(config: CreateAnalystAiConfig): AxAIService;
1991
-
1992
- /**
1993
- * Deterministic behavioral metrics over OTLP spans — pure arithmetic, no LLM.
1994
- *
1995
- * These are the model-independent multiplier: the four trace-quality signals a
1996
- * tolerant analyzer (e.g. HALO) re-derives per run inside the model — token
1997
- * growth, output decay, tool monoculture, missing self-verification — computed
1998
- * here once, in TypeScript, with zero model judgment. A finding that falls out
1999
- * of arithmetic is trivially model-agnostic and cannot hallucinate the trend.
2000
- *
2001
- * General, not trace-specific: the detectors key off token trajectories and
2002
- * tool usage present in any agentic OTLP trace, not any one benchmark.
2003
- */
2004
-
2005
- type SuboptimalCode = 'monotonic-input-growth' | 'output-length-decay' | 'single-tool-dependency' | 'no-self-verification';
2006
- interface SuboptimalSignal {
2007
- code: SuboptimalCode;
2008
- severity: 'high' | 'medium' | 'low';
2009
- /** Human-readable claim, with the backing numbers inlined. */
2010
- detail: string;
2011
- /** The exact figures the detector fired on — auditable, no model in the loop. */
2012
- evidence: Record<string, number | string | boolean>;
2013
- }
2014
- interface BehavioralMetrics {
2015
- /** The only trace represented by these metrics; null when spans are empty. */
2016
- traceId: string | null;
2017
- llmCallCount: number;
2018
- /** Causally serial LLM timelines. Parallel branches are never joined. */
2019
- tokenSequences: BehavioralTokenSequence[];
2020
- /** Token values from the longest serial timeline, retained for convenience. */
2021
- inputTokenTrajectory: number[];
2022
- outputTokenTrajectory: number[];
2023
- toolHistogram: Record<string, number>;
2024
- totalToolCalls: number;
2025
- distinctTools: number;
2026
- /** distinct/total tool calls; 1.0 when there are no tool calls. */
2027
- toolDiversityRatio: number;
2028
- hasSelfVerification: boolean;
2029
- signals: SuboptimalSignal[];
2030
- }
2031
- interface BehavioralTokenSequence {
2032
- scopeId: string;
2033
- spanIds: string[];
2034
- inputTokenTrajectory: Array<number | null>;
2035
- outputTokenTrajectory: Array<number | null>;
2036
- }
2037
-
2038
- /**
2039
- * `behavioralAnalyst` — a DETERMINISTIC analyst (cost.kind = 'deterministic',
2040
- * never calls the LLM). It produces the efficiency/behavioral findings a
2041
- * tolerant agentic analyzer (HALO) re-derives per run inside the model —
2042
- * context bloat, output decay, tool monoculture, missing self-verification —
2043
- * directly from arithmetic over spans (`computeTraceMetrics`).
2044
- *
2045
- * Why it matters: these findings are model-agnostic BY CONSTRUCTION (no model
2046
- * in the loop), so they cannot return 0 on a weak model the way the Ax-RLM
2047
- * does — and they are strictly more reliable than HALO, which spends tokens
2048
- * re-deriving the same numbers and can hallucinate the trend. The agentic
2049
- * RLM kinds remain for SEMANTIC findings that genuinely need a model; this
2050
- * analyst owns the behavioral class.
2051
- */
2052
-
2053
- /**
2054
- * Map computed signals → structured AnalystFindings. Pure: no LLM, no clock
2055
- * dependence beyond `produced_at` (overridable for deterministic tests).
2056
- */
2057
- declare function deriveEfficiencyFindings(metrics: BehavioralMetrics, opts?: {
2058
- analystId?: string;
2059
- producedAt?: string;
2060
- }): AnalystFinding[];
2061
- /** The deterministic behavioral/efficiency analyst (no LLM, any-model). */
2062
- declare function behavioralAnalyst(): Analyst<TraceAnalysisStore>;
2063
-
2064
- /**
2065
- * Typed Ax output for analyst findings.
2066
- *
2067
- * Replaces the legacy `findings:string[]` pattern (where every bullet
2068
- * became a flat-severity `AnalystFinding`) with a structured object
2069
- * array. Ax binds the field as `findings:json[]` so the provider emits
2070
- * native structured output; at the kind-factory boundary we Zod-validate
2071
- * each emitted finding so malformed rows fail loud instead of being
2072
- * silently lifted with default severity.
2073
- *
2074
- * Why not `f.object().array()` directly in the signature? The Ax
2075
- * signature string `question:string -> findings:json[]` already lets
2076
- * the provider emit JSON arrays. A Zod boundary is required either
2077
- * way (the provider can return any JSON), and Zod gives us a single
2078
- * validation surface independent of which Ax version is installed.
2079
- */
2080
-
2081
- declare const ANALYST_SEVERITIES: readonly ["critical", "high", "medium", "low", "info"];
2082
- declare const RawAnalystEvidenceSchema: z.ZodObject<{
2083
- uri: z.ZodString;
2084
- excerpt: z.ZodOptional<z.ZodString>;
2085
- }, z.core.$strict>;
2086
- type RawAnalystEvidence = z.infer<typeof RawAnalystEvidenceSchema>;
2087
- /** Original public schema retained for stored rows and callback contracts. */
2088
- declare const RawAnalystFindingSchema: z.ZodObject<{
2089
- evidence_uri: z.ZodString;
2090
- evidence_excerpt: z.ZodOptional<z.ZodString>;
2091
- severity: z.ZodEnum<{
2092
- critical: "critical";
2093
- info: "info";
2094
- low: "low";
2095
- high: "high";
2096
- medium: "medium";
2097
- }>;
2098
- claim: z.ZodString;
2099
- subject: z.ZodOptional<z.ZodString>;
2100
- confidence: z.ZodNumber;
2101
- rationale: z.ZodOptional<z.ZodString>;
2102
- recommended_action: z.ZodOptional<z.ZodString>;
2103
- }, z.core.$strict>;
2104
- type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
2105
- /**
2106
- * Canonical plural-evidence contract. The preprocessor accepts the original
2107
- * `evidence_uri` / `evidence_excerpt` pair and normalizes it into one evidence
2108
- * item so persisted rows and older model fixtures remain readable. New output
2109
- * always receives the plural shape.
2110
- */
2111
- declare const CanonicalRawAnalystFindingSchema: z.ZodPipe<z.ZodTransform<unknown, unknown>, z.ZodObject<{
2112
- evidence: z.ZodArray<z.ZodObject<{
2113
- uri: z.ZodString;
2114
- excerpt: z.ZodOptional<z.ZodString>;
2115
- }, z.core.$strict>>;
2116
- severity: z.ZodEnum<{
2117
- critical: "critical";
2118
- info: "info";
2119
- low: "low";
2120
- high: "high";
2121
- medium: "medium";
2122
- }>;
2123
- claim: z.ZodString;
2124
- subject: z.ZodOptional<z.ZodString>;
2125
- confidence: z.ZodNumber;
2126
- rationale: z.ZodOptional<z.ZodString>;
2127
- recommended_action: z.ZodOptional<z.ZodString>;
2128
- }, z.core.$strict>>;
2129
- type CanonicalRawAnalystFinding = z.infer<typeof CanonicalRawAnalystFindingSchema>;
2130
- /**
2131
- * Description embedded into the actor prompt so the LLM knows what
2132
- * shape to emit. Kept here so kinds share one source of truth rather
2133
- * than restating the schema in every prompt.
2134
- */
2135
- declare const RAW_FINDING_SCHEMA_PROMPT = "Each finding MUST be a strict JSON object with:\n - severity: \"critical\" | \"high\" | \"medium\" | \"low\" | \"info\"\n - claim: one-sentence statement (max 2000 chars)\n - subject?: one exact subject form listed by this kind; omit rather than guess\n - evidence: REQUIRED non-empty array of {\"uri\": string, \"excerpt\"?: string}. Use real identifiers with span://, event://, artifact://, metric://, or finding://. Include a short exact quote in excerpt when available. If nothing is citable, do not emit the finding.\n - confidence: number 0..1 (0.9+ exact evidence; 0.6-0.8 inferred pattern; <0.5 speculative)\n - rationale?: one or two reasoning sentences\n - recommended_action?: concrete imperative change; omit for descriptive findings\n\nUnknown fields are rejected. Do not emit area; the factory assigns it. Emit [] when there are no findings. Never fabricate evidence.";
2136
- /** Convert canonical raw citations into the public finding evidence envelope. */
2137
- declare function evidenceRefsFromRawFinding(finding: CanonicalRawAnalystFinding): EvidenceRef[];
2138
- /**
2139
- * Validate the original singular-evidence shape. This public parser retains
2140
- * its pre-canonicalization result type so existing callback code and stored
2141
- * rows continue to receive exactly the object accepted by
2142
- * {@link RawAnalystFindingSchema}.
2143
- */
2144
- declare function parseRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): RawAnalystFinding | null;
2145
- /** Validate model output and normalize original singular citations. */
2146
- declare function parseCanonicalRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): CanonicalRawAnalystFinding | null;
2147
-
2148
- /**
2149
- * Analyst-kind factory — the typed way to define trace analysts.
2150
- *
2151
- * A "kind" is a specialized analyst whose actor prompt, tool subset,
2152
- * and bounded Ax subqueries target one failure-mode lens (failure-mode
2153
- * classification, knowledge gap discovery, knowledge poisoning,
2154
- * self-improvement, ...). Kinds emit findings in the typed
2155
- * `CanonicalRawAnalystFinding` shape via a JSON-array Ax output; the factory
2156
- * validates each row with Zod and lifts it into `AnalystFinding[]`.
2157
- *
2158
- * Composition rules:
2159
- * - Each kind owns its actor description. No generic "answer this
2160
- * question" prompt — the prompt names the failure lens.
2161
- * - Each kind picks a narrow tool subset from `ANALYST_TOOL_GROUPS`.
2162
- * A kind that never needs full-trace dumps can drop `viewTrace` /
2163
- * `viewSpans` and stay cheap.
2164
- * - Each kind declares its subquery + parallelism budget. Discovery-heavy
2165
- * kinds can fan out more bounded semantic questions than narrow lenses.
2166
- *
2167
- * Optimizer hook: kinds may declare `goldens` — labeled examples used
2168
- * by `AxBootstrapFewShot` / `AxGEPA` to fit the actor
2169
- * description programmatically. Stored on the kind, not the registry,
2170
- * because the right metric is kind-specific.
2171
- */
2172
-
2173
- /**
2174
- * Per-kind specification. The factory turns this into a regular
2175
- * `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
2176
- */
2177
- interface TraceAnalystKindSpec {
2178
- /** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
2179
- id: string;
2180
- /** One-sentence description shown in `registry.list()`. */
2181
- description: string;
2182
- /** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
2183
- area: string;
2184
- /** Bump on any breaking change to the actor prompt or output schema. */
2185
- version: string;
2186
- /** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
2187
- actorDescription: string;
2188
- /** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
2189
- buildTools: (store: TraceAnalysisStore) => AxFunction[];
2190
- /** Bounded semantic subqueries. `maxCalls: 0` disables model fan-out. */
2191
- subqueries?: {
2192
- maxCalls: number;
2193
- maxParallel?: number;
2194
- };
2195
- /** Actor turn cap. Default 12. */
2196
- maxTurns?: number;
2197
- /** Runtime char cap. Default 6000. */
2198
- maxRuntimeChars?: number;
2199
- /** Maximum output tokens for every actor and subquery model call. Default 4096. */
2200
- maxOutputTokens?: number;
2201
- /** Cost classification surfaced in `registry.list()` and budget enforcement. */
2202
- cost: AnalystCost;
2203
- /** Per-finding-row hook — kinds may reject / rewrite before lifting. */
2204
- postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
2205
- /** Minimum citations per finding. Default 1; rows below it are rejected. */
2206
- minimumEvidenceCitations?: number;
2207
- /** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
2208
- goldens?: TraceAnalystGolden[];
2209
- }
2210
- /**
2211
- * One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
2212
- * Each input is the same `{question}` an analyst would receive; `expected`
2213
- * is the ground-truth finding set a fitted prompt should produce on this
2214
- * input. Metric: kind-specific (default: F1 on `finding_id` overlap).
2215
- */
2216
- interface TraceAnalystGolden {
2217
- question: string;
2218
- expected: ReadonlyArray<Omit<CanonicalRawAnalystFinding, 'confidence'>>;
2219
- }
2220
- interface CreateTraceAnalystKindOpts {
2221
- /** AxAIService bound at registration time. */
2222
- ai: AxAIService;
2223
- /** Required unless `ai` was created by {@link createAnalystAi}. */
2224
- model?: string;
2225
- /** Override the spec's `version` (e.g. when an optimizer has fitted a new prompt). */
2226
- versionSuffix?: string;
2227
- /**
2228
- * Optional two-phase recovery: when the agentic harvest is empty but the
2229
- * actor produced a substantive free-form `report`, extract findings from that
2230
- * prose via a tolerant chat-completions pass (`structureFindings`) — no
2231
- * strict-emission contract, so it works on weak models. Omit to leave the
2232
- * actor's harvest as-is (the report is still surfaced fail-loud either way).
2233
- */
2234
- recovery?: {
2235
- baseUrl: string;
2236
- apiKey?: string;
2237
- model?: string;
2238
- fetchImpl?: typeof fetch;
2239
- };
2240
- /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
2241
- settlementTimeoutMs?: number;
2242
- }
2243
- /**
2244
- * Build an `Analyst<TraceAnalysisStore>` from a kind spec.
2245
- *
2246
- * Lifts the Ax pipeline once at registration time so the registry
2247
- * gets a stateless analyst. The Ax agent is freshly constructed per
2248
- * `analyze()` call (the agent carries chat-log + usage state we don't
2249
- * want shared across analyst runs).
2250
- */
2251
- declare function createTraceAnalystKind(spec: TraceAnalystKindSpec, opts: CreateTraceAnalystKindOpts): Analyst<TraceAnalysisStore>;
2252
- /**
2253
- * Render a compact prior-findings block the actor reads alongside its
2254
- * brief. Each row is one line so the actor can scan dozens cheaply.
2255
- * The kind's prompt instructs the actor to (a) check whether a new
2256
- * cluster matches a prior `finding_id` (carry the id forward via
2257
- * `id_basis` to keep diffs stable) and (b) raise severity / confidence
2258
- * when a prior finding has reappeared without remediation.
2259
- *
2260
- * Returns the empty string when there are no prior findings — most
2261
- * runs are "first-of-its-kind" and the prompt stays unchanged.
2262
- *
2263
- * Exported for tests + for consumers that build their own actor
2264
- * prompts (e.g. specialized analysts living outside the default kinds).
2265
- */
2266
- declare function renderPriorFindings(prior: AnalystContext['priorFindings']): string;
2267
- /** Render findings produced earlier in this same registry run. */
2268
- declare function renderUpstreamFindings(upstream: AnalystContext['upstreamFindings']): string;
2269
-
2270
- /**
2271
- * AnalystRegistry — orchestrate N analysts against one run.
2272
- *
2273
- * Owns three responsibilities and only three:
2274
- * 1. Registration — ids must be unique; bad registrations fail loudly
2275
- * at register-time, not run-time.
2276
- * 2. Routing — each analyst declares its `inputKind`; the registry
2277
- * picks the matching field from AnalystRunInputs and skips the
2278
- * analyst with a logged reason if it's missing.
2279
- * 3. Isolation — one analyst's exception MUST NOT stop other analysts.
2280
- * Failed analysts produce zero findings + a 'failed' summary row.
2281
- *
2282
- * Cross-cutting concerns (telemetry, error → finding conversion, cost
2283
- * ingestion, storage rotation) live in `AnalystHooks`. Budget shaping
2284
- * (equal split vs weighted vs custom) lives in `BudgetPolicy`. Both
2285
- * have sensible defaults; consumers override only what they need.
2286
- */
2287
-
2288
- interface AnalystHooks {
2289
- /** Before analyze() — last chance to mutate ctx (e.g. inject tags, override budget). */
2290
- onBeforeAnalyze?(args: {
2291
- analyst: Analyst;
2292
- ctx: AnalystContext;
2293
- runId: string;
2294
- }): void | Promise<void>;
2295
- /** After every analyst (ok | failed | skipped). Use for telemetry, ingestion, rotation. */
2296
- onAfterAnalyze?(args: {
2297
- analyst: Analyst;
2298
- summary: AnalystRunSummary;
2299
- findings: AnalystFinding[];
2300
- runId: string;
2301
- }): void | Promise<void>;
2302
- /**
2303
- * On analyst exception. Hook MAY return findings to convert the
2304
- * error into structured findings; the summary still reports 'failed'.
2305
- * Return void to keep the default empty-findings behavior.
2306
- */
2307
- onError?(args: {
2308
- analyst: Analyst;
2309
- error: Error;
2310
- runId: string;
2311
- }): AnalystFinding[] | undefined | Promise<AnalystFinding[] | undefined>;
2312
- /** Once after registry.run() completes. Use for final aggregation, persistence. */
2313
- onComplete?(args: {
2314
- result: AnalystRunResult;
2315
- }): void | Promise<void>;
2316
- }
2317
- interface BudgetPolicy {
2318
- /** Overall USD cap across the registry.run(). */
2319
- totalUsd?: number;
2320
- /** Per-analyst weight for the default allocator. Missing ids get weight 1. */
2321
- weights?: Record<string, number>;
2322
- /**
2323
- * Custom allocator — receives the analyst, remaining/total budget, and
2324
- * the count of analysts that will run. Returns the per-analyst budget
2325
- * (or undefined only when the run has no overall cap). Overrides weights
2326
- * when set.
2327
- */
2328
- allocate?: (args: {
2329
- analyst: Analyst;
2330
- totalUsd: number | undefined;
2331
- remainingUsd: number | undefined;
2332
- runningCount: number;
2333
- }) => number | undefined;
2334
- }
2335
- interface AnalystRegistryOptions {
2336
- /** Shared chat client passed to every LLM analyst via AnalystContext. */
2337
- chat?: ChatClient;
2338
- /** Logger callback. Defaults to a no-op. */
2339
- log?: (msg: string, fields?: Record<string, unknown>) => void;
2340
- /** Hooks invoked around analyze() — observability + customization seam. */
2341
- hooks?: AnalystHooks;
2342
- /** Default budget when run() doesn't override. */
2343
- defaultBudget?: BudgetPolicy;
2344
- }
2345
- interface RegistryRunOpts {
2346
- /** Restrict to a subset of registered analysts by id. */
2347
- only?: string[];
2348
- /** Skip these analysts even if registered. Useful for cheap iteration. */
2349
- skip?: string[];
2350
- /** Budget policy — totalUsd + optional weights/allocator. Falls back to options.defaultBudget. */
2351
- budget?: BudgetPolicy;
2352
- /** Active-work cap for the complete registry run. Model receipt settlement may follow. */
2353
- timeoutMs?: number;
2354
- /** Abort signal — forwarded into every analyst's context. */
2355
- signal?: AbortSignal;
2356
- /** Shared paid-call account forwarded to every analyst. */
2357
- costLedger?: CostLedgerHandle;
2358
- /** Attribution phase for calls written to `costLedger`. */
2359
- costPhase?: string;
2360
- /** Tags echoed into AnalystContext.tags — useful for tracking environment/version in findings. */
2361
- tags?: Record<string, string>;
2362
- /**
2363
- * Prior-run findings made available as retrieval context to every
2364
- * analyst via `ctx.priorFindings`. The registry forwards the slice
2365
- * whose `analyst_id` matches each registered analyst so a kind sees
2366
- * only its own history. Pass `{ '*': findings }` to broadcast to
2367
- * every analyst (useful when several kinds share the same historical
2368
- * context). For findings from this run, use `chainFindings` instead.
2369
- */
2370
- priorFindings?: ReadonlyArray<AnalystFinding> | Record<string, ReadonlyArray<AnalystFinding>>;
2371
- /**
2372
- * Pass findings produced earlier in this registry run to each later analyst
2373
- * via `ctx.upstreamFindings`. Registration order is dependency order.
2374
- * Disabled by default because independent analyst suites must opt in.
2375
- */
2376
- chainFindings?: boolean;
2377
- }
2378
- declare class AnalystRegistry {
2379
- private readonly analysts;
2380
- private readonly options;
2381
- constructor(options?: AnalystRegistryOptions);
2382
- register(analyst: Analyst): void;
2383
- list(): ReadonlyArray<{
2384
- id: string;
2385
- description: string;
2386
- version: string;
2387
- cost: Analyst['cost'];
2388
- }>;
2389
- run(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): Promise<AnalystRunResult>;
2390
- /**
2391
- * Streaming counterpart to `run()`. Emits `AnalystRunEvent` values
2392
- * in real time — `run-started`, then per-analyst `skipped` /
2393
- * `started` / `completed`, then a terminal `run-completed` whose
2394
- * payload is the full `AnalystRunResult`. UIs use this to render
2395
- * progress; persistence consumers use `run()` and read the result.
2396
- *
2397
- * Hooks (`onBeforeAnalyze` / `onAfterAnalyze` / `onError` /
2398
- * `onComplete`) fire as before — streaming is additive, not a hook
2399
- * replacement.
2400
- */
2401
- runStream(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): AsyncGenerator<AnalystRunEvent, void, void>;
2402
- private selectAnalysts;
2403
- private routeInput;
2404
- }
2405
-
2406
- /**
2407
- * `buildDefaultAnalystRegistry` — the canonical analyst suite, so consumers
2408
- * stop hand-wiring `new AnalystRegistry()` + per-kind `createTraceAnalystKind`.
2409
- *
2410
- * The deterministic `behavioralAnalyst` is ALWAYS registered (it needs no
2411
- * model and is model-agnostic by construction). The agentic RLM kinds are
2412
- * registered only when an `ai` service is supplied — so a caller with no LLM
2413
- * still gets the full behavioral/efficiency diagnosis, and the substrate's
2414
- * "any model (including no model)" guarantee holds at the suite level.
2415
- */
2416
-
2417
- interface DefaultAnalystRegistryOptions {
2418
- /** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
2419
- ai?: AxAIService;
2420
- /** Required unless `ai` was created by `createAnalystAi`. */
2421
- model?: string;
2422
- /** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
2423
- kinds?: readonly TraceAnalystKindSpec[];
2424
- /** Set false to omit the deterministic behavioral analyst (default: include). */
2425
- includeBehavioral?: boolean;
2426
- /** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
2427
- registry?: AnalystRegistryOptions;
2428
- }
2429
- declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
2430
-
2431
- /**
2432
- * Typed `FindingSubject` — the canonical grammar every analyst kind emits.
2433
- *
2434
- * Background: kind actor prompts have always documented a subject grammar
2435
- * (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`) but the
2436
- * LLM was unconstrained — it could emit `subject: "fix the prompt"`
2437
- * (prose) and downstream adapters routed on `startsWith(...)` would
2438
- * silently skip it. Every per-vertical `ImprovementAdapter` had a
2439
- * routing table that mostly caught nothing.
2440
- *
2441
- * This module fixes that:
2442
- * - `parseFindingSubject(raw)` — returns the typed `FindingSubject`
2443
- * when `raw` matches the grammar, else `null`. Used at the
2444
- * `RawAnalystFindingSchema` boundary so malformed subjects are
2445
- * rejected loudly instead of silently lifted into the registry.
2446
- * - `FindingSubjectKind` — the union of valid locus categories. Each
2447
- * variant carries the typed components downstream adapters resolve
2448
- * against the agent's surface manifest (no string parsing in the
2449
- * adapter).
2450
- * - `FINDING_SUBJECT_GRAMMAR_PROMPT` — single source of truth for the
2451
- * grammar string embedded in kind actor prompts. Drift between
2452
- * prompt and parser is impossible if every kind imports this.
2453
- *
2454
- * The grammar is intentionally NARROW — only loci the substrate's
2455
- * default `ImprovementAdapter` / `KnowledgeAdapter` can act on. A
2456
- * finding with a subject outside this set fails the parser; the kind
2457
- * author either extends the grammar here (and adds adapter routing)
2458
- * or rephrases the prompt to map onto an existing variant.
2459
- *
2460
- * `failure-mode` is the one exception — its subjects are free-form
2461
- * cluster labels, not loci. The schema preserves them as
2462
- * `{ kind: 'cluster', label }` and the adapters skip them (cluster
2463
- * findings are evidence, not actionable mutations).
2464
- */
2465
-
2466
- /**
2467
- * Discriminated union of every locus the substrate can route findings to.
2468
- *
2469
- * Adapters narrow on `kind` and use the typed components (no string
2470
- * parsing). Adding a variant here REQUIRES updating the parser, the
2471
- * grammar prompt, and at least one adapter — by design.
2472
- */
2473
- type FindingSubject = {
2474
- kind: 'knowledge.wiki';
2475
- slug: string;
2476
- heading?: string;
2477
- } | {
2478
- kind: 'knowledge.claim';
2479
- topic: string;
2480
- } | {
2481
- kind: 'knowledge.raw';
2482
- sourceId: string;
2483
- } | {
2484
- kind: 'knowledge.stale';
2485
- slug: string;
2486
- } | {
2487
- kind: 'system-prompt';
2488
- section: string;
2489
- } | {
2490
- kind: 'skill';
2491
- name: string;
2492
- } | {
2493
- kind: 'tool-doc';
2494
- tool: string;
2495
- aspect?: string;
2496
- } | {
2497
- kind: 'new-tool';
2498
- name: string;
2499
- } | {
2500
- kind: 'mcp';
2501
- server: string;
2502
- tool?: string;
2503
- } | {
2504
- kind: 'hook';
2505
- name: string;
2506
- } | {
2507
- kind: 'subagent';
2508
- name: string;
2509
- } | {
2510
- kind: 'workflow';
2511
- name: string;
2512
- } | {
2513
- kind: 'rollout-policy';
2514
- field: string;
2515
- } | {
2516
- kind: 'agent-profile';
2517
- field: string;
2518
- } | {
2519
- kind: 'code';
2520
- path: string;
2521
- } | {
2522
- kind: 'rag';
2523
- corpus: string;
2524
- docId: string;
2525
- } | {
2526
- kind: 'memory';
2527
- key: string;
2528
- } | {
2529
- kind: 'scaffolding';
2530
- concern: string;
2531
- } | {
2532
- kind: 'output-schema';
2533
- field: string;
2534
- } | {
2535
- kind: 'websearch.outdated';
2536
- topic: string;
2537
- } | {
2538
- kind: 'prior-run-summary';
2539
- topic: string;
2540
- } | {
2541
- kind: 'cluster';
2542
- label: string;
2543
- };
2544
- type FindingSubjectKind = FindingSubject['kind'];
2545
- declare const FINDING_SUBJECT_KINDS: ReadonlyArray<FindingSubjectKind>;
2546
- /**
2547
- * Parse a raw subject string emitted by an analyst kind's actor.
2548
- *
2549
- * Returns the typed `FindingSubject` when `raw` matches the grammar,
2550
- * else `null`. Callers use the `null` return as a signal to either
2551
- * (a) reject the finding at parse time (kinds that emit typed loci —
2552
- * knowledge-gap, improvement, knowledge-poisoning) or (b) lift it as
2553
- * a cluster label (failure-mode).
2554
- *
2555
- * Slugs are constrained to `[a-z0-9-]+` (lowercase kebab) to keep file
2556
- * paths sane downstream. Topics / keys / sections allow any non-empty
2557
- * string (free-form for the LLM's voice) but get trimmed.
2558
- *
2559
- * Empty / whitespace-only inputs return `null`. `undefined` returns
2560
- * `null`. Both are surfaced by the caller as a rejected subject.
2561
- */
2562
- declare function parseFindingSubject(raw: string | null | undefined): FindingSubject | null;
2563
- /**
2564
- * Render the parsed subject back to its canonical string form. Inverse
2565
- * of `parseFindingSubject`; useful when the substrate constructs new
2566
- * findings programmatically (e.g. for tests, replays, or
2567
- * `id_basis` carry-forward).
2568
- */
2569
- declare function renderFindingSubject(s: FindingSubject): string;
2570
- /**
2571
- * The grammar text embedded into kind actor prompts. Kinds opt into
2572
- * the subset of variants they emit (e.g. `improvement` excludes the
2573
- * cluster variant; `failure-mode` includes ONLY the cluster variant).
2574
- *
2575
- * Drift between prompt and parser is impossible: every kind imports
2576
- * this constant + the matching `expects` set, and the unit tests below
2577
- * lock the table to the parser.
2578
- */
2579
- declare const FINDING_SUBJECT_SYNTAX: Readonly<Record<FindingSubjectKind, string>>;
2580
- declare const FINDING_SUBJECT_GRAMMAR_PROMPT: string;
2581
- /**
2582
- * The variants each kind is allowed to emit. Used at the kind factory
2583
- * boundary so a knowledge-gap finding can't sneak in a `system-prompt:*`
2584
- * subject (the improvement-analyst's job) and vice versa.
2585
- *
2586
- * `failure-mode` is restricted to `cluster` — the only kind that emits
2587
- * a non-locus subject.
2588
- */
2589
- declare const KIND_EXPECTED_SUBJECTS: Record<string, ReadonlyArray<FindingSubjectKind>>;
2590
- /** Render only the subject forms one analyst kind is permitted to emit. */
2591
- declare function findingSubjectGrammarPromptFor(kindId: string): string;
2592
- /**
2593
- * Zod schema that validates a raw subject string and returns the parsed
2594
- * `FindingSubject`. Embedded in `RawAnalystFindingSchema` via
2595
- * `transform`, so `subject` arrives at the kind factory either as a
2596
- * typed locus or as a parse error attached to a single Zod issue.
2597
- *
2598
- * Optionality is preserved: subjects ARE optional on the wire (some
2599
- * findings are descriptive, not actionable). When present, they MUST
2600
- * parse — emitting a malformed subject is a contract violation, not a
2601
- * soft signal.
2602
- */
2603
- declare const FindingSubjectStringSchema: z.ZodString;
2604
-
2605
- /**
2606
- * FindingsStore — durable persistence for AnalystFinding rows + a diff
2607
- * helper so we can answer "what changed since the last run?" without
2608
- * recomputing analysts.
2609
- *
2610
- * On-disk shape is JSONL: one finding per line, append-only, locked via
2611
- * LockedJsonlAppender. Operators get crash-safety (no partial JSON),
2612
- * cheap reads (sequential parse), and trivial backup (rsync the file).
2613
- *
2614
- * Reads are non-locking: a reader sees a consistent snapshot of all
2615
- * fully-written lines and skips an incomplete trailing line if the
2616
- * writer is mid-append. Cross-process locking is intentionally out of
2617
- * scope (see locked-jsonl-appender.ts).
2618
- *
2619
- * The store is run-scoped: callers pass `runId` on append and on load,
2620
- * which keeps multi-run files cleanly partitioned. The `diffFindings`
2621
- * helper compares two run-id sets using stable `finding_id` semantics —
2622
- * the diff is the cross-run signal the regression dashboard renders.
2623
- */
2624
-
2625
- /**
2626
- * One persisted row. We attach `run_id` on disk so a single file can
2627
- * hold multiple runs and the diff helper can query without re-walking
2628
- * separate files.
2629
- */
2630
- interface PersistedFinding extends AnalystFinding {
2631
- run_id: string;
2632
- }
2633
- declare class FindingsStore {
2634
- readonly path: string;
2635
- private readonly appender;
2636
- constructor(path: string);
2637
- append(runId: string, findings: AnalystFinding[]): Promise<void>;
2638
- /** Load every persisted finding. Discards malformed trailing lines silently. */
2639
- loadAll(): PersistedFinding[];
2640
- /** Filter to a single run. */
2641
- loadRun(runId: string): PersistedFinding[];
2642
- }
2643
- interface FindingsDiff {
2644
- /** New finding ids in `current` that weren't in `previous`. */
2645
- appeared: PersistedFinding[];
2646
- /** Finding ids in `previous` that aren't in `current`. */
2647
- disappeared: PersistedFinding[];
2648
- /** Same finding id present in both runs and unchanged per the materiality test. */
2649
- persisted: PersistedFinding[];
2650
- /**
2651
- * Same finding id in both runs but at least one non-identity field
2652
- * shifted per `DiffPolicy.isMaterial`. Reported as [previous, current].
2653
- */
2654
- changed: Array<{
2655
- previous: PersistedFinding;
2656
- current: PersistedFinding;
2657
- }>;
2658
- }
2659
- interface DiffPolicy {
2660
- /**
2661
- * Predicate that decides whether two findings (same finding_id) count
2662
- * as a material change. Defaults to {@link defaultIsMaterial}: severity
2663
- * shift, confidence Δ > 0.05, or evidence count change. Compliance /
2664
- * perf consumers MAY supply a stricter predicate (e.g. rationale text
2665
- * diff, metric Δ thresholds).
2666
- */
2667
- isMaterial?: (previous: AnalystFinding, current: AnalystFinding) => boolean;
2668
- }
2669
- /**
2670
- * Default materiality test. Deliberately narrow so LLM-reword churn
2671
- * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.
2672
- */
2673
- declare function defaultIsMaterial(a: AnalystFinding, b: AnalystFinding): boolean;
2674
- /**
2675
- * Diff two findings sets by stable finding_id. Callers typically load
2676
- * the two run-id slices from the same store and pass them in.
2677
- */
2678
- declare function diffFindings(previous: PersistedFinding[], current: PersistedFinding[], policy?: DiffPolicy): FindingsDiff;
2679
-
2680
- /**
2681
- * Failure-mode analyst — classifies what went wrong and why.
2682
- *
2683
- * Brief: read the trace dataset, identify the top failure modes across
2684
- * runs, classify each with severity + evidence, and surface them as
2685
- * findings. The actor's job is *taxonomy + evidence*, not fix-design —
2686
- * that's the improvement-analyst's job.
2687
- *
2688
- * Eight bounded model subqueries let the actor compare candidate
2689
- * clusters in parallel after it has loaded representative evidence.
2690
- */
2691
-
2692
- declare const FAILURE_MODE_KIND_SPEC: TraceAnalystKindSpec;
2693
-
2694
- /**
2695
- * Improvement analyst — actionable self-improvement findings.
2696
- *
2697
- * Brief: read findings from upstream analysts (failure-mode,
2698
- * knowledge-gap, knowledge-poisoning) AND the trace dataset itself,
2699
- * then propose **concrete edits** to the agent's runtime: prompt
2700
- * additions, RAG documents to ingest, tool descriptions to rewrite,
2701
- * scaffolding changes to make, memory entries to invalidate. Each
2702
- * finding is one proposed edit with the locus, the diff, and the
2703
- * expected effect.
2704
- *
2705
- * This is the self-improvement loop's last mile: the prior
2706
- * kinds describe *what's wrong*; this kind describes *what to change*.
2707
- *
2708
- * Eight bounded model subqueries let the actor compare competing fix
2709
- * directions over the same cited evidence before recommending one.
2710
- */
2711
-
2712
- declare const IMPROVEMENT_KIND_SPEC: TraceAnalystKindSpec;
2713
-
2714
- /**
2715
- * Knowledge-gap analyst — what did the agent NOT know that it needed?
2716
- *
2717
- * Brief: find moments in the trace where the agent had to guess, ask
2718
- * the user to fill in context, recover from a wrong assumption, or
2719
- * loop on a retrieval. Each finding names a *missing or outdated piece
2720
- * of knowledge* the agent's curated knowledge base should have held —
2721
- * or a downstream lookup (web, docs, tool description) that surfaced
2722
- * stale or outdated information.
2723
- *
2724
- * The primary expected store is `@tangle-network/agent-knowledge`: a
2725
- * Karpathy-style wiki the agent maintains with raw ↔ curated pages,
2726
- * source anchors, and claim/relation triples. A gap is anything the
2727
- * agent had to discover at run-time that should already have lived
2728
- * there. Secondary loci: web-search results that returned outdated
2729
- * pages, tool descriptions that omitted critical behavior, system-
2730
- * prompt sections that didn't cover the case.
2731
- *
2732
- * Distinct from failure-mode: failure-mode classifies *how* it broke;
2733
- * knowledge-gap names the *information* whose absence (or staleness)
2734
- * caused the break. One failure-mode often maps to several gaps.
2735
- *
2736
- * Five bounded model subqueries let the actor compare candidate gaps
2737
- * across source layers after it has loaded the relevant excerpts.
2738
- */
2739
-
2740
- declare const KNOWLEDGE_GAP_KIND_SPEC: TraceAnalystKindSpec;
2741
-
2742
- /**
2743
- * Knowledge-poisoning analyst — what FALSE information misled the agent?
2744
- *
2745
- * Brief: find moments where the agent acted on information that was
2746
- * *wrong* — stale memory, RAG documents that contradicted ground truth,
2747
- * tool descriptions that lied about return shapes, system-prompt
2748
- * instructions that no longer matched reality, prior-run summaries that
2749
- * cached a wrong decision.
2750
- *
2751
- * Distinct from knowledge-gap: a gap is "the agent didn't know X"; a
2752
- * poisoning is "the agent confidently used X, but X was wrong." Gaps
2753
- * surface as questions / self-correction; poisonings surface as
2754
- * confident-but-wrong actions that downstream evidence contradicts.
2755
- *
2756
- * Eight bounded model subqueries let the actor independently assess
2757
- * the action and contradiction excerpts for candidate poisonings.
2758
- */
2759
-
2760
- declare const KNOWLEDGE_POISONING_KIND_SPEC: TraceAnalystKindSpec;
2761
-
2762
- /**
2763
- * Default analyst kinds focused on agent failure + recursive
2764
- * self-improvement.
2765
- *
2766
- * The four kinds chain: failure-mode classifies; knowledge-gap and
2767
- * knowledge-poisoning explain *why* in two orthogonal ways; improvement
2768
- * proposes concrete edits. Register all four against the same trace
2769
- * store in this order and run the registry with `chainFindings: true`
2770
- * to pass each completed kind's findings to the kinds that follow it.
2771
- */
2772
-
2773
- /**
2774
- * The default kind suite. Order is the run order operators should
2775
- * use: failure-mode first (no upstream deps), gap + poisoning next
2776
- * (both depend on failures), improvement last (chains all three).
2777
- */
2778
- declare const DEFAULT_TRACE_ANALYST_KINDS: readonly TraceAnalystKindSpec[];
2779
-
2780
- /**
2781
- * Skill-usage analyst — a DETERMINISTIC `Analyst` over a Claude/Codex skill
2782
- * library + its trace corpus. Unlike the trace-store kinds (failure-mode,
2783
- * improvement, ...) this kind calls no LLM: it mines real usage and skill
2784
- * structure and emits findings by rule.
2785
- *
2786
- * It exists because the naive "Skill-tool invocation count" lies low — it
2787
- * misses orchestrated sub-dispatch (a leaf skill run BY /pursue or /governor
2788
- * logs under the parent), slash-command entry, local-script bypass, and
2789
- * on-disk artifacts. The 2026-05-30 skill audit found 39/53 skills at zero
2790
- * direct invocations, yet only one was a genuine cut: the rest were
2791
- * measurement-invisible or discovery-limited. This analyst encodes that
2792
- * lesson as a multi-signal usage model so a cheap repeatable pass can keep
2793
- * the library honest, and so the expensive audit workflow's verdicts can
2794
- * GEPA-distill it toward agreement (see `gold/skill-verdicts.gold.jsonl`).
2795
- *
2796
- * Report-building (`buildSkillUsageReport`, an fs scan) is separated from
2797
- * finding emission (`SkillUsageAnalyst.analyze`, pure) so the slow scan runs
2798
- * once at the registry boundary and the rule logic stays unit-testable.
2799
- */
2800
-
2801
- type SkillKind = 'public' | 'private';
2802
- /** One skill's multi-signal usage + structure. All counts are deterministic. */
2803
- interface SkillUsageRecord {
2804
- name: string;
2805
- kind: SkillKind;
2806
- /** Absolute path to the skill's SKILL.md. */
2807
- path: string;
2808
- lines: number;
2809
- /** `"skill":"<name>"` Skill-tool invocations across the trace corpus. */
2810
- directInvocations: number;
2811
- /** `<command-name>/<name>` slash invocations across the trace corpus. */
2812
- slashInvocations: number;
2813
- /** Sibling skills whose SKILL.md dispatches to this one (`/<name>`). Proxy
2814
- * for orchestrated sub-dispatch the per-skill counter cannot see. */
2815
- inboundRefs: number;
2816
- /** On-disk artifacts attributable to the skill (e.g. `.evolve/<name>/**`). */
2817
- artifactCount: number;
2818
- /** Tangle-private reference count in the body (leak signal for public skills). */
2819
- tanglePrivateRefs: number;
2820
- hasReferencesDir: boolean;
2821
- hasEvalsDir: boolean;
2822
- /** Body mentions `skill-runs.jsonl` (visible to /reflect + /governor). */
2823
- logsRuns: boolean;
2824
- /** Description carries an explicit `Triggers:` clause / trigger phrases. */
2825
- hasTriggerPhrases: boolean;
2826
- }
2827
- interface SkillUsageReport {
2828
- generatedFromTraces: number;
2829
- records: SkillUsageRecord[];
2830
- }
2831
- interface SkillUsageScanConfig {
2832
- /** Dirs holding `*.jsonl` transcripts (Claude `~/.claude/projects`, Codex sessions). */
2833
- transcriptDirs: string[];
2834
- /** Skill roots to scan; each dir directly under `root` with a `SKILL.md` is a skill. */
2835
- skillRoots: {
2836
- root: string;
2837
- kind: SkillKind;
2838
- }[];
2839
- /** Roots scanned for `<root>/.evolve/<skill>` artifact dirs. */
2840
- artifactRoots?: string[];
2841
- /** Token-prefixed mappings: skill name → extra artifact subpaths under an artifactRoot
2842
- * (e.g. reflect → `.evolve/reflections`). Catches non-eponymous artifact dirs. */
2843
- artifactAliases?: Record<string, string[]>;
2844
- /** Cap files read per transcript dir (bounds a huge corpus); 0 = unbounded. */
2845
- maxTranscriptsPerDir?: number;
2846
- }
2847
- /** Scan the corpus + skill roots into a {@link SkillUsageReport}. Deterministic. */
2848
- declare function buildSkillUsageReport(config: SkillUsageScanConfig): SkillUsageReport;
2849
- /** Pure rule pass over a report → findings. Exported for direct/unit use. */
2850
- declare function emitSkillUsageFindings(report: SkillUsageReport, producedAt: string): AnalystFinding[];
2851
- declare class SkillUsageAnalyst implements Analyst<SkillUsageReport> {
2852
- readonly id = "skill-usage";
2853
- readonly description = "Deterministic multi-signal skill-usage analysis: flags dead skills, measurement-invisible (orchestrated) usage, discovery gaps, public-repo leaks, bloat, missing evals, and missing run-logging.";
2854
- readonly inputKind: "custom";
2855
- readonly cost: {
2856
- kind: "deterministic";
2857
- est_usd_per_run: number;
2858
- };
2859
- readonly version = "1.0.0";
2860
- analyze(input: SkillUsageReport, ctx: AnalystContext): Promise<AnalystFinding[]>;
2861
- }
2862
- declare const SKILL_USAGE_ANALYST: SkillUsageAnalyst;
2863
-
2864
- /**
2865
- * Forgiving pre-parse for analyst findings. Weak models routinely emit
2866
- * schema-correct content in an unusable wrapper — fenced ```json blocks, a
2867
- * single object where an array is expected, trailing commas. Measured: GPT-4o
2868
- * drops to 0% usable output purely from markdown-fence wrapping
2869
- * (arXiv:2605.02363). A five-line de-fence recovers most of it. This module is
2870
- * the de-fence/coerce step that runs BEFORE Zod, so a recoverable finding is
2871
- * repaired, not dropped.
2872
- *
2873
- * Pure + deterministic. No model, no network.
2874
- */
2875
- /** Strip a ```lang ... ``` (or bare ``` ... ```) code fence, if the string is one. */
2876
- declare function stripCodeFences(text: string): string;
2877
- /**
2878
- * Best-effort parse of a string into JSON. De-fences, drops trailing commas,
2879
- * then `JSON.parse`. Returns `undefined` (never throws) when unrecoverable.
2880
- */
2881
- declare function coerceJson(text: string): unknown;
2882
- /**
2883
- * Coerce arbitrary actor/structurer output into an array of candidate finding
2884
- * rows: a JSON string → parse; a single object → 1-element array; an array →
2885
- * as-is; anything else → []. Callers still run each row through Zod
2886
- * (`parseCanonicalRawFinding`) — this only fixes the SHAPE, never invents fields.
2887
- */
2888
- declare function coerceToFindingRows(raw: unknown): unknown[];
2889
-
2890
- type PolicyEditSchemaVersion = 'policy-edit/v1';
2891
- declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
2892
- type PolicyEditAxis = (typeof POLICY_EDIT_AXES)[number];
2893
- declare const POLICY_EDIT_TARGET_SURFACES: readonly ["prompt", "tool-contract", "runtime-config", "memory", "agent-profile", "code", "deployment"];
2894
- type PolicyEditTargetSurface = (typeof POLICY_EDIT_TARGET_SURFACES)[number];
2895
- type PolicyEditRisk = 'low' | 'medium' | 'high' | 'unknown';
2896
- type PolicyEditGainDirection = 'increase' | 'decrease';
2897
- type PolicyEditGainUnit = 'absolute' | 'relative' | 'percent' | 'score';
2898
- interface PolicyEditTarget {
2899
- surface: PolicyEditTargetSurface;
2900
- /** Stable path inside the target surface, for example `system-prompt:tools`
2901
- * or `budget.maxTurns`. */
2902
- path?: string;
2903
- /** Optional canonical deployment identity. Store the existing cell, not a
2904
- * local profile shape. */
2905
- agentProfileCell?: AgentProfileCell;
2906
- /** Human label when the path is not enough for a readable audit trail. */
2907
- label?: string;
2908
- }
2909
- type PolicyEditChange = {
2910
- kind: 'text';
2911
- mode: 'append' | 'prepend' | 'replace';
2912
- value: string;
2913
- /** Required when `mode === 'replace'`; exact match only. */
2914
- find?: string;
2915
- } | {
2916
- kind: 'json';
2917
- mode: 'set' | 'merge' | 'remove';
2918
- path: string;
2919
- value?: AgentProfileJson;
2920
- };
2921
- interface PolicyEditExpectedGain {
2922
- /** Metric this edit is expected to move, e.g. `holdout.composite`. */
2923
- metric: string;
2924
- direction: PolicyEditGainDirection;
2925
- /** Positive magnitude in the metric's native units. */
2926
- amount: number;
2927
- unit?: PolicyEditGainUnit;
2928
- rationale?: string;
2929
- }
2930
- interface PolicyEditSource {
2931
- findingIds: string[];
2932
- analystIds: string[];
2933
- evidenceRefs: EvidenceRef[];
2934
- /** Mirrors `AnalystFinding.derived_from_judge`; admission rejects it. */
2935
- derivedFromJudge?: boolean;
2936
- }
2937
- interface PolicyEdit {
2938
- schemaVersion: PolicyEditSchemaVersion;
2939
- editId: string;
2940
- axis: PolicyEditAxis;
2941
- target: PolicyEditTarget;
2942
- change: PolicyEditChange;
2943
- claim: string;
2944
- expectedGain: PolicyEditExpectedGain;
2945
- confidence: number;
2946
- risk: PolicyEditRisk;
2947
- source: PolicyEditSource;
2948
- rationale?: string;
2949
- validationPlan?: string;
2950
- metadata?: Record<string, unknown>;
2951
- }
2952
- declare const POLICY_EDIT_CANDIDATE_RECORD_SCHEMA: "tangle.policy-edit-candidate.v1";
2953
- /** JSON-safe attribution carried with a measured candidate and its scores. */
2954
- interface PolicyEditCandidateRecord {
2955
- schema: typeof POLICY_EDIT_CANDIDATE_RECORD_SCHEMA;
2956
- policyEdit: PolicyEdit;
2957
- }
2958
- type PolicyEditInit = Omit<PolicyEdit, 'schemaVersion' | 'editId'> & {
2959
- schemaVersion?: PolicyEditSchemaVersion;
2960
- editId?: string;
2961
- };
2962
- declare class PolicyEditValidationError extends ValidationError {
2963
- readonly path: string;
2964
- constructor(message: string, path?: string);
2965
- }
2966
- interface FindingToPolicyEditOptions {
2967
- expectedGain?: PolicyEditExpectedGain | ((finding: AnalystFinding) => PolicyEditExpectedGain | null | undefined);
2968
- risk?: PolicyEditRisk | ((finding: AnalystFinding) => PolicyEditRisk);
2969
- defaultAxis?: PolicyEditAxis;
2970
- defaultTargetSurface?: PolicyEditTargetSurface;
2971
- }
2972
- interface PolicyEditAdmissionOptions {
2973
- minScore?: number;
2974
- minExpectedGain?: number;
2975
- allowHighRisk?: boolean;
2976
- requireEvidence?: boolean;
2977
- }
2978
- interface PolicyEditAdmission {
2979
- edit: PolicyEdit;
2980
- decision: 'admit' | 'reject';
2981
- score: number;
2982
- reasons: string[];
2983
- }
2984
- declare function makePolicyEdit(init: PolicyEditInit): PolicyEdit;
2985
- declare function computePolicyEditId(edit: Omit<PolicyEdit, 'editId'> | PolicyEdit): string;
2986
- declare function validatePolicyEdit(input: unknown): PolicyEdit;
2987
- declare function makePolicyEditCandidateRecord(edit: PolicyEdit): PolicyEditCandidateRecord;
2988
- declare function validatePolicyEditCandidateRecord(input: unknown): PolicyEditCandidateRecord;
2989
- declare function isPolicyEdit(input: unknown): input is PolicyEdit;
2990
- declare function policyEditsFromFindings(findings: ReadonlyArray<AnalystFinding>, opts?: FindingToPolicyEditOptions): PolicyEdit[];
2991
- declare function policyEditFromFinding(finding: AnalystFinding, opts?: FindingToPolicyEditOptions): PolicyEdit | null;
2992
- declare function scorePolicyEditReadiness(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): number;
2993
- declare function admitPolicyEdit(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): PolicyEditAdmission;
2994
- declare function applyPolicyEditToSurface(surface: unknown, edit: PolicyEdit): unknown;
2995
-
2996
- /** DESCRIPTIVE predicate: does the finding cite at least one observable
2997
- * (span/event/artifact) evidence ref. Useful for ranking evidence quality or
2998
- * rendering — it is NOT the steer gate. Evidence presence is the WRONG
2999
- * discriminator for steering: a legitimate trace-analyst observation may cite
3000
- * nothing (it would be wrongly rejected), and a judge verdict may cite an
3001
- * artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate
3002
- * steering; use this only where "is this grounded in observable evidence" is the
3003
- * literal question. */
3004
- declare function isTraceObservable(finding: AnalystFinding): boolean;
3005
- /** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a
3006
- * finding), identified by provenance set at the lift site — independent of
3007
- * whatever evidence it cites. */
3008
- declare function isJudgeVerdict(finding: AnalystFinding): boolean;
3009
- /**
3010
- * THE steer firewall. Fail-loud guard for any path that admits analyst findings
3011
- * as STEERING input (the `f(trace)` role): rejects — naming the offenders — any
3012
- * finding whose provenance is a judge verdict, rather than let `J` leak into the
3013
- * loop. Returns the findings unchanged for chaining.
3014
- *
3015
- * Call this at the chokepoint where a detector that ALSO scores/gates has its
3016
- * findings turned into a steer (the judge-and-steer dual-role case). It keys on
3017
- * provenance, so it correctly admits evidence-less trace-analyst observations and
3018
- * correctly rejects an artifact-citing judge verdict — the cases an evidence
3019
- * check gets backwards.
3020
- *
3021
- * It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge
3022
- * whose output is laundered through a hand-built finding with no provenance flag
3023
- * is out of its reach — provenance must be honestly set at every judge→finding
3024
- * lift (today: createJudgeAdapter). That is why the integrity rule lives at the
3025
- * lift site, and why ProposeContext.judgeScores?: never is the complementary
3026
- * compile-time tripwire on the obvious direct channel.
3027
- */
3028
- declare function assertNoJudgeVerdict(findings: ReadonlyArray<AnalystFinding>, context?: string): ReadonlyArray<AnalystFinding>;
3029
-
3030
- /**
3031
- * `structureFindings` — the deferred structuring pass (DSPy TwoStepAdapter /
3032
- * HALO `synthesize_traces` analog). The agentic actor reasons FREE-FORM and
3033
- * emits a prose `report` (which any model does reliably); this separate, cheap
3034
- * call's ONLY job is to turn that report into `AnalystFinding[]`. Decoupling
3035
- * reasoning from structuring is what makes the SEMANTIC findings model-agnostic
3036
- * — the reasoning model never has to satisfy a strict typed-array contract
3037
- * while it diagnoses.
3038
- *
3039
- * Forgiving: the response runs through `coerceToFindingRows` (de-fence, lift
3040
- * single→array) before Zod, and on a zero-finding extraction from a substantive
3041
- * report it reasks ONCE with the schema restated. Returns a typed outcome so a
3042
- * legitimate "nothing to report" is distinguishable from a failed extraction
3043
- * (no silent empty).
3044
- */
3045
-
3046
- interface StructureFindingsOptions {
3047
- /** The actor's free-form diagnosis prose. */
3048
- report: string;
3049
- analystId: string;
3050
- /** Coarse classification stamped on every extracted finding. */
3051
- area: string;
3052
- model: string;
3053
- baseUrl: string;
3054
- apiKey?: string;
3055
- /** Optional ledger for direct use. */
3056
- costLedger?: CostLedgerHandle;
3057
- costPhase?: string;
3058
- costTags?: Record<string, string>;
3059
- maxTokens?: number;
3060
- signal?: AbortSignal;
3061
- /** Max reask attempts after a zero/invalid extraction. Default 1. */
3062
- maxReasks?: number;
3063
- /** Apply the caller's normal finding rules before a recovered row is lifted. */
3064
- processRow?: (row: RawAnalystFinding) => RawAnalystFinding | null;
3065
- /** Apply canonical multi-citation rules after any original callback. */
3066
- processCanonicalRow?: (row: CanonicalRawAnalystFinding) => CanonicalRawAnalystFinding | null;
3067
- /** Provenance copied onto every recovered finding. */
3068
- findingMetadata?: Record<string, unknown>;
3069
- /** Test seam: inject a fetch (no network in unit tests). */
3070
- fetchImpl?: LlmClientOptions['fetch'];
3071
- }
3072
- interface StructureFindingsResult {
3073
- findings: AnalystFinding[];
3074
- outcome: 'ok' | 'extraction_failed';
3075
- }
3076
- declare function structureFindings(opts: StructureFindingsOptions): Promise<StructureFindingsResult>;
3077
-
3078
- /**
3079
- * Pre-curated tool subsets for analyst kinds.
3080
- *
3081
- * The full trace-analyst tool set is seven functions. Most kinds only
3082
- * need three or four. Picking from named groups instead of importing
3083
- * the whole bundle keeps every kind's actor-context budget tight and
3084
- * makes "what can this analyst see?" obvious at registration time.
3085
- *
3086
- * Each function in the group keeps its full `name`/`description` from
3087
- * `buildTraceAnalystTools` — we filter, we don't re-implement.
3088
- */
3089
-
3090
- /** Named tool sets. Kinds pass `tools: TRACE_TOOL_GROUPS.failureForensics` etc. */
3091
- type TraceToolGroupName =
3092
- /** All seven tools. Use for open-ended discovery kinds. */
3093
- 'all'
3094
- /** Overview + paginated query + count. No deep reads. Cheap. */
3095
- | 'discovery'
3096
- /** Discovery + viewTrace + viewSpans. Deep-read but no regex search. */
3097
- | 'discoveryAndRead'
3098
- /** Discovery + search tools. For pattern-matching across many traces. */
3099
- | 'discoveryAndSearch'
3100
- /** Discovery + viewSpans + searchSpan. Targeted-span work after another kind narrows down. */
3101
- | 'targeted';
3102
- /**
3103
- * Build the tool set for a named group bound to a specific trace store.
3104
- *
3105
- * `all` returns every tool. Other groups filter `buildTraceAnalystTools`
3106
- * by name to the documented subset. An unrecognised group name throws —
3107
- * silently returning all tools would defeat the cost-control point.
3108
- */
3109
- declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): AxFunction[];
3110
-
3111
- export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type CanonicalRawAnalystFinding, CanonicalRawAnalystFindingSchema, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingToPolicyEditOptions, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, POLICY_EDIT_AXES, POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, POLICY_EDIT_TARGET_SURFACES, type PersistedFinding, type PolicyEdit, type PolicyEditAdmission, type PolicyEditAdmissionOptions, type PolicyEditAxis, type PolicyEditCandidateRecord, type PolicyEditChange, type PolicyEditExpectedGain, type PolicyEditGainDirection, type PolicyEditGainUnit, type PolicyEditInit, type PolicyEditRisk, type PolicyEditSchemaVersion, type PolicyEditSource, type PolicyEditTarget, type PolicyEditTargetSurface, PolicyEditValidationError, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts, admitPolicyEdit, applyPolicyEditToSurface, assertNoJudgeVerdict, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, computePolicyEditId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isJudgeVerdict, isPolicyEdit, isTraceObservable, liftSeverity, makeFinding, makePolicyEdit, makePolicyEditCandidateRecord, parseCanonicalRawFinding, parseFindingSubject, parseRawFinding, policyEditFromFinding, policyEditsFromFindings, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, scorePolicyEditReadiness, stripCodeFences, structureFindings, validatePolicyEdit, validatePolicyEditCandidateRecord };