@tangle-network/agent-eval 0.120.0 → 0.120.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/package.json +1 -1
  3. package/dist/analyst/index.d.ts +0 -3111
  4. package/dist/analyst/index.js +0 -403
  5. package/dist/analyst/index.js.map +0 -1
  6. package/dist/authenticity/index.d.ts +0 -161
  7. package/dist/authenticity/index.js +0 -215
  8. package/dist/authenticity/index.js.map +0 -1
  9. package/dist/belief-state/index.d.ts +0 -1301
  10. package/dist/belief-state/index.js +0 -2152
  11. package/dist/belief-state/index.js.map +0 -1
  12. package/dist/benchmarks/index.d.ts +0 -974
  13. package/dist/benchmarks/index.js +0 -60
  14. package/dist/benchmarks/index.js.map +0 -1
  15. package/dist/builder-eval/index.d.ts +0 -695
  16. package/dist/builder-eval/index.js +0 -366
  17. package/dist/builder-eval/index.js.map +0 -1
  18. package/dist/campaign/index.d.ts +0 -7454
  19. package/dist/campaign/index.js +0 -272
  20. package/dist/campaign/index.js.map +0 -1
  21. package/dist/chunk-32BZXMSO.js +0 -3878
  22. package/dist/chunk-32BZXMSO.js.map +0 -1
  23. package/dist/chunk-3A246TSA.js +0 -998
  24. package/dist/chunk-3A246TSA.js.map +0 -1
  25. package/dist/chunk-3RF76KTD.js +0 -84
  26. package/dist/chunk-3RF76KTD.js.map +0 -1
  27. package/dist/chunk-3YYRZDON.js +0 -45
  28. package/dist/chunk-3YYRZDON.js.map +0 -1
  29. package/dist/chunk-4I2E3LLO.js +0 -1030
  30. package/dist/chunk-4I2E3LLO.js.map +0 -1
  31. package/dist/chunk-ARU2PZFM.js +0 -312
  32. package/dist/chunk-ARU2PZFM.js.map +0 -1
  33. package/dist/chunk-BOD4O7OF.js +0 -40
  34. package/dist/chunk-BOD4O7OF.js.map +0 -1
  35. package/dist/chunk-DPZAEKA6.js +0 -880
  36. package/dist/chunk-DPZAEKA6.js.map +0 -1
  37. package/dist/chunk-DTJ6QUQB.js +0 -131
  38. package/dist/chunk-DTJ6QUQB.js.map +0 -1
  39. package/dist/chunk-GGE4NNQT.js +0 -65
  40. package/dist/chunk-GGE4NNQT.js.map +0 -1
  41. package/dist/chunk-H5UD2323.js +0 -286
  42. package/dist/chunk-H5UD2323.js.map +0 -1
  43. package/dist/chunk-HHWE3POT.js +0 -94
  44. package/dist/chunk-HHWE3POT.js.map +0 -1
  45. package/dist/chunk-HKUCJ437.js +0 -787
  46. package/dist/chunk-HKUCJ437.js.map +0 -1
  47. package/dist/chunk-JHCHEVET.js +0 -274
  48. package/dist/chunk-JHCHEVET.js.map +0 -1
  49. package/dist/chunk-JHOJHHU7.js +0 -867
  50. package/dist/chunk-JHOJHHU7.js.map +0 -1
  51. package/dist/chunk-JM2SKQMS.js +0 -750
  52. package/dist/chunk-JM2SKQMS.js.map +0 -1
  53. package/dist/chunk-JN2FCO5W.js +0 -7958
  54. package/dist/chunk-JN2FCO5W.js.map +0 -1
  55. package/dist/chunk-K4DBDHLK.js +0 -158
  56. package/dist/chunk-K4DBDHLK.js.map +0 -1
  57. package/dist/chunk-K6N6XJJX.js +0 -306
  58. package/dist/chunk-K6N6XJJX.js.map +0 -1
  59. package/dist/chunk-MA6HLL3S.js +0 -65
  60. package/dist/chunk-MA6HLL3S.js.map +0 -1
  61. package/dist/chunk-MAZ26DC7.js +0 -99
  62. package/dist/chunk-MAZ26DC7.js.map +0 -1
  63. package/dist/chunk-MOXWMGPC.js +0 -577
  64. package/dist/chunk-MOXWMGPC.js.map +0 -1
  65. package/dist/chunk-NJC7U437.js +0 -626
  66. package/dist/chunk-NJC7U437.js.map +0 -1
  67. package/dist/chunk-NPCTHQIO.js +0 -91
  68. package/dist/chunk-NPCTHQIO.js.map +0 -1
  69. package/dist/chunk-ONWEPEDO.js +0 -57
  70. package/dist/chunk-ONWEPEDO.js.map +0 -1
  71. package/dist/chunk-OYZAPX5G.js +0 -1526
  72. package/dist/chunk-OYZAPX5G.js.map +0 -1
  73. package/dist/chunk-PC4UYEBM.js +0 -166
  74. package/dist/chunk-PC4UYEBM.js.map +0 -1
  75. package/dist/chunk-PICTDURQ.js +0 -766
  76. package/dist/chunk-PICTDURQ.js.map +0 -1
  77. package/dist/chunk-PJQFMIOX.js +0 -1182
  78. package/dist/chunk-PJQFMIOX.js.map +0 -1
  79. package/dist/chunk-PXD6ZFNY.js +0 -1107
  80. package/dist/chunk-PXD6ZFNY.js.map +0 -1
  81. package/dist/chunk-PXE2VKMX.js +0 -140
  82. package/dist/chunk-PXE2VKMX.js.map +0 -1
  83. package/dist/chunk-PZ5AY32C.js +0 -10
  84. package/dist/chunk-PZ5AY32C.js.map +0 -1
  85. package/dist/chunk-QBRSJK47.js +0 -622
  86. package/dist/chunk-QBRSJK47.js.map +0 -1
  87. package/dist/chunk-QWMPPZ3X.js +0 -550
  88. package/dist/chunk-QWMPPZ3X.js.map +0 -1
  89. package/dist/chunk-S3UZOQ5Y.js +0 -328
  90. package/dist/chunk-S3UZOQ5Y.js.map +0 -1
  91. package/dist/chunk-S5TT5R3L.js +0 -2668
  92. package/dist/chunk-S5TT5R3L.js.map +0 -1
  93. package/dist/chunk-T4SQEITX.js +0 -95
  94. package/dist/chunk-T4SQEITX.js.map +0 -1
  95. package/dist/chunk-TT4KNT67.js +0 -124
  96. package/dist/chunk-TT4KNT67.js.map +0 -1
  97. package/dist/chunk-U5CHZ5M3.js +0 -357
  98. package/dist/chunk-U5CHZ5M3.js.map +0 -1
  99. package/dist/chunk-ULOKLHIQ.js +0 -1937
  100. package/dist/chunk-ULOKLHIQ.js.map +0 -1
  101. package/dist/chunk-VI2UW6B6.js +0 -162
  102. package/dist/chunk-VI2UW6B6.js.map +0 -1
  103. package/dist/chunk-VQMK5FMP.js +0 -247
  104. package/dist/chunk-VQMK5FMP.js.map +0 -1
  105. package/dist/chunk-VSMTAMNK.js +0 -53
  106. package/dist/chunk-VSMTAMNK.js.map +0 -1
  107. package/dist/chunk-VZSRQ272.js +0 -149
  108. package/dist/chunk-VZSRQ272.js.map +0 -1
  109. package/dist/chunk-WW2A73HW.js +0 -159
  110. package/dist/chunk-WW2A73HW.js.map +0 -1
  111. package/dist/chunk-X4UCIOTZ.js +0 -136
  112. package/dist/chunk-X4UCIOTZ.js.map +0 -1
  113. package/dist/chunk-XDIRG3TO.js +0 -1266
  114. package/dist/chunk-XDIRG3TO.js.map +0 -1
  115. package/dist/chunk-XJYR7XFV.js +0 -317
  116. package/dist/chunk-XJYR7XFV.js.map +0 -1
  117. package/dist/chunk-ZET2UAYW.js +0 -89
  118. package/dist/chunk-ZET2UAYW.js.map +0 -1
  119. package/dist/chunk-ZZUXHH3R.js +0 -99
  120. package/dist/chunk-ZZUXHH3R.js.map +0 -1
  121. package/dist/cli.d.ts +0 -1
  122. package/dist/cli.js +0 -112
  123. package/dist/cli.js.map +0 -1
  124. package/dist/contract/index.d.ts +0 -4969
  125. package/dist/contract/index.js +0 -1653
  126. package/dist/contract/index.js.map +0 -1
  127. package/dist/control.d.ts +0 -1013
  128. package/dist/control.js +0 -34
  129. package/dist/control.js.map +0 -1
  130. package/dist/fuzz.d.ts +0 -759
  131. package/dist/fuzz.js +0 -714
  132. package/dist/fuzz.js.map +0 -1
  133. package/dist/hosted/index.d.ts +0 -730
  134. package/dist/hosted/index.js +0 -14
  135. package/dist/hosted/index.js.map +0 -1
  136. package/dist/index.d.ts +0 -16780
  137. package/dist/index.js +0 -12168
  138. package/dist/index.js.map +0 -1
  139. package/dist/matrix/index.d.ts +0 -155
  140. package/dist/matrix/index.js +0 -8
  141. package/dist/matrix/index.js.map +0 -1
  142. package/dist/meta-eval/index.d.ts +0 -1030
  143. package/dist/meta-eval/index.js +0 -417
  144. package/dist/meta-eval/index.js.map +0 -1
  145. package/dist/multishot/index.d.ts +0 -579
  146. package/dist/multishot/index.js +0 -589
  147. package/dist/multishot/index.js.map +0 -1
  148. package/dist/openapi.json +0 -992
  149. package/dist/pipelines/index.d.ts +0 -567
  150. package/dist/pipelines/index.js +0 -515
  151. package/dist/pipelines/index.js.map +0 -1
  152. package/dist/reporting.d.ts +0 -1277
  153. package/dist/reporting.js +0 -48
  154. package/dist/reporting.js.map +0 -1
  155. package/dist/rl.d.ts +0 -4092
  156. package/dist/rl.js +0 -1724
  157. package/dist/rl.js.map +0 -1
  158. package/dist/run-campaign-HNFPJET4.js +0 -14
  159. package/dist/run-campaign-HNFPJET4.js.map +0 -1
  160. package/dist/storyboard/index.d.ts +0 -279
  161. package/dist/storyboard/index.js +0 -767
  162. package/dist/storyboard/index.js.map +0 -1
  163. package/dist/trace-attributes.d.ts +0 -52
  164. package/dist/trace-attributes.js +0 -62
  165. package/dist/trace-attributes.js.map +0 -1
  166. package/dist/traces.d.ts +0 -2343
  167. package/dist/traces.js +0 -249
  168. package/dist/traces.js.map +0 -1
  169. package/dist/wire/index.d.ts +0 -1252
  170. package/dist/wire/index.js +0 -81
  171. package/dist/wire/index.js.map +0 -1
@@ -1,1252 +0,0 @@
1
- import { z } from 'zod';
2
- import { OpenAPIObject } from 'openapi3-ts/oas31';
3
- import * as hono_types from 'hono/types';
4
- import { ServerType } from '@hono/node-server';
5
- import { Hono } from 'hono';
6
-
7
- type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
8
- interface CostUsage {
9
- inputTokens: number;
10
- /** Includes reasoning tokens when the provider bills them as output. */
11
- outputTokens: number;
12
- /** Reasoning-token subset of outputTokens, when reported. */
13
- reasoningTokens?: number;
14
- /** Prompt tokens served from a provider cache. */
15
- cachedTokens?: number;
16
- /** Prompt tokens written into a provider cache. */
17
- cacheWriteTokens?: number;
18
- }
19
- interface CostCallBase {
20
- callId: string;
21
- channel: CostChannel;
22
- phase: string;
23
- actor: string;
24
- model: string;
25
- maximumCostUsd?: number;
26
- tags?: Record<string, string>;
27
- timestamp: number;
28
- }
29
- interface CostReceipt extends CostCallBase, CostUsage {
30
- status: 'settled';
31
- costUsd: number;
32
- costUnknown: boolean;
33
- usageUnknown?: boolean;
34
- pricing?: {
35
- inputUsdPerThousand: number;
36
- outputUsdPerThousand: number;
37
- };
38
- actualCostUsd?: number;
39
- error?: string;
40
- }
41
- interface CostReceiptInput extends CostUsage {
42
- model: string;
43
- actualCostUsd?: number;
44
- costUnknown?: boolean;
45
- usageUnknown?: boolean;
46
- }
47
- type MaximumCharge = {
48
- externallyEnforcedMaximumUsd: number;
49
- } | ({
50
- model: string;
51
- } & CostUsage);
52
- interface RunPaidCallInput<T> {
53
- callId?: string;
54
- channel: CostChannel;
55
- phase: string;
56
- actor: string;
57
- /** Used before a provider receipt exists and on failures without one. */
58
- model?: string;
59
- tags?: Record<string, string>;
60
- signal?: AbortSignal;
61
- /** Provider-enforced dollar maximum, or maximum priced token usage. Required when capped. */
62
- maximumCharge?: MaximumCharge;
63
- /** `callId` can be forwarded as the provider's idempotency key. */
64
- execute(signal: AbortSignal, callId: string): Promise<T>;
65
- receipt(value: T): CostReceiptInput;
66
- receiptFromError?(error: Error): CostReceiptInput | undefined;
67
- }
68
- type PaidCallResult<T> = {
69
- succeeded: true;
70
- callId: string;
71
- value: T;
72
- receipt: CostReceipt;
73
- } | {
74
- succeeded: false;
75
- callId?: string;
76
- error: Error;
77
- receipt?: CostReceipt;
78
- };
79
- interface ChannelRollup {
80
- channel: CostChannel;
81
- calls: number;
82
- inputTokens: number;
83
- outputTokens: number;
84
- reasoningTokens?: number;
85
- cachedTokens: number;
86
- cacheWriteTokens?: number;
87
- costUsd: number;
88
- unpricedCalls: number;
89
- unknownUsageCalls: number;
90
- }
91
- interface CostLedgerSummary {
92
- totalCalls: number;
93
- pendingCalls: number;
94
- unresolvedCalls: number;
95
- reservedCostUsd: number;
96
- inputTokens: number;
97
- outputTokens: number;
98
- reasoningTokens?: number;
99
- cachedTokens: number;
100
- cacheWriteTokens?: number;
101
- totalCostUsd: number;
102
- byChannel: ChannelRollup[];
103
- unpricedModels: string[];
104
- fullyPriced: boolean;
105
- usageComplete: boolean;
106
- accountingComplete: boolean;
107
- incompleteReasons: string[];
108
- }
109
- interface CostLedgerFilter {
110
- channel?: CostChannel;
111
- phase?: string;
112
- tags?: Record<string, string>;
113
- }
114
- interface CostLedgerWaitOptions {
115
- /** Maximum time to wait for active provider calls. Default 5 seconds. */
116
- timeoutMs?: number;
117
- }
118
- /** Append-only storage. `append` must atomically reject stale revisions. */
119
- interface CostLedgerPersistence {
120
- read(): {
121
- revision: string;
122
- events: string;
123
- };
124
- append(expectedRevision: string, event: string): string | undefined;
125
- }
126
- interface CostLedgerOptions {
127
- costCeilingUsd?: number;
128
- persistence?: CostLedgerPersistence;
129
- /** Import already-settled receipts without admitting new paid work. */
130
- receipts?: readonly CostReceipt[];
131
- }
132
- /** Run-wide paid-call admission, durable call state, receipts, and summaries. */
133
- declare class CostLedger {
134
- private readonly records;
135
- private readonly activeCallIds;
136
- private readonly lateCallIds;
137
- private readonly idleWaiters;
138
- private completedTasks;
139
- private revision;
140
- private costLimitPersisted;
141
- readonly costCeilingUsd?: number;
142
- private readonly persistence?;
143
- constructor(input?: number | CostLedgerOptions);
144
- runPaidCall<T>(input: RunPaidCallInput<T>): Promise<PaidCallResult<T>>;
145
- /** Wait until every call started by this ledger has produced a durable outcome. */
146
- waitForIdle(options?: CostLedgerWaitOptions): Promise<boolean>;
147
- /** Settle a call left pending by a crashed process after reconciling with the provider. */
148
- reconcile(callId: string, observed: CostReceiptInput, options?: {
149
- error?: string;
150
- }): CostReceipt;
151
- list(filter?: CostLedgerFilter): CostReceipt[];
152
- summary(filter?: CostLedgerFilter): CostLedgerSummary;
153
- markCompleted(count?: number): void;
154
- costPerCompletedTask(): number | null;
155
- private execute;
156
- private captureLateOutcome;
157
- private releaseActiveCall;
158
- private commitOutcome;
159
- private captureFailure;
160
- private commitReceipt;
161
- private resolveMaximum;
162
- private hasIncompleteSettledCall;
163
- private appendRecord;
164
- private ensureCostLimitPersisted;
165
- private appendEvent;
166
- }
167
- /** Public callback surface for a shared cost ledger.
168
- *
169
- * Declaration bundles may expose this type through multiple package subpaths.
170
- * Keeping callback contracts structural lets those subpaths compose while the
171
- * concrete {@link CostLedger} retains its private durable state.
172
- */
173
- type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'waitForIdle'>> & Partial<Pick<CostLedger, 'waitForIdle'>>;
174
-
175
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
176
- interface BudgetSpec {
177
- tokens?: number;
178
- wallMs?: number;
179
- calls?: number;
180
- usd?: number;
181
- }
182
- interface RunOutcome {
183
- score?: number;
184
- pass?: boolean;
185
- failureClass?: FailureClass;
186
- notes?: string;
187
- }
188
- /**
189
- * Layer — optional classification in a nested build workflow.
190
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
191
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
192
- * `app-runtime`: a run of the generated agent against a domain scenario.
193
- * `meta`: any meta-eval (judge replay, correlation analysis).
194
- */
195
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
196
- interface Run {
197
- runId: string;
198
- /**
199
- * Stable identifier of the scenario being executed.
200
- *
201
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
202
- * input WITHOUT this field, substituting a sensible default
203
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
204
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
205
- * keeps the persisted shape unambiguous for downstream filters + aggregations
206
- * while removing the boilerplate of inventing placeholder ids at the call site.
207
- */
208
- scenarioId: string;
209
- variantId?: string;
210
- datasetVersion?: string;
211
- /** Git SHA of agent code at run time. */
212
- codeSha?: string;
213
- /** Hash of the prompt template + any system prompt. */
214
- promptSha?: string;
215
- /** Model id + date + system-prompt hash, concatenated. */
216
- modelFingerprint?: string;
217
- seed?: number;
218
- /** Arbitrary environment markers (shell, docker version, tz). */
219
- envFingerprint?: Record<string, string>;
220
- /** Version of the redaction rules applied to this run. */
221
- redactionVersion?: string;
222
- /** Parent run in a nested build workflow. A builder run's children are
223
- * app-build runs; those children are app-runtime runs. */
224
- parentRunId?: string;
225
- /** Stable project identifier — groups runs across chats + sessions. */
226
- projectId?: string;
227
- /** Chat/conversation identifier within a project. */
228
- chatId?: string;
229
- /** Layer classification — hint for aggregation; not enforced. */
230
- layer?: RunLayer;
231
- startedAt: number;
232
- endedAt?: number;
233
- status: RunStatus;
234
- outcome?: RunOutcome;
235
- budget?: BudgetSpec;
236
- /** Free-form labels for downstream grouping. */
237
- tags?: Record<string, string>;
238
- }
239
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
240
- type SpanStatus = 'ok' | 'error';
241
- interface SpanBase {
242
- spanId: string;
243
- parentSpanId?: string;
244
- runId: string;
245
- kind: SpanKind;
246
- name: string;
247
- startedAt: number;
248
- endedAt?: number;
249
- status?: SpanStatus;
250
- error?: string;
251
- /** Anything not covered by typed fields. Kept deliberately free-form. */
252
- attributes?: Record<string, unknown>;
253
- }
254
- interface Message {
255
- role: 'system' | 'user' | 'assistant' | 'tool';
256
- content: string;
257
- tokens?: number;
258
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
259
- images?: Array<{
260
- artifactId?: string;
261
- url?: string;
262
- mime?: string;
263
- }>;
264
- }
265
- interface LlmSpan extends SpanBase {
266
- kind: 'llm';
267
- model: string;
268
- messages: Message[];
269
- output?: string;
270
- inputTokens?: number;
271
- /** All generated tokens, including the reasoning subset when present. */
272
- outputTokens?: number;
273
- cachedTokens?: number;
274
- cacheWriteTokens?: number;
275
- /** Reasoning-token subset of `outputTokens`. */
276
- reasoningTokens?: number;
277
- costUsd?: number;
278
- finishReason?: string;
279
- }
280
- interface ToolSpan extends SpanBase {
281
- kind: 'tool';
282
- toolName: string;
283
- args: unknown;
284
- /** False when the source observed the call but did not capture its arguments. */
285
- argsCaptured?: boolean;
286
- result?: unknown;
287
- latencyMs?: number;
288
- }
289
- interface RetrievalSpan extends SpanBase {
290
- kind: 'retrieval';
291
- query: string;
292
- hits: Array<{
293
- docId: string;
294
- score: number;
295
- content?: string;
296
- }>;
297
- }
298
- interface JudgeSpan extends SpanBase {
299
- kind: 'judge';
300
- judgeId: string;
301
- /** Span this judgment applies to. */
302
- targetSpanId: string;
303
- dimension: string;
304
- /** Numeric score (free-range; interpretation up to the judge). */
305
- score: number;
306
- rationale?: string;
307
- evidence?: string;
308
- }
309
- interface SandboxSpan extends SpanBase {
310
- kind: 'sandbox';
311
- image?: string;
312
- command?: string;
313
- exitCode?: number;
314
- testsTotal?: number;
315
- testsPassed?: number;
316
- stdoutHash?: string;
317
- stderrHash?: string;
318
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
319
- wallMs?: number;
320
- }
321
- interface GenericSpan extends SpanBase {
322
- kind: 'agent' | 'custom';
323
- }
324
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
325
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
326
- interface TraceEvent$1 {
327
- eventId: string;
328
- runId: string;
329
- spanId?: string;
330
- kind: EventKind;
331
- timestamp: number;
332
- payload: Record<string, unknown>;
333
- }
334
- interface BudgetLedgerEntry {
335
- runId: string;
336
- dimension: keyof BudgetSpec;
337
- limit: number;
338
- consumed: number;
339
- remaining: number;
340
- timestamp: number;
341
- breached: boolean;
342
- /** Span that triggered this entry, if any. */
343
- spanId?: string;
344
- }
345
- interface Artifact {
346
- artifactId: string;
347
- runId: string;
348
- spanId?: string;
349
- contentType: string;
350
- sizeBytes: number;
351
- /** sha256 in hex. */
352
- hash: string;
353
- /** External storage URL (R2, S3, filesystem path). */
354
- storageUrl?: string;
355
- /** Inline content for small blobs — keep under ~64KB. */
356
- inlineContent?: string;
357
- }
358
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
359
-
360
- interface RunFilter {
361
- scenarioId?: string;
362
- variantId?: string;
363
- status?: RunStatus;
364
- since?: number;
365
- until?: number;
366
- tag?: {
367
- key: string;
368
- value: string;
369
- };
370
- parentRunId?: string;
371
- projectId?: string;
372
- chatId?: string;
373
- layer?: RunLayer;
374
- }
375
- interface SpanFilter {
376
- runId?: string;
377
- parentSpanId?: string;
378
- kind?: SpanKind;
379
- name?: string;
380
- toolName?: string;
381
- judgeId?: string;
382
- since?: number;
383
- until?: number;
384
- }
385
- interface EventFilter {
386
- runId?: string;
387
- spanId?: string;
388
- kind?: EventKind;
389
- since?: number;
390
- until?: number;
391
- }
392
- interface TraceStore {
393
- appendRun(run: Run): Promise<void>;
394
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
395
- appendSpan(span: Span): Promise<void>;
396
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
397
- appendEvent(event: TraceEvent$1): Promise<void>;
398
- appendArtifact(artifact: Artifact): Promise<void>;
399
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
400
- getRun(runId: string): Promise<Run | undefined>;
401
- listRuns(filter?: RunFilter): Promise<Run[]>;
402
- spans(filter?: SpanFilter): Promise<Span[]>;
403
- events(filter?: EventFilter): Promise<TraceEvent$1[]>;
404
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
405
- artifacts(runId: string): Promise<Artifact[]>;
406
- }
407
-
408
- /**
409
- * Policy-based agent control runtime.
410
- *
411
- * This is the minimal reusable loop behind driver-agent patterns:
412
- *
413
- * observe state -> validate -> decide next action -> act -> observe -> ...
414
- *
415
- * It deliberately does not model named "topologies". Direct execution,
416
- * critic/revise, driver intervention, specialist calls, and human escalation
417
- * are all just actions chosen by the control policy.
418
- */
419
-
420
- type ControlSeverity = 'info' | 'warning' | 'error' | 'critical';
421
- interface ControlEvalResult {
422
- /** Stable validator or judge id. */
423
- id: string;
424
- /** Whether this check passed. */
425
- passed: boolean;
426
- /** Optional normalized score. 1 = best, 0 = worst. */
427
- score?: number;
428
- /** Objective validators should usually be "error" or "critical" when failed. */
429
- severity?: ControlSeverity;
430
- /** Human-readable result. */
431
- detail?: string;
432
- /** Small evidence string or pointer. Avoid large payloads. */
433
- evidence?: string;
434
- /** True when the result came from deterministic state, not LLM judgment. */
435
- objective?: boolean;
436
- /** Structured details for downstream control policies and reports. */
437
- metadata?: Record<string, unknown>;
438
- }
439
-
440
- /**
441
- * Dataset — versioned, sliceable, content-hashed scenario collection.
442
- *
443
- * Scenarios stop being ephemeral arrays and become first-class
444
- * artifacts. Every Dataset carries:
445
- * - content hash (sha256 over canonicalized scenario array)
446
- * - provenance (contributor, createdAt, sourceUrl)
447
- * - split labels (train | dev | test | holdout)
448
- * - difficulty tiers (easy | medium | hard | extreme)
449
- * - tags (free-form, per-scenario)
450
- *
451
- * `Dataset.slice({ difficulty, split, holdout, seed })` returns a
452
- * deterministic, reproducible subset. Holdout slices are locked: you
453
- * can read them but `mutate` throws, which prevents "oh I'll just
454
- * tweak that one scenario" contamination drift.
455
- */
456
- type DatasetSplit = 'train' | 'dev' | 'test' | 'holdout';
457
-
458
- type FeedbackArtifactType = 'text' | 'code' | 'plan' | 'research' | 'action' | 'ui' | 'decision' | 'data' | 'other';
459
- type FeedbackLabelSource = 'user' | 'judge' | 'environment' | 'metric' | 'policy' | 'system';
460
- type FeedbackLabelKind = 'approve' | 'reject' | 'select' | 'edit' | 'rank' | 'rate' | 'comment' | 'metric_outcome' | 'policy_block' | 'revision_request';
461
- type FeedbackSeverity = 'info' | 'warning' | 'error' | 'critical';
462
- interface FeedbackTask {
463
- intent: string;
464
- context?: unknown;
465
- }
466
- interface ProposedSideEffect {
467
- type: string;
468
- risk?: 'low' | 'medium' | 'high';
469
- costUsd?: number;
470
- externalSideEffect?: boolean;
471
- requiresApproval?: boolean;
472
- metadata?: Record<string, unknown>;
473
- }
474
- interface FeedbackLabel {
475
- id?: string;
476
- source: FeedbackLabelSource;
477
- kind: FeedbackLabelKind;
478
- value: unknown;
479
- reason?: string;
480
- severity?: FeedbackSeverity;
481
- createdAt: string;
482
- metadata?: Record<string, unknown>;
483
- }
484
- interface FeedbackAttempt {
485
- id: string;
486
- stepIndex: number;
487
- artifactType: FeedbackArtifactType;
488
- artifact: unknown;
489
- options?: unknown[];
490
- proposedAction?: ProposedSideEffect;
491
- evals?: ControlEvalResult[];
492
- feedback?: FeedbackLabel[];
493
- createdAt: string;
494
- metadata?: Record<string, unknown>;
495
- }
496
- interface FeedbackOutcome {
497
- success?: boolean;
498
- score?: number;
499
- metrics?: Record<string, number>;
500
- costUsd?: number;
501
- detail?: string;
502
- observedAt?: string;
503
- metadata?: Record<string, unknown>;
504
- }
505
- interface FeedbackTrajectory$1 {
506
- id: string;
507
- projectId?: string;
508
- scenarioId?: string;
509
- task: FeedbackTask;
510
- attempts: FeedbackAttempt[];
511
- labels: FeedbackLabel[];
512
- outcome?: FeedbackOutcome;
513
- split?: DatasetSplit;
514
- tags?: Record<string, string>;
515
- createdAt: string;
516
- updatedAt?: string;
517
- metadata?: Record<string, unknown>;
518
- }
519
- interface FeedbackTrajectoryStore {
520
- save(trajectory: FeedbackTrajectory$1): Promise<void>;
521
- get(id: string): Promise<FeedbackTrajectory$1 | null>;
522
- list(filter?: FeedbackTrajectoryFilter): Promise<FeedbackTrajectory$1[]>;
523
- appendAttempt(id: string, attempt: FeedbackAttempt): Promise<FeedbackTrajectory$1>;
524
- appendLabel(id: string, label: FeedbackLabel, attemptId?: string): Promise<FeedbackTrajectory$1>;
525
- }
526
- interface FeedbackTrajectoryFilter {
527
- projectId?: string;
528
- scenarioId?: string;
529
- split?: DatasetSplit;
530
- tag?: [string, string];
531
- }
532
-
533
- /**
534
- * RawProviderSink — first-class persistence for the actual HTTP-level
535
- * request/response bodies of every LLM provider call.
536
- *
537
- * Why this is a separate sink from the structured `LlmSpan`:
538
- *
539
- * - `LlmSpan` records the *intent* — model name, messages, output text,
540
- * usage. It's what dashboards read; it's NOT enough for forensics.
541
- * - When a downstream consumer reports "the verifier used the wrong route"
542
- * or "tokens look right but reasoning was missing," the only way to
543
- * answer is the raw HTTP body. Span fields can lie (a proxy can echo
544
- * a different `model` value than what actually answered); the raw
545
- * response is ground truth.
546
- *
547
- * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
548
- * matrix runner / BuilderSession sets it up automatically) and every
549
- * request, response, and error is recorded — including retries, with the
550
- * attempt index attached so a flaky call's full event chain is recoverable.
551
- *
552
- * Redaction is enforced at sink time. The default redactor strips
553
- * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
554
- * payload field whose key matches `apiKey | api_key | bearer | password |
555
- * secret | token` (case-insensitive). Override via the sink constructor or
556
- * the per-call `redactor`. The `redactedFields` array on the persisted
557
- * event lets a reviewer see what was stripped without exposing the values.
558
- */
559
- type RawProviderDirection = 'request' | 'response' | 'error';
560
- interface RawProviderEvent {
561
- /** Stable id. Generated by the sink if omitted. */
562
- eventId: string;
563
- /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
564
- runId?: string;
565
- spanId?: string;
566
- /**
567
- * Logical provider name. Free-form so callers can use whatever id matches
568
- * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
569
- * omitted, derived from `baseUrl` in `LlmClientOptions`.
570
- */
571
- provider: string;
572
- model: string;
573
- /** Endpoint path, e.g. `'/v1/chat/completions'`. */
574
- endpoint: string;
575
- /** Base URL used for the call (already-normalised — no trailing slash). */
576
- baseUrl: string;
577
- /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
578
- attemptIndex: number;
579
- direction: RawProviderDirection;
580
- /** Unix ms. */
581
- timestamp: number;
582
- /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
583
- durationMs?: number;
584
- statusCode?: number;
585
- requestHeaders?: Record<string, string>;
586
- requestBody?: unknown;
587
- responseHeaders?: Record<string, string>;
588
- responseBody?: unknown;
589
- /** Set on `direction: 'error'` events. */
590
- errorMessage?: string;
591
- /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
592
- redactedFields: string[];
593
- }
594
- interface RawProviderSinkFilter {
595
- runId?: string;
596
- spanId?: string;
597
- direction?: RawProviderDirection;
598
- attemptIndex?: number;
599
- }
600
- interface RawProviderSink {
601
- record(event: RawProviderEvent): Promise<void>;
602
- /** Optional listing — implementations that durably persist (file, db) should support this. */
603
- list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
604
- /** Optional teardown for backed implementations. */
605
- close?(): Promise<void>;
606
- }
607
- type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
608
-
609
- /**
610
- * LLM client with graceful degrade.
611
- *
612
- * OpenAI-compatible `/v1/chat/completions` client with:
613
- * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
614
- * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
615
- * - Graceful json_schema → json_object degrade on 400 with schema-reject body.
616
- * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
617
- * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
618
- * directly, cli-bridge subscriptions, and any router that speaks the spec.
619
- *
620
- * Usage:
621
- * const { value, result } = await callLlmJson<MyType>(
622
- * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
623
- * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
624
- * )
625
- *
626
- * This is THE llm-calling seam for agent-eval primitives that need structured
627
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
628
- * that need free-form text use `callLlm` and parse output themselves.
629
- */
630
-
631
- interface LlmClientOptions {
632
- /** Base URL (without trailing slash). Must end at the `/v1` prefix. */
633
- baseUrl?: string;
634
- /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
635
- apiKey?: string;
636
- bearer?: string;
637
- /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
638
- authHeader?: {
639
- name: string;
640
- value: string;
641
- };
642
- /** Stable provider idempotency key, reused across retries of this logical call. */
643
- idempotencyKey?: string;
644
- /** Default timeout in ms. Per-call can override. */
645
- defaultTimeoutMs?: number;
646
- /**
647
- * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
648
- * each attempt's per-attempt timeout controller, so aborting it cancels
649
- * the in-flight fetch. A caller abort is FATAL: it is not retried even
650
- * though an AbortError otherwise matches the transient patterns.
651
- */
652
- signal?: AbortSignal;
653
- /**
654
- * Cross-attempt wall-clock budget in ms, measured from the first attempt.
655
- * Before launching each attempt the loop checks the remaining budget and
656
- * stops retrying once it is exhausted, rather than waiting the full
657
- * per-attempt timeout on every retry. Bounds total time independent of
658
- * total attempts × `timeoutMs`.
659
- */
660
- deadlineMs?: number;
661
- /** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
662
- maxRetries?: number;
663
- /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
664
- fetch?: typeof fetch;
665
- /**
666
- * Optional raw HTTP capture sink. When provided, every request, response,
667
- * and error (across all retry attempts) is recorded to the sink, with auth
668
- * headers and credential-shaped body fields redacted by default. This is
669
- * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
670
- * raw events record what actually crossed the wire.
671
- */
672
- rawSink?: RawProviderSink;
673
- /**
674
- * Logical provider id attached to raw events. When omitted, derived from
675
- * `baseUrl` via `providerFromBaseUrl`.
676
- */
677
- provider?: string;
678
- /** Trace context attached to raw events; populated by emitter-aware callers. */
679
- traceContext?: {
680
- runId?: string;
681
- spanId?: string;
682
- };
683
- /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
684
- redactor?: ProviderRedactor;
685
- }
686
-
687
- declare const RubricDimensionSchema: z.ZodObject<{
688
- id: z.ZodString;
689
- description: z.ZodString;
690
- weight: z.ZodDefault<z.ZodNumber>;
691
- min: z.ZodDefault<z.ZodNumber>;
692
- max: z.ZodDefault<z.ZodNumber>;
693
- }, z.core.$strip>;
694
- declare const FailureModeSchema: z.ZodObject<{
695
- id: z.ZodString;
696
- description: z.ZodString;
697
- }, z.core.$strip>;
698
- declare const RubricSchema: z.ZodObject<{
699
- name: z.ZodString;
700
- description: z.ZodString;
701
- systemPrompt: z.ZodString;
702
- dimensions: z.ZodArray<z.ZodObject<{
703
- id: z.ZodString;
704
- description: z.ZodString;
705
- weight: z.ZodDefault<z.ZodNumber>;
706
- min: z.ZodDefault<z.ZodNumber>;
707
- max: z.ZodDefault<z.ZodNumber>;
708
- }, z.core.$strip>>;
709
- failureModes: z.ZodDefault<z.ZodArray<z.ZodObject<{
710
- id: z.ZodString;
711
- description: z.ZodString;
712
- }, z.core.$strip>>>;
713
- wins: z.ZodDefault<z.ZodArray<z.ZodObject<{
714
- id: z.ZodString;
715
- description: z.ZodString;
716
- }, z.core.$strip>>>;
717
- }, z.core.$strip>;
718
- declare const JudgeRequestSchema: z.ZodObject<{
719
- rubricName: z.ZodOptional<z.ZodString>;
720
- rubric: z.ZodOptional<z.ZodObject<{
721
- name: z.ZodString;
722
- description: z.ZodString;
723
- systemPrompt: z.ZodString;
724
- dimensions: z.ZodArray<z.ZodObject<{
725
- id: z.ZodString;
726
- description: z.ZodString;
727
- weight: z.ZodDefault<z.ZodNumber>;
728
- min: z.ZodDefault<z.ZodNumber>;
729
- max: z.ZodDefault<z.ZodNumber>;
730
- }, z.core.$strip>>;
731
- failureModes: z.ZodDefault<z.ZodArray<z.ZodObject<{
732
- id: z.ZodString;
733
- description: z.ZodString;
734
- }, z.core.$strip>>>;
735
- wins: z.ZodDefault<z.ZodArray<z.ZodObject<{
736
- id: z.ZodString;
737
- description: z.ZodString;
738
- }, z.core.$strip>>>;
739
- }, z.core.$strip>>;
740
- content: z.ZodString;
741
- context: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
742
- model: z.ZodOptional<z.ZodString>;
743
- }, z.core.$strip>;
744
- declare const JudgeResultSchema: z.ZodObject<{
745
- composite: z.ZodNumber;
746
- dimensions: z.ZodRecord<z.ZodString, z.ZodNumber>;
747
- failureModes: z.ZodDefault<z.ZodArray<z.ZodString>>;
748
- wins: z.ZodDefault<z.ZodArray<z.ZodString>>;
749
- rationale: z.ZodString;
750
- rubricVersion: z.ZodString;
751
- model: z.ZodString;
752
- durationMs: z.ZodNumber;
753
- }, z.core.$strip>;
754
- declare const RubricInfoSchema: z.ZodObject<{
755
- name: z.ZodString;
756
- description: z.ZodString;
757
- dimensions: z.ZodArray<z.ZodObject<{
758
- id: z.ZodString;
759
- description: z.ZodString;
760
- weight: z.ZodNumber;
761
- }, z.core.$strip>>;
762
- failureModes: z.ZodDefault<z.ZodArray<z.ZodString>>;
763
- rubricVersion: z.ZodString;
764
- }, z.core.$strip>;
765
- declare const ListRubricsResponseSchema: z.ZodObject<{
766
- rubrics: z.ZodArray<z.ZodObject<{
767
- name: z.ZodString;
768
- description: z.ZodString;
769
- dimensions: z.ZodArray<z.ZodObject<{
770
- id: z.ZodString;
771
- description: z.ZodString;
772
- weight: z.ZodNumber;
773
- }, z.core.$strip>>;
774
- failureModes: z.ZodDefault<z.ZodArray<z.ZodString>>;
775
- rubricVersion: z.ZodString;
776
- }, z.core.$strip>>;
777
- }, z.core.$strip>;
778
- declare const VersionResponseSchema: z.ZodObject<{
779
- package: z.ZodString;
780
- version: z.ZodString;
781
- wireVersion: z.ZodString;
782
- apiSurface: z.ZodArray<z.ZodString>;
783
- }, z.core.$strip>;
784
- declare const HealthResponseSchema: z.ZodObject<{
785
- status: z.ZodLiteral<"ok">;
786
- uptimeSec: z.ZodNumber;
787
- }, z.core.$strip>;
788
- /**
789
- * Minimal `TraceEvent` shape that the production runtime emits.
790
- * Matches `trace/schema.ts` `TraceEvent` but is duplicated here as a
791
- * wire schema so non-TypeScript clients can validate without depending
792
- * on internal types.
793
- */
794
- declare const TraceEventSchema: z.ZodObject<{
795
- eventId: z.ZodString;
796
- runId: z.ZodString;
797
- spanId: z.ZodOptional<z.ZodString>;
798
- kind: z.ZodEnum<{
799
- error: "error";
800
- custom: "custom";
801
- policy_violation: "policy_violation";
802
- log: "log";
803
- budget_decrement: "budget_decrement";
804
- budget_breach: "budget_breach";
805
- state_mutation: "state_mutation";
806
- redaction_applied: "redaction_applied";
807
- }>;
808
- timestamp: z.ZodNumber;
809
- payload: z.ZodRecord<z.ZodString, z.ZodUnknown>;
810
- }, z.core.$strip>;
811
- declare const TracesIngestRequestSchema: z.ZodObject<{
812
- events: z.ZodArray<z.ZodObject<{
813
- eventId: z.ZodString;
814
- runId: z.ZodString;
815
- spanId: z.ZodOptional<z.ZodString>;
816
- kind: z.ZodEnum<{
817
- error: "error";
818
- custom: "custom";
819
- policy_violation: "policy_violation";
820
- log: "log";
821
- budget_decrement: "budget_decrement";
822
- budget_breach: "budget_breach";
823
- state_mutation: "state_mutation";
824
- redaction_applied: "redaction_applied";
825
- }>;
826
- timestamp: z.ZodNumber;
827
- payload: z.ZodRecord<z.ZodString, z.ZodUnknown>;
828
- }, z.core.$strip>>;
829
- }, z.core.$strip>;
830
- declare const TracesIngestResponseSchema: z.ZodObject<{
831
- accepted: z.ZodNumber;
832
- rejected: z.ZodNumber;
833
- errors: z.ZodDefault<z.ZodArray<z.ZodObject<{
834
- eventId: z.ZodString;
835
- message: z.ZodString;
836
- }, z.core.$strip>>>;
837
- }, z.core.$strip>;
838
- declare const FeedbackLabelSchema: z.ZodObject<{
839
- id: z.ZodOptional<z.ZodString>;
840
- source: z.ZodEnum<{
841
- judge: "judge";
842
- user: "user";
843
- system: "system";
844
- policy: "policy";
845
- environment: "environment";
846
- metric: "metric";
847
- }>;
848
- kind: z.ZodEnum<{
849
- approve: "approve";
850
- reject: "reject";
851
- select: "select";
852
- edit: "edit";
853
- rank: "rank";
854
- rate: "rate";
855
- comment: "comment";
856
- metric_outcome: "metric_outcome";
857
- policy_block: "policy_block";
858
- revision_request: "revision_request";
859
- }>;
860
- value: z.ZodUnknown;
861
- reason: z.ZodOptional<z.ZodString>;
862
- severity: z.ZodOptional<z.ZodEnum<{
863
- error: "error";
864
- info: "info";
865
- critical: "critical";
866
- warning: "warning";
867
- }>>;
868
- createdAt: z.ZodString;
869
- metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
870
- }, z.core.$strip>;
871
- declare const FeedbackAttemptSchema: z.ZodObject<{
872
- id: z.ZodString;
873
- stepIndex: z.ZodNumber;
874
- artifactType: z.ZodEnum<{
875
- text: "text";
876
- code: "code";
877
- action: "action";
878
- decision: "decision";
879
- plan: "plan";
880
- research: "research";
881
- ui: "ui";
882
- data: "data";
883
- other: "other";
884
- }>;
885
- artifact: z.ZodUnknown;
886
- options: z.ZodOptional<z.ZodArray<z.ZodUnknown>>;
887
- proposedAction: z.ZodOptional<z.ZodObject<{
888
- type: z.ZodString;
889
- risk: z.ZodOptional<z.ZodEnum<{
890
- medium: "medium";
891
- low: "low";
892
- high: "high";
893
- }>>;
894
- costUsd: z.ZodOptional<z.ZodNumber>;
895
- externalSideEffect: z.ZodOptional<z.ZodBoolean>;
896
- requiresApproval: z.ZodOptional<z.ZodBoolean>;
897
- metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
898
- }, z.core.$strip>>;
899
- feedback: z.ZodOptional<z.ZodArray<z.ZodObject<{
900
- id: z.ZodOptional<z.ZodString>;
901
- source: z.ZodEnum<{
902
- judge: "judge";
903
- user: "user";
904
- system: "system";
905
- policy: "policy";
906
- environment: "environment";
907
- metric: "metric";
908
- }>;
909
- kind: z.ZodEnum<{
910
- approve: "approve";
911
- reject: "reject";
912
- select: "select";
913
- edit: "edit";
914
- rank: "rank";
915
- rate: "rate";
916
- comment: "comment";
917
- metric_outcome: "metric_outcome";
918
- policy_block: "policy_block";
919
- revision_request: "revision_request";
920
- }>;
921
- value: z.ZodUnknown;
922
- reason: z.ZodOptional<z.ZodString>;
923
- severity: z.ZodOptional<z.ZodEnum<{
924
- error: "error";
925
- info: "info";
926
- critical: "critical";
927
- warning: "warning";
928
- }>>;
929
- createdAt: z.ZodString;
930
- metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
931
- }, z.core.$strip>>>;
932
- createdAt: z.ZodString;
933
- metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
934
- }, z.core.$strip>;
935
- declare const FeedbackTrajectorySchema: z.ZodObject<{
936
- id: z.ZodString;
937
- projectId: z.ZodOptional<z.ZodString>;
938
- scenarioId: z.ZodOptional<z.ZodString>;
939
- task: z.ZodObject<{
940
- intent: z.ZodString;
941
- context: z.ZodOptional<z.ZodUnknown>;
942
- }, z.core.$strip>;
943
- attempts: z.ZodDefault<z.ZodArray<z.ZodObject<{
944
- id: z.ZodString;
945
- stepIndex: z.ZodNumber;
946
- artifactType: z.ZodEnum<{
947
- text: "text";
948
- code: "code";
949
- action: "action";
950
- decision: "decision";
951
- plan: "plan";
952
- research: "research";
953
- ui: "ui";
954
- data: "data";
955
- other: "other";
956
- }>;
957
- artifact: z.ZodUnknown;
958
- options: z.ZodOptional<z.ZodArray<z.ZodUnknown>>;
959
- proposedAction: z.ZodOptional<z.ZodObject<{
960
- type: z.ZodString;
961
- risk: z.ZodOptional<z.ZodEnum<{
962
- medium: "medium";
963
- low: "low";
964
- high: "high";
965
- }>>;
966
- costUsd: z.ZodOptional<z.ZodNumber>;
967
- externalSideEffect: z.ZodOptional<z.ZodBoolean>;
968
- requiresApproval: z.ZodOptional<z.ZodBoolean>;
969
- metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
970
- }, z.core.$strip>>;
971
- feedback: z.ZodOptional<z.ZodArray<z.ZodObject<{
972
- id: z.ZodOptional<z.ZodString>;
973
- source: z.ZodEnum<{
974
- judge: "judge";
975
- user: "user";
976
- system: "system";
977
- policy: "policy";
978
- environment: "environment";
979
- metric: "metric";
980
- }>;
981
- kind: z.ZodEnum<{
982
- approve: "approve";
983
- reject: "reject";
984
- select: "select";
985
- edit: "edit";
986
- rank: "rank";
987
- rate: "rate";
988
- comment: "comment";
989
- metric_outcome: "metric_outcome";
990
- policy_block: "policy_block";
991
- revision_request: "revision_request";
992
- }>;
993
- value: z.ZodUnknown;
994
- reason: z.ZodOptional<z.ZodString>;
995
- severity: z.ZodOptional<z.ZodEnum<{
996
- error: "error";
997
- info: "info";
998
- critical: "critical";
999
- warning: "warning";
1000
- }>>;
1001
- createdAt: z.ZodString;
1002
- metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
1003
- }, z.core.$strip>>>;
1004
- createdAt: z.ZodString;
1005
- metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
1006
- }, z.core.$strip>>>;
1007
- labels: z.ZodDefault<z.ZodArray<z.ZodObject<{
1008
- id: z.ZodOptional<z.ZodString>;
1009
- source: z.ZodEnum<{
1010
- judge: "judge";
1011
- user: "user";
1012
- system: "system";
1013
- policy: "policy";
1014
- environment: "environment";
1015
- metric: "metric";
1016
- }>;
1017
- kind: z.ZodEnum<{
1018
- approve: "approve";
1019
- reject: "reject";
1020
- select: "select";
1021
- edit: "edit";
1022
- rank: "rank";
1023
- rate: "rate";
1024
- comment: "comment";
1025
- metric_outcome: "metric_outcome";
1026
- policy_block: "policy_block";
1027
- revision_request: "revision_request";
1028
- }>;
1029
- value: z.ZodUnknown;
1030
- reason: z.ZodOptional<z.ZodString>;
1031
- severity: z.ZodOptional<z.ZodEnum<{
1032
- error: "error";
1033
- info: "info";
1034
- critical: "critical";
1035
- warning: "warning";
1036
- }>>;
1037
- createdAt: z.ZodString;
1038
- metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
1039
- }, z.core.$strip>>>;
1040
- outcome: z.ZodOptional<z.ZodObject<{
1041
- success: z.ZodOptional<z.ZodBoolean>;
1042
- score: z.ZodOptional<z.ZodNumber>;
1043
- metrics: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodNumber>>;
1044
- costUsd: z.ZodOptional<z.ZodNumber>;
1045
- detail: z.ZodOptional<z.ZodString>;
1046
- observedAt: z.ZodOptional<z.ZodString>;
1047
- metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
1048
- }, z.core.$strip>>;
1049
- split: z.ZodOptional<z.ZodEnum<{
1050
- train: "train";
1051
- dev: "dev";
1052
- test: "test";
1053
- holdout: "holdout";
1054
- }>>;
1055
- tags: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
1056
- createdAt: z.ZodString;
1057
- updatedAt: z.ZodOptional<z.ZodString>;
1058
- metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
1059
- }, z.core.$strip>;
1060
- declare const FeedbackIngestResponseSchema: z.ZodObject<{
1061
- id: z.ZodString;
1062
- persisted: z.ZodBoolean;
1063
- }, z.core.$strip>;
1064
- type TraceEvent = z.infer<typeof TraceEventSchema>;
1065
- type TracesIngestRequest = z.infer<typeof TracesIngestRequestSchema>;
1066
- type TracesIngestResponse = z.infer<typeof TracesIngestResponseSchema>;
1067
- type FeedbackTrajectory = z.infer<typeof FeedbackTrajectorySchema>;
1068
- type FeedbackIngestResponse = z.infer<typeof FeedbackIngestResponseSchema>;
1069
- declare const ErrorResponseSchema: z.ZodObject<{
1070
- error: z.ZodObject<{
1071
- code: z.ZodString;
1072
- message: z.ZodString;
1073
- details: z.ZodOptional<z.ZodUnknown>;
1074
- }, z.core.$strip>;
1075
- }, z.core.$strip>;
1076
- type RubricDimension = z.infer<typeof RubricDimensionSchema>;
1077
- type FailureMode = z.infer<typeof FailureModeSchema>;
1078
- type Rubric = z.infer<typeof RubricSchema>;
1079
- type JudgeRequest = z.infer<typeof JudgeRequestSchema>;
1080
- type JudgeResult = z.infer<typeof JudgeResultSchema>;
1081
- type RubricInfo = z.infer<typeof RubricInfoSchema>;
1082
- type ListRubricsResponse = z.infer<typeof ListRubricsResponseSchema>;
1083
- type VersionResponse = z.infer<typeof VersionResponseSchema>;
1084
- type ErrorResponse = z.infer<typeof ErrorResponseSchema>;
1085
- /**
1086
- * Bump on any breaking change to a request/response schema.
1087
- * Non-breaking (additive) changes don't require a bump.
1088
- */
1089
- declare const WIRE_VERSION = "1.0.0";
1090
- /**
1091
- * Stable hash of a rubric. Used to make scores comparable across runs:
1092
- * if the rubricVersion matches, the rubric was identical.
1093
- */
1094
- declare function hashRubric(rubric: Rubric): string;
1095
-
1096
- /**
1097
- * Pure handler functions — the "business logic" behind every wire-protocol
1098
- * method. The HTTP server (`server.ts`) and the stdio RPC (`rpc.ts`) both
1099
- * call these. Tests call these directly without spinning a server.
1100
- *
1101
- * Each handler:
1102
- * - Takes a parsed request (already Zod-validated by the transport).
1103
- * - Returns a result that matches the response schema.
1104
- * - Throws `WireError` for caller-fixable errors (404, 400, 422).
1105
- * - Lets unexpected errors bubble — the transport maps them to 500.
1106
- */
1107
-
1108
- /** Caller-fixable error. The transport renders this to 4xx + ErrorResponse. */
1109
- declare class WireError extends Error {
1110
- readonly code: string;
1111
- readonly status: number;
1112
- readonly details?: unknown | undefined;
1113
- constructor(code: string, message: string, status?: number, details?: unknown | undefined);
1114
- }
1115
- interface HandleJudgeOptions {
1116
- costLedger?: CostLedgerHandle;
1117
- costPhase?: string;
1118
- llm?: LlmClientOptions;
1119
- signal?: AbortSignal;
1120
- }
1121
- declare function handleJudge(req: JudgeRequest, options?: HandleJudgeOptions): Promise<JudgeResult>;
1122
- declare function handleListRubrics(): ListRubricsResponse;
1123
- declare function handleVersion(): VersionResponse;
1124
- /**
1125
- * Pluggable stores the wire layer routes ingestion writes into. Both
1126
- * are optional — when omitted, the corresponding endpoint returns 503.
1127
- *
1128
- * Production deployments wire a `FileSystemTraceStore` and
1129
- * `FileSystemFeedbackTrajectoryStore` here. Tests substitute in-memory
1130
- * stores.
1131
- */
1132
- interface IngestionStores {
1133
- traceStore?: TraceStore;
1134
- feedbackStore?: FeedbackTrajectoryStore;
1135
- }
1136
- /**
1137
- * `POST /v1/traces/ingest` — accept a batch of `TraceEvent`s from the
1138
- * production runtime. Best-effort: each event is appended independently;
1139
- * one bad event does not poison the batch.
1140
- *
1141
- * Idempotency: the underlying store is append-only; consumers retrying
1142
- * the same payload will get duplicate events. Consumers should
1143
- * de-duplicate by `eventId` downstream — production traces frequently
1144
- * land via at-least-once buses (Kafka, SQS) where dedup is unavoidable.
1145
- */
1146
- declare function handleTracesIngest(req: TracesIngestRequest, stores: IngestionStores): Promise<TracesIngestResponse>;
1147
- /**
1148
- * `POST /v1/feedback` — accept a single `FeedbackTrajectory` from the
1149
- * production runtime. Idempotent on `id`: re-posting the same trajectory
1150
- * replaces the prior record.
1151
- */
1152
- declare function handleFeedbackIngest(req: FeedbackTrajectory, stores: IngestionStores): Promise<FeedbackIngestResponse>;
1153
-
1154
- declare function buildOpenApi(packageVersion: string): OpenAPIObject;
1155
-
1156
- interface RpcRequest {
1157
- method: 'judge' | 'listRubrics' | 'version';
1158
- params?: unknown;
1159
- }
1160
- interface RpcSuccess {
1161
- result: unknown;
1162
- }
1163
- interface RpcError {
1164
- error: {
1165
- code: string;
1166
- message: string;
1167
- details?: unknown;
1168
- };
1169
- }
1170
- declare function dispatchRpc(req: RpcRequest): Promise<RpcSuccess | RpcError>;
1171
- /** Read one JSON request from stdin, write one JSON response to stdout. */
1172
- declare function runRpcOnce(method?: string): Promise<number>;
1173
- /** Read JSONL requests from stdin, write JSONL responses to stdout. */
1174
- declare function runRpcBatch(method?: string): Promise<number>;
1175
-
1176
- /**
1177
- * Built-in rubrics shipped with agent-eval.
1178
- *
1179
- * A rubric is a set of scoring axes plus a system prompt that tells the
1180
- * judging LLM how to grade against those axes. Built-in rubrics are
1181
- * curated for use cases that recur across Tangle projects — call them
1182
- * by name from any client.
1183
- *
1184
- * Adding a rubric:
1185
- * 1. Define the Rubric object below with a clear `description` and
1186
- * named `dimensions`.
1187
- * 2. Register it in `BUILTIN_RUBRICS` at the bottom.
1188
- * 3. Add a test in `tests/wire/rubrics.test.ts`.
1189
- *
1190
- * Custom rubrics: callers pass `rubric` inline to /v1/judge instead of
1191
- * `rubricName` — see schemas.ts.
1192
- */
1193
-
1194
- declare const BUILTIN_RUBRICS: Record<string, Rubric>;
1195
- /** Get a built-in rubric by name, or undefined. */
1196
- declare function getBuiltinRubric(name: string): Rubric | undefined;
1197
- /** List built-in rubrics with their stable versions. */
1198
- declare function listBuiltinRubrics(): {
1199
- name: string;
1200
- description: string;
1201
- dimensions: {
1202
- id: string;
1203
- description: string;
1204
- weight: number;
1205
- }[];
1206
- failureModes: string[];
1207
- rubricVersion: string;
1208
- }[];
1209
-
1210
- interface CreateAppOptions {
1211
- /** Stores wired to the ingestion endpoints. */
1212
- stores?: IngestionStores;
1213
- /**
1214
- * Bearer-token auth. When provided, every endpoint EXCEPT `/healthz`
1215
- * and `/v1/version` requires `Authorization: Bearer <token>`. The
1216
- * token may be a static string OR a function for time-bounded /
1217
- * rotating tokens.
1218
- *
1219
- * Recommended for any server that accepts ingestion writes from the
1220
- * public internet. Read-only deployments may omit it.
1221
- */
1222
- auth?: {
1223
- bearer: string | ((token: string) => boolean | Promise<boolean>);
1224
- };
1225
- }
1226
- declare function createApp(opts?: CreateAppOptions): Hono<hono_types.BlankEnv, hono_types.BlankSchema, "/">;
1227
- interface ServeOptions extends CreateAppOptions {
1228
- /** Default 5005. */
1229
- port?: number;
1230
- /** Default '127.0.0.1'. Set to '0.0.0.0' to listen on all interfaces. */
1231
- host?: string;
1232
- }
1233
- declare function startServer(opts?: ServeOptions): ServerType;
1234
- interface StartedServer {
1235
- server: ServerType;
1236
- /** The OS-assigned port. When opts.port was 0, this is the actual port the
1237
- * kernel bound — callers that need to dial back (smoke tests, sidecars
1238
- * registering with a parent) read this rather than guessing a free port. */
1239
- port: number;
1240
- /** Resolved host the server bound to (defaults to 127.0.0.1). */
1241
- host: string;
1242
- /** Close the server. Resolves once active connections have drained. */
1243
- close(): Promise<void>;
1244
- }
1245
- /**
1246
- * Promise-returning variant of `startServer` that resolves once the server is
1247
- * listening and surfaces the resolved bound port. Use this from smoke tests
1248
- * (`startServerAsync({ port: 0 })`) and any caller that needs to dial back.
1249
- */
1250
- declare function startServerAsync(opts?: ServeOptions): Promise<StartedServer>;
1251
-
1252
- export { BUILTIN_RUBRICS, type ErrorResponse, ErrorResponseSchema, type FailureMode, FailureModeSchema, FeedbackAttemptSchema, type FeedbackIngestResponse, FeedbackIngestResponseSchema, FeedbackLabelSchema, type FeedbackTrajectory, FeedbackTrajectorySchema, type HandleJudgeOptions, HealthResponseSchema, type IngestionStores, type JudgeRequest, JudgeRequestSchema, type JudgeResult, JudgeResultSchema, type ListRubricsResponse, ListRubricsResponseSchema, type Rubric, type RubricDimension, RubricDimensionSchema, type RubricInfo, RubricInfoSchema, RubricSchema, type ServeOptions, type StartedServer, type TraceEvent, TraceEventSchema, type TracesIngestRequest, TracesIngestRequestSchema, type TracesIngestResponse, TracesIngestResponseSchema, type VersionResponse, VersionResponseSchema, WIRE_VERSION, WireError, buildOpenApi, createApp, dispatchRpc, getBuiltinRubric, handleFeedbackIngest, handleJudge, handleListRubrics, handleTracesIngest, handleVersion, hashRubric, listBuiltinRubrics, runRpcBatch, runRpcOnce, startServer, startServerAsync };