@tangle-network/agent-eval 0.120.1 → 0.120.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/package.json +1 -1
- package/dist/analyst/index.d.ts +0 -3111
- package/dist/analyst/index.js +0 -403
- package/dist/analyst/index.js.map +0 -1
- package/dist/authenticity/index.d.ts +0 -161
- package/dist/authenticity/index.js +0 -215
- package/dist/authenticity/index.js.map +0 -1
- package/dist/belief-state/index.d.ts +0 -1301
- package/dist/belief-state/index.js +0 -2152
- package/dist/belief-state/index.js.map +0 -1
- package/dist/benchmarks/index.d.ts +0 -974
- package/dist/benchmarks/index.js +0 -60
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/builder-eval/index.d.ts +0 -695
- package/dist/builder-eval/index.js +0 -366
- package/dist/builder-eval/index.js.map +0 -1
- package/dist/campaign/index.d.ts +0 -7454
- package/dist/campaign/index.js +0 -272
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-32BZXMSO.js +0 -3878
- package/dist/chunk-32BZXMSO.js.map +0 -1
- package/dist/chunk-3A246TSA.js +0 -998
- package/dist/chunk-3A246TSA.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-3YYRZDON.js +0 -45
- package/dist/chunk-3YYRZDON.js.map +0 -1
- package/dist/chunk-4I2E3LLO.js +0 -1030
- package/dist/chunk-4I2E3LLO.js.map +0 -1
- package/dist/chunk-ARU2PZFM.js +0 -312
- package/dist/chunk-ARU2PZFM.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-DPZAEKA6.js +0 -880
- package/dist/chunk-DPZAEKA6.js.map +0 -1
- package/dist/chunk-DTJ6QUQB.js +0 -131
- package/dist/chunk-DTJ6QUQB.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H5UD2323.js +0 -286
- package/dist/chunk-H5UD2323.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HKUCJ437.js +0 -787
- package/dist/chunk-HKUCJ437.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JHOJHHU7.js +0 -867
- package/dist/chunk-JHOJHHU7.js.map +0 -1
- package/dist/chunk-JM2SKQMS.js +0 -750
- package/dist/chunk-JM2SKQMS.js.map +0 -1
- package/dist/chunk-JN2FCO5W.js +0 -7958
- package/dist/chunk-JN2FCO5W.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MOXWMGPC.js +0 -577
- package/dist/chunk-MOXWMGPC.js.map +0 -1
- package/dist/chunk-NJC7U437.js +0 -626
- package/dist/chunk-NJC7U437.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OYZAPX5G.js +0 -1526
- package/dist/chunk-OYZAPX5G.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PICTDURQ.js +0 -766
- package/dist/chunk-PICTDURQ.js.map +0 -1
- package/dist/chunk-PJQFMIOX.js +0 -1182
- package/dist/chunk-PJQFMIOX.js.map +0 -1
- package/dist/chunk-PXD6ZFNY.js +0 -1107
- package/dist/chunk-PXD6ZFNY.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QBRSJK47.js +0 -622
- package/dist/chunk-QBRSJK47.js.map +0 -1
- package/dist/chunk-QWMPPZ3X.js +0 -550
- package/dist/chunk-QWMPPZ3X.js.map +0 -1
- package/dist/chunk-S3UZOQ5Y.js +0 -328
- package/dist/chunk-S3UZOQ5Y.js.map +0 -1
- package/dist/chunk-S5TT5R3L.js +0 -2668
- package/dist/chunk-S5TT5R3L.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-U5CHZ5M3.js +0 -357
- package/dist/chunk-U5CHZ5M3.js.map +0 -1
- package/dist/chunk-ULOKLHIQ.js +0 -1937
- package/dist/chunk-ULOKLHIQ.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VSMTAMNK.js +0 -53
- package/dist/chunk-VSMTAMNK.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WW2A73HW.js +0 -159
- package/dist/chunk-WW2A73HW.js.map +0 -1
- package/dist/chunk-X4UCIOTZ.js +0 -136
- package/dist/chunk-X4UCIOTZ.js.map +0 -1
- package/dist/chunk-XDIRG3TO.js +0 -1266
- package/dist/chunk-XDIRG3TO.js.map +0 -1
- package/dist/chunk-XJYR7XFV.js +0 -317
- package/dist/chunk-XJYR7XFV.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZZUXHH3R.js +0 -99
- package/dist/chunk-ZZUXHH3R.js.map +0 -1
- package/dist/cli.d.ts +0 -1
- package/dist/cli.js +0 -112
- package/dist/cli.js.map +0 -1
- package/dist/contract/index.d.ts +0 -4972
- package/dist/contract/index.js +0 -1654
- package/dist/contract/index.js.map +0 -1
- package/dist/control.d.ts +0 -1013
- package/dist/control.js +0 -34
- package/dist/control.js.map +0 -1
- package/dist/fuzz.d.ts +0 -759
- package/dist/fuzz.js +0 -714
- package/dist/fuzz.js.map +0 -1
- package/dist/hosted/index.d.ts +0 -730
- package/dist/hosted/index.js +0 -14
- package/dist/hosted/index.js.map +0 -1
- package/dist/index.d.ts +0 -16780
- package/dist/index.js +0 -12168
- package/dist/index.js.map +0 -1
- package/dist/matrix/index.d.ts +0 -155
- package/dist/matrix/index.js +0 -8
- package/dist/matrix/index.js.map +0 -1
- package/dist/meta-eval/index.d.ts +0 -1030
- package/dist/meta-eval/index.js +0 -417
- package/dist/meta-eval/index.js.map +0 -1
- package/dist/multishot/index.d.ts +0 -579
- package/dist/multishot/index.js +0 -589
- package/dist/multishot/index.js.map +0 -1
- package/dist/openapi.json +0 -992
- package/dist/pipelines/index.d.ts +0 -567
- package/dist/pipelines/index.js +0 -515
- package/dist/pipelines/index.js.map +0 -1
- package/dist/reporting.d.ts +0 -1277
- package/dist/reporting.js +0 -48
- package/dist/reporting.js.map +0 -1
- package/dist/rl.d.ts +0 -4092
- package/dist/rl.js +0 -1724
- package/dist/rl.js.map +0 -1
- package/dist/run-campaign-HNFPJET4.js +0 -14
- package/dist/run-campaign-HNFPJET4.js.map +0 -1
- package/dist/storyboard/index.d.ts +0 -279
- package/dist/storyboard/index.js +0 -767
- package/dist/storyboard/index.js.map +0 -1
- package/dist/trace-attributes.d.ts +0 -52
- package/dist/trace-attributes.js +0 -62
- package/dist/trace-attributes.js.map +0 -1
- package/dist/traces.d.ts +0 -2343
- package/dist/traces.js +0 -249
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.d.ts +0 -1252
- package/dist/wire/index.js +0 -81
- package/dist/wire/index.js.map +0 -1
package/dist/wire/index.d.ts
DELETED
|
@@ -1,1252 +0,0 @@
|
|
|
1
|
-
import { z } from 'zod';
|
|
2
|
-
import { OpenAPIObject } from 'openapi3-ts/oas31';
|
|
3
|
-
import * as hono_types from 'hono/types';
|
|
4
|
-
import { ServerType } from '@hono/node-server';
|
|
5
|
-
import { Hono } from 'hono';
|
|
6
|
-
|
|
7
|
-
type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
|
|
8
|
-
interface CostUsage {
|
|
9
|
-
inputTokens: number;
|
|
10
|
-
/** Includes reasoning tokens when the provider bills them as output. */
|
|
11
|
-
outputTokens: number;
|
|
12
|
-
/** Reasoning-token subset of outputTokens, when reported. */
|
|
13
|
-
reasoningTokens?: number;
|
|
14
|
-
/** Prompt tokens served from a provider cache. */
|
|
15
|
-
cachedTokens?: number;
|
|
16
|
-
/** Prompt tokens written into a provider cache. */
|
|
17
|
-
cacheWriteTokens?: number;
|
|
18
|
-
}
|
|
19
|
-
interface CostCallBase {
|
|
20
|
-
callId: string;
|
|
21
|
-
channel: CostChannel;
|
|
22
|
-
phase: string;
|
|
23
|
-
actor: string;
|
|
24
|
-
model: string;
|
|
25
|
-
maximumCostUsd?: number;
|
|
26
|
-
tags?: Record<string, string>;
|
|
27
|
-
timestamp: number;
|
|
28
|
-
}
|
|
29
|
-
interface CostReceipt extends CostCallBase, CostUsage {
|
|
30
|
-
status: 'settled';
|
|
31
|
-
costUsd: number;
|
|
32
|
-
costUnknown: boolean;
|
|
33
|
-
usageUnknown?: boolean;
|
|
34
|
-
pricing?: {
|
|
35
|
-
inputUsdPerThousand: number;
|
|
36
|
-
outputUsdPerThousand: number;
|
|
37
|
-
};
|
|
38
|
-
actualCostUsd?: number;
|
|
39
|
-
error?: string;
|
|
40
|
-
}
|
|
41
|
-
interface CostReceiptInput extends CostUsage {
|
|
42
|
-
model: string;
|
|
43
|
-
actualCostUsd?: number;
|
|
44
|
-
costUnknown?: boolean;
|
|
45
|
-
usageUnknown?: boolean;
|
|
46
|
-
}
|
|
47
|
-
type MaximumCharge = {
|
|
48
|
-
externallyEnforcedMaximumUsd: number;
|
|
49
|
-
} | ({
|
|
50
|
-
model: string;
|
|
51
|
-
} & CostUsage);
|
|
52
|
-
interface RunPaidCallInput<T> {
|
|
53
|
-
callId?: string;
|
|
54
|
-
channel: CostChannel;
|
|
55
|
-
phase: string;
|
|
56
|
-
actor: string;
|
|
57
|
-
/** Used before a provider receipt exists and on failures without one. */
|
|
58
|
-
model?: string;
|
|
59
|
-
tags?: Record<string, string>;
|
|
60
|
-
signal?: AbortSignal;
|
|
61
|
-
/** Provider-enforced dollar maximum, or maximum priced token usage. Required when capped. */
|
|
62
|
-
maximumCharge?: MaximumCharge;
|
|
63
|
-
/** `callId` can be forwarded as the provider's idempotency key. */
|
|
64
|
-
execute(signal: AbortSignal, callId: string): Promise<T>;
|
|
65
|
-
receipt(value: T): CostReceiptInput;
|
|
66
|
-
receiptFromError?(error: Error): CostReceiptInput | undefined;
|
|
67
|
-
}
|
|
68
|
-
type PaidCallResult<T> = {
|
|
69
|
-
succeeded: true;
|
|
70
|
-
callId: string;
|
|
71
|
-
value: T;
|
|
72
|
-
receipt: CostReceipt;
|
|
73
|
-
} | {
|
|
74
|
-
succeeded: false;
|
|
75
|
-
callId?: string;
|
|
76
|
-
error: Error;
|
|
77
|
-
receipt?: CostReceipt;
|
|
78
|
-
};
|
|
79
|
-
interface ChannelRollup {
|
|
80
|
-
channel: CostChannel;
|
|
81
|
-
calls: number;
|
|
82
|
-
inputTokens: number;
|
|
83
|
-
outputTokens: number;
|
|
84
|
-
reasoningTokens?: number;
|
|
85
|
-
cachedTokens: number;
|
|
86
|
-
cacheWriteTokens?: number;
|
|
87
|
-
costUsd: number;
|
|
88
|
-
unpricedCalls: number;
|
|
89
|
-
unknownUsageCalls: number;
|
|
90
|
-
}
|
|
91
|
-
interface CostLedgerSummary {
|
|
92
|
-
totalCalls: number;
|
|
93
|
-
pendingCalls: number;
|
|
94
|
-
unresolvedCalls: number;
|
|
95
|
-
reservedCostUsd: number;
|
|
96
|
-
inputTokens: number;
|
|
97
|
-
outputTokens: number;
|
|
98
|
-
reasoningTokens?: number;
|
|
99
|
-
cachedTokens: number;
|
|
100
|
-
cacheWriteTokens?: number;
|
|
101
|
-
totalCostUsd: number;
|
|
102
|
-
byChannel: ChannelRollup[];
|
|
103
|
-
unpricedModels: string[];
|
|
104
|
-
fullyPriced: boolean;
|
|
105
|
-
usageComplete: boolean;
|
|
106
|
-
accountingComplete: boolean;
|
|
107
|
-
incompleteReasons: string[];
|
|
108
|
-
}
|
|
109
|
-
interface CostLedgerFilter {
|
|
110
|
-
channel?: CostChannel;
|
|
111
|
-
phase?: string;
|
|
112
|
-
tags?: Record<string, string>;
|
|
113
|
-
}
|
|
114
|
-
interface CostLedgerWaitOptions {
|
|
115
|
-
/** Maximum time to wait for active provider calls. Default 5 seconds. */
|
|
116
|
-
timeoutMs?: number;
|
|
117
|
-
}
|
|
118
|
-
/** Append-only storage. `append` must atomically reject stale revisions. */
|
|
119
|
-
interface CostLedgerPersistence {
|
|
120
|
-
read(): {
|
|
121
|
-
revision: string;
|
|
122
|
-
events: string;
|
|
123
|
-
};
|
|
124
|
-
append(expectedRevision: string, event: string): string | undefined;
|
|
125
|
-
}
|
|
126
|
-
interface CostLedgerOptions {
|
|
127
|
-
costCeilingUsd?: number;
|
|
128
|
-
persistence?: CostLedgerPersistence;
|
|
129
|
-
/** Import already-settled receipts without admitting new paid work. */
|
|
130
|
-
receipts?: readonly CostReceipt[];
|
|
131
|
-
}
|
|
132
|
-
/** Run-wide paid-call admission, durable call state, receipts, and summaries. */
|
|
133
|
-
declare class CostLedger {
|
|
134
|
-
private readonly records;
|
|
135
|
-
private readonly activeCallIds;
|
|
136
|
-
private readonly lateCallIds;
|
|
137
|
-
private readonly idleWaiters;
|
|
138
|
-
private completedTasks;
|
|
139
|
-
private revision;
|
|
140
|
-
private costLimitPersisted;
|
|
141
|
-
readonly costCeilingUsd?: number;
|
|
142
|
-
private readonly persistence?;
|
|
143
|
-
constructor(input?: number | CostLedgerOptions);
|
|
144
|
-
runPaidCall<T>(input: RunPaidCallInput<T>): Promise<PaidCallResult<T>>;
|
|
145
|
-
/** Wait until every call started by this ledger has produced a durable outcome. */
|
|
146
|
-
waitForIdle(options?: CostLedgerWaitOptions): Promise<boolean>;
|
|
147
|
-
/** Settle a call left pending by a crashed process after reconciling with the provider. */
|
|
148
|
-
reconcile(callId: string, observed: CostReceiptInput, options?: {
|
|
149
|
-
error?: string;
|
|
150
|
-
}): CostReceipt;
|
|
151
|
-
list(filter?: CostLedgerFilter): CostReceipt[];
|
|
152
|
-
summary(filter?: CostLedgerFilter): CostLedgerSummary;
|
|
153
|
-
markCompleted(count?: number): void;
|
|
154
|
-
costPerCompletedTask(): number | null;
|
|
155
|
-
private execute;
|
|
156
|
-
private captureLateOutcome;
|
|
157
|
-
private releaseActiveCall;
|
|
158
|
-
private commitOutcome;
|
|
159
|
-
private captureFailure;
|
|
160
|
-
private commitReceipt;
|
|
161
|
-
private resolveMaximum;
|
|
162
|
-
private hasIncompleteSettledCall;
|
|
163
|
-
private appendRecord;
|
|
164
|
-
private ensureCostLimitPersisted;
|
|
165
|
-
private appendEvent;
|
|
166
|
-
}
|
|
167
|
-
/** Public callback surface for a shared cost ledger.
|
|
168
|
-
*
|
|
169
|
-
* Declaration bundles may expose this type through multiple package subpaths.
|
|
170
|
-
* Keeping callback contracts structural lets those subpaths compose while the
|
|
171
|
-
* concrete {@link CostLedger} retains its private durable state.
|
|
172
|
-
*/
|
|
173
|
-
type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'waitForIdle'>> & Partial<Pick<CostLedger, 'waitForIdle'>>;
|
|
174
|
-
|
|
175
|
-
type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
|
|
176
|
-
interface BudgetSpec {
|
|
177
|
-
tokens?: number;
|
|
178
|
-
wallMs?: number;
|
|
179
|
-
calls?: number;
|
|
180
|
-
usd?: number;
|
|
181
|
-
}
|
|
182
|
-
interface RunOutcome {
|
|
183
|
-
score?: number;
|
|
184
|
-
pass?: boolean;
|
|
185
|
-
failureClass?: FailureClass;
|
|
186
|
-
notes?: string;
|
|
187
|
-
}
|
|
188
|
-
/**
|
|
189
|
-
* Layer — optional classification in a nested build workflow.
|
|
190
|
-
* `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
|
|
191
|
-
* `app-build`: sandbox harness that compiled + tested the generated scaffold.
|
|
192
|
-
* `app-runtime`: a run of the generated agent against a domain scenario.
|
|
193
|
-
* `meta`: any meta-eval (judge replay, correlation analysis).
|
|
194
|
-
*/
|
|
195
|
-
type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
|
|
196
|
-
interface Run {
|
|
197
|
-
runId: string;
|
|
198
|
-
/**
|
|
199
|
-
* Stable identifier of the scenario being executed.
|
|
200
|
-
*
|
|
201
|
-
* Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
|
|
202
|
-
* input WITHOUT this field, substituting a sensible default
|
|
203
|
-
* (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
|
|
204
|
-
* curated scenario to anchor to (runtime / operator / meta-eval runs). This
|
|
205
|
-
* keeps the persisted shape unambiguous for downstream filters + aggregations
|
|
206
|
-
* while removing the boilerplate of inventing placeholder ids at the call site.
|
|
207
|
-
*/
|
|
208
|
-
scenarioId: string;
|
|
209
|
-
variantId?: string;
|
|
210
|
-
datasetVersion?: string;
|
|
211
|
-
/** Git SHA of agent code at run time. */
|
|
212
|
-
codeSha?: string;
|
|
213
|
-
/** Hash of the prompt template + any system prompt. */
|
|
214
|
-
promptSha?: string;
|
|
215
|
-
/** Model id + date + system-prompt hash, concatenated. */
|
|
216
|
-
modelFingerprint?: string;
|
|
217
|
-
seed?: number;
|
|
218
|
-
/** Arbitrary environment markers (shell, docker version, tz). */
|
|
219
|
-
envFingerprint?: Record<string, string>;
|
|
220
|
-
/** Version of the redaction rules applied to this run. */
|
|
221
|
-
redactionVersion?: string;
|
|
222
|
-
/** Parent run in a nested build workflow. A builder run's children are
|
|
223
|
-
* app-build runs; those children are app-runtime runs. */
|
|
224
|
-
parentRunId?: string;
|
|
225
|
-
/** Stable project identifier — groups runs across chats + sessions. */
|
|
226
|
-
projectId?: string;
|
|
227
|
-
/** Chat/conversation identifier within a project. */
|
|
228
|
-
chatId?: string;
|
|
229
|
-
/** Layer classification — hint for aggregation; not enforced. */
|
|
230
|
-
layer?: RunLayer;
|
|
231
|
-
startedAt: number;
|
|
232
|
-
endedAt?: number;
|
|
233
|
-
status: RunStatus;
|
|
234
|
-
outcome?: RunOutcome;
|
|
235
|
-
budget?: BudgetSpec;
|
|
236
|
-
/** Free-form labels for downstream grouping. */
|
|
237
|
-
tags?: Record<string, string>;
|
|
238
|
-
}
|
|
239
|
-
type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
|
|
240
|
-
type SpanStatus = 'ok' | 'error';
|
|
241
|
-
interface SpanBase {
|
|
242
|
-
spanId: string;
|
|
243
|
-
parentSpanId?: string;
|
|
244
|
-
runId: string;
|
|
245
|
-
kind: SpanKind;
|
|
246
|
-
name: string;
|
|
247
|
-
startedAt: number;
|
|
248
|
-
endedAt?: number;
|
|
249
|
-
status?: SpanStatus;
|
|
250
|
-
error?: string;
|
|
251
|
-
/** Anything not covered by typed fields. Kept deliberately free-form. */
|
|
252
|
-
attributes?: Record<string, unknown>;
|
|
253
|
-
}
|
|
254
|
-
interface Message {
|
|
255
|
-
role: 'system' | 'user' | 'assistant' | 'tool';
|
|
256
|
-
content: string;
|
|
257
|
-
tokens?: number;
|
|
258
|
-
/** Multi-modal content descriptors; blobs themselves live in Artifacts. */
|
|
259
|
-
images?: Array<{
|
|
260
|
-
artifactId?: string;
|
|
261
|
-
url?: string;
|
|
262
|
-
mime?: string;
|
|
263
|
-
}>;
|
|
264
|
-
}
|
|
265
|
-
interface LlmSpan extends SpanBase {
|
|
266
|
-
kind: 'llm';
|
|
267
|
-
model: string;
|
|
268
|
-
messages: Message[];
|
|
269
|
-
output?: string;
|
|
270
|
-
inputTokens?: number;
|
|
271
|
-
/** All generated tokens, including the reasoning subset when present. */
|
|
272
|
-
outputTokens?: number;
|
|
273
|
-
cachedTokens?: number;
|
|
274
|
-
cacheWriteTokens?: number;
|
|
275
|
-
/** Reasoning-token subset of `outputTokens`. */
|
|
276
|
-
reasoningTokens?: number;
|
|
277
|
-
costUsd?: number;
|
|
278
|
-
finishReason?: string;
|
|
279
|
-
}
|
|
280
|
-
interface ToolSpan extends SpanBase {
|
|
281
|
-
kind: 'tool';
|
|
282
|
-
toolName: string;
|
|
283
|
-
args: unknown;
|
|
284
|
-
/** False when the source observed the call but did not capture its arguments. */
|
|
285
|
-
argsCaptured?: boolean;
|
|
286
|
-
result?: unknown;
|
|
287
|
-
latencyMs?: number;
|
|
288
|
-
}
|
|
289
|
-
interface RetrievalSpan extends SpanBase {
|
|
290
|
-
kind: 'retrieval';
|
|
291
|
-
query: string;
|
|
292
|
-
hits: Array<{
|
|
293
|
-
docId: string;
|
|
294
|
-
score: number;
|
|
295
|
-
content?: string;
|
|
296
|
-
}>;
|
|
297
|
-
}
|
|
298
|
-
interface JudgeSpan extends SpanBase {
|
|
299
|
-
kind: 'judge';
|
|
300
|
-
judgeId: string;
|
|
301
|
-
/** Span this judgment applies to. */
|
|
302
|
-
targetSpanId: string;
|
|
303
|
-
dimension: string;
|
|
304
|
-
/** Numeric score (free-range; interpretation up to the judge). */
|
|
305
|
-
score: number;
|
|
306
|
-
rationale?: string;
|
|
307
|
-
evidence?: string;
|
|
308
|
-
}
|
|
309
|
-
interface SandboxSpan extends SpanBase {
|
|
310
|
-
kind: 'sandbox';
|
|
311
|
-
image?: string;
|
|
312
|
-
command?: string;
|
|
313
|
-
exitCode?: number;
|
|
314
|
-
testsTotal?: number;
|
|
315
|
-
testsPassed?: number;
|
|
316
|
-
stdoutHash?: string;
|
|
317
|
-
stderrHash?: string;
|
|
318
|
-
/** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
|
|
319
|
-
wallMs?: number;
|
|
320
|
-
}
|
|
321
|
-
interface GenericSpan extends SpanBase {
|
|
322
|
-
kind: 'agent' | 'custom';
|
|
323
|
-
}
|
|
324
|
-
type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
|
|
325
|
-
type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
|
|
326
|
-
interface TraceEvent$1 {
|
|
327
|
-
eventId: string;
|
|
328
|
-
runId: string;
|
|
329
|
-
spanId?: string;
|
|
330
|
-
kind: EventKind;
|
|
331
|
-
timestamp: number;
|
|
332
|
-
payload: Record<string, unknown>;
|
|
333
|
-
}
|
|
334
|
-
interface BudgetLedgerEntry {
|
|
335
|
-
runId: string;
|
|
336
|
-
dimension: keyof BudgetSpec;
|
|
337
|
-
limit: number;
|
|
338
|
-
consumed: number;
|
|
339
|
-
remaining: number;
|
|
340
|
-
timestamp: number;
|
|
341
|
-
breached: boolean;
|
|
342
|
-
/** Span that triggered this entry, if any. */
|
|
343
|
-
spanId?: string;
|
|
344
|
-
}
|
|
345
|
-
interface Artifact {
|
|
346
|
-
artifactId: string;
|
|
347
|
-
runId: string;
|
|
348
|
-
spanId?: string;
|
|
349
|
-
contentType: string;
|
|
350
|
-
sizeBytes: number;
|
|
351
|
-
/** sha256 in hex. */
|
|
352
|
-
hash: string;
|
|
353
|
-
/** External storage URL (R2, S3, filesystem path). */
|
|
354
|
-
storageUrl?: string;
|
|
355
|
-
/** Inline content for small blobs — keep under ~64KB. */
|
|
356
|
-
inlineContent?: string;
|
|
357
|
-
}
|
|
358
|
-
type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
|
|
359
|
-
|
|
360
|
-
interface RunFilter {
|
|
361
|
-
scenarioId?: string;
|
|
362
|
-
variantId?: string;
|
|
363
|
-
status?: RunStatus;
|
|
364
|
-
since?: number;
|
|
365
|
-
until?: number;
|
|
366
|
-
tag?: {
|
|
367
|
-
key: string;
|
|
368
|
-
value: string;
|
|
369
|
-
};
|
|
370
|
-
parentRunId?: string;
|
|
371
|
-
projectId?: string;
|
|
372
|
-
chatId?: string;
|
|
373
|
-
layer?: RunLayer;
|
|
374
|
-
}
|
|
375
|
-
interface SpanFilter {
|
|
376
|
-
runId?: string;
|
|
377
|
-
parentSpanId?: string;
|
|
378
|
-
kind?: SpanKind;
|
|
379
|
-
name?: string;
|
|
380
|
-
toolName?: string;
|
|
381
|
-
judgeId?: string;
|
|
382
|
-
since?: number;
|
|
383
|
-
until?: number;
|
|
384
|
-
}
|
|
385
|
-
interface EventFilter {
|
|
386
|
-
runId?: string;
|
|
387
|
-
spanId?: string;
|
|
388
|
-
kind?: EventKind;
|
|
389
|
-
since?: number;
|
|
390
|
-
until?: number;
|
|
391
|
-
}
|
|
392
|
-
interface TraceStore {
|
|
393
|
-
appendRun(run: Run): Promise<void>;
|
|
394
|
-
updateRun(runId: string, patch: Partial<Run>): Promise<void>;
|
|
395
|
-
appendSpan(span: Span): Promise<void>;
|
|
396
|
-
updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
|
|
397
|
-
appendEvent(event: TraceEvent$1): Promise<void>;
|
|
398
|
-
appendArtifact(artifact: Artifact): Promise<void>;
|
|
399
|
-
appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
|
|
400
|
-
getRun(runId: string): Promise<Run | undefined>;
|
|
401
|
-
listRuns(filter?: RunFilter): Promise<Run[]>;
|
|
402
|
-
spans(filter?: SpanFilter): Promise<Span[]>;
|
|
403
|
-
events(filter?: EventFilter): Promise<TraceEvent$1[]>;
|
|
404
|
-
budget(runId: string): Promise<BudgetLedgerEntry[]>;
|
|
405
|
-
artifacts(runId: string): Promise<Artifact[]>;
|
|
406
|
-
}
|
|
407
|
-
|
|
408
|
-
/**
|
|
409
|
-
* Policy-based agent control runtime.
|
|
410
|
-
*
|
|
411
|
-
* This is the minimal reusable loop behind driver-agent patterns:
|
|
412
|
-
*
|
|
413
|
-
* observe state -> validate -> decide next action -> act -> observe -> ...
|
|
414
|
-
*
|
|
415
|
-
* It deliberately does not model named "topologies". Direct execution,
|
|
416
|
-
* critic/revise, driver intervention, specialist calls, and human escalation
|
|
417
|
-
* are all just actions chosen by the control policy.
|
|
418
|
-
*/
|
|
419
|
-
|
|
420
|
-
type ControlSeverity = 'info' | 'warning' | 'error' | 'critical';
|
|
421
|
-
interface ControlEvalResult {
|
|
422
|
-
/** Stable validator or judge id. */
|
|
423
|
-
id: string;
|
|
424
|
-
/** Whether this check passed. */
|
|
425
|
-
passed: boolean;
|
|
426
|
-
/** Optional normalized score. 1 = best, 0 = worst. */
|
|
427
|
-
score?: number;
|
|
428
|
-
/** Objective validators should usually be "error" or "critical" when failed. */
|
|
429
|
-
severity?: ControlSeverity;
|
|
430
|
-
/** Human-readable result. */
|
|
431
|
-
detail?: string;
|
|
432
|
-
/** Small evidence string or pointer. Avoid large payloads. */
|
|
433
|
-
evidence?: string;
|
|
434
|
-
/** True when the result came from deterministic state, not LLM judgment. */
|
|
435
|
-
objective?: boolean;
|
|
436
|
-
/** Structured details for downstream control policies and reports. */
|
|
437
|
-
metadata?: Record<string, unknown>;
|
|
438
|
-
}
|
|
439
|
-
|
|
440
|
-
/**
|
|
441
|
-
* Dataset — versioned, sliceable, content-hashed scenario collection.
|
|
442
|
-
*
|
|
443
|
-
* Scenarios stop being ephemeral arrays and become first-class
|
|
444
|
-
* artifacts. Every Dataset carries:
|
|
445
|
-
* - content hash (sha256 over canonicalized scenario array)
|
|
446
|
-
* - provenance (contributor, createdAt, sourceUrl)
|
|
447
|
-
* - split labels (train | dev | test | holdout)
|
|
448
|
-
* - difficulty tiers (easy | medium | hard | extreme)
|
|
449
|
-
* - tags (free-form, per-scenario)
|
|
450
|
-
*
|
|
451
|
-
* `Dataset.slice({ difficulty, split, holdout, seed })` returns a
|
|
452
|
-
* deterministic, reproducible subset. Holdout slices are locked: you
|
|
453
|
-
* can read them but `mutate` throws, which prevents "oh I'll just
|
|
454
|
-
* tweak that one scenario" contamination drift.
|
|
455
|
-
*/
|
|
456
|
-
type DatasetSplit = 'train' | 'dev' | 'test' | 'holdout';
|
|
457
|
-
|
|
458
|
-
type FeedbackArtifactType = 'text' | 'code' | 'plan' | 'research' | 'action' | 'ui' | 'decision' | 'data' | 'other';
|
|
459
|
-
type FeedbackLabelSource = 'user' | 'judge' | 'environment' | 'metric' | 'policy' | 'system';
|
|
460
|
-
type FeedbackLabelKind = 'approve' | 'reject' | 'select' | 'edit' | 'rank' | 'rate' | 'comment' | 'metric_outcome' | 'policy_block' | 'revision_request';
|
|
461
|
-
type FeedbackSeverity = 'info' | 'warning' | 'error' | 'critical';
|
|
462
|
-
interface FeedbackTask {
|
|
463
|
-
intent: string;
|
|
464
|
-
context?: unknown;
|
|
465
|
-
}
|
|
466
|
-
interface ProposedSideEffect {
|
|
467
|
-
type: string;
|
|
468
|
-
risk?: 'low' | 'medium' | 'high';
|
|
469
|
-
costUsd?: number;
|
|
470
|
-
externalSideEffect?: boolean;
|
|
471
|
-
requiresApproval?: boolean;
|
|
472
|
-
metadata?: Record<string, unknown>;
|
|
473
|
-
}
|
|
474
|
-
interface FeedbackLabel {
|
|
475
|
-
id?: string;
|
|
476
|
-
source: FeedbackLabelSource;
|
|
477
|
-
kind: FeedbackLabelKind;
|
|
478
|
-
value: unknown;
|
|
479
|
-
reason?: string;
|
|
480
|
-
severity?: FeedbackSeverity;
|
|
481
|
-
createdAt: string;
|
|
482
|
-
metadata?: Record<string, unknown>;
|
|
483
|
-
}
|
|
484
|
-
interface FeedbackAttempt {
|
|
485
|
-
id: string;
|
|
486
|
-
stepIndex: number;
|
|
487
|
-
artifactType: FeedbackArtifactType;
|
|
488
|
-
artifact: unknown;
|
|
489
|
-
options?: unknown[];
|
|
490
|
-
proposedAction?: ProposedSideEffect;
|
|
491
|
-
evals?: ControlEvalResult[];
|
|
492
|
-
feedback?: FeedbackLabel[];
|
|
493
|
-
createdAt: string;
|
|
494
|
-
metadata?: Record<string, unknown>;
|
|
495
|
-
}
|
|
496
|
-
interface FeedbackOutcome {
|
|
497
|
-
success?: boolean;
|
|
498
|
-
score?: number;
|
|
499
|
-
metrics?: Record<string, number>;
|
|
500
|
-
costUsd?: number;
|
|
501
|
-
detail?: string;
|
|
502
|
-
observedAt?: string;
|
|
503
|
-
metadata?: Record<string, unknown>;
|
|
504
|
-
}
|
|
505
|
-
interface FeedbackTrajectory$1 {
|
|
506
|
-
id: string;
|
|
507
|
-
projectId?: string;
|
|
508
|
-
scenarioId?: string;
|
|
509
|
-
task: FeedbackTask;
|
|
510
|
-
attempts: FeedbackAttempt[];
|
|
511
|
-
labels: FeedbackLabel[];
|
|
512
|
-
outcome?: FeedbackOutcome;
|
|
513
|
-
split?: DatasetSplit;
|
|
514
|
-
tags?: Record<string, string>;
|
|
515
|
-
createdAt: string;
|
|
516
|
-
updatedAt?: string;
|
|
517
|
-
metadata?: Record<string, unknown>;
|
|
518
|
-
}
|
|
519
|
-
interface FeedbackTrajectoryStore {
|
|
520
|
-
save(trajectory: FeedbackTrajectory$1): Promise<void>;
|
|
521
|
-
get(id: string): Promise<FeedbackTrajectory$1 | null>;
|
|
522
|
-
list(filter?: FeedbackTrajectoryFilter): Promise<FeedbackTrajectory$1[]>;
|
|
523
|
-
appendAttempt(id: string, attempt: FeedbackAttempt): Promise<FeedbackTrajectory$1>;
|
|
524
|
-
appendLabel(id: string, label: FeedbackLabel, attemptId?: string): Promise<FeedbackTrajectory$1>;
|
|
525
|
-
}
|
|
526
|
-
interface FeedbackTrajectoryFilter {
|
|
527
|
-
projectId?: string;
|
|
528
|
-
scenarioId?: string;
|
|
529
|
-
split?: DatasetSplit;
|
|
530
|
-
tag?: [string, string];
|
|
531
|
-
}
|
|
532
|
-
|
|
533
|
-
/**
|
|
534
|
-
* RawProviderSink — first-class persistence for the actual HTTP-level
|
|
535
|
-
* request/response bodies of every LLM provider call.
|
|
536
|
-
*
|
|
537
|
-
* Why this is a separate sink from the structured `LlmSpan`:
|
|
538
|
-
*
|
|
539
|
-
* - `LlmSpan` records the *intent* — model name, messages, output text,
|
|
540
|
-
* usage. It's what dashboards read; it's NOT enough for forensics.
|
|
541
|
-
* - When a downstream consumer reports "the verifier used the wrong route"
|
|
542
|
-
* or "tokens look right but reasoning was missing," the only way to
|
|
543
|
-
* answer is the raw HTTP body. Span fields can lie (a proxy can echo
|
|
544
|
-
* a different `model` value than what actually answered); the raw
|
|
545
|
-
* response is ground truth.
|
|
546
|
-
*
|
|
547
|
-
* Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
|
|
548
|
-
* matrix runner / BuilderSession sets it up automatically) and every
|
|
549
|
-
* request, response, and error is recorded — including retries, with the
|
|
550
|
-
* attempt index attached so a flaky call's full event chain is recoverable.
|
|
551
|
-
*
|
|
552
|
-
* Redaction is enforced at sink time. The default redactor strips
|
|
553
|
-
* `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
|
|
554
|
-
* payload field whose key matches `apiKey | api_key | bearer | password |
|
|
555
|
-
* secret | token` (case-insensitive). Override via the sink constructor or
|
|
556
|
-
* the per-call `redactor`. The `redactedFields` array on the persisted
|
|
557
|
-
* event lets a reviewer see what was stripped without exposing the values.
|
|
558
|
-
*/
|
|
559
|
-
type RawProviderDirection = 'request' | 'response' | 'error';
|
|
560
|
-
interface RawProviderEvent {
|
|
561
|
-
/** Stable id. Generated by the sink if omitted. */
|
|
562
|
-
eventId: string;
|
|
563
|
-
/** Trace context populated by `LlmClient` when the call is wrapped in a span. */
|
|
564
|
-
runId?: string;
|
|
565
|
-
spanId?: string;
|
|
566
|
-
/**
|
|
567
|
-
* Logical provider name. Free-form so callers can use whatever id matches
|
|
568
|
-
* their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
|
|
569
|
-
* omitted, derived from `baseUrl` in `LlmClientOptions`.
|
|
570
|
-
*/
|
|
571
|
-
provider: string;
|
|
572
|
-
model: string;
|
|
573
|
-
/** Endpoint path, e.g. `'/v1/chat/completions'`. */
|
|
574
|
-
endpoint: string;
|
|
575
|
-
/** Base URL used for the call (already-normalised — no trailing slash). */
|
|
576
|
-
baseUrl: string;
|
|
577
|
-
/** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
|
|
578
|
-
attemptIndex: number;
|
|
579
|
-
direction: RawProviderDirection;
|
|
580
|
-
/** Unix ms. */
|
|
581
|
-
timestamp: number;
|
|
582
|
-
/** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
|
|
583
|
-
durationMs?: number;
|
|
584
|
-
statusCode?: number;
|
|
585
|
-
requestHeaders?: Record<string, string>;
|
|
586
|
-
requestBody?: unknown;
|
|
587
|
-
responseHeaders?: Record<string, string>;
|
|
588
|
-
responseBody?: unknown;
|
|
589
|
-
/** Set on `direction: 'error'` events. */
|
|
590
|
-
errorMessage?: string;
|
|
591
|
-
/** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
|
|
592
|
-
redactedFields: string[];
|
|
593
|
-
}
|
|
594
|
-
interface RawProviderSinkFilter {
|
|
595
|
-
runId?: string;
|
|
596
|
-
spanId?: string;
|
|
597
|
-
direction?: RawProviderDirection;
|
|
598
|
-
attemptIndex?: number;
|
|
599
|
-
}
|
|
600
|
-
interface RawProviderSink {
|
|
601
|
-
record(event: RawProviderEvent): Promise<void>;
|
|
602
|
-
/** Optional listing — implementations that durably persist (file, db) should support this. */
|
|
603
|
-
list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
604
|
-
/** Optional teardown for backed implementations. */
|
|
605
|
-
close?(): Promise<void>;
|
|
606
|
-
}
|
|
607
|
-
type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
|
|
608
|
-
|
|
609
|
-
/**
|
|
610
|
-
* LLM client with graceful degrade.
|
|
611
|
-
*
|
|
612
|
-
* OpenAI-compatible `/v1/chat/completions` client with:
|
|
613
|
-
* - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
|
|
614
|
-
* - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
|
|
615
|
-
* - Graceful json_schema → json_object degrade on 400 with schema-reject body.
|
|
616
|
-
* - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
|
|
617
|
-
* - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
|
|
618
|
-
* directly, cli-bridge subscriptions, and any router that speaks the spec.
|
|
619
|
-
*
|
|
620
|
-
* Usage:
|
|
621
|
-
* const { value, result } = await callLlmJson<MyType>(
|
|
622
|
-
* { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
|
|
623
|
-
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
624
|
-
* )
|
|
625
|
-
*
|
|
626
|
-
* This is THE llm-calling seam for agent-eval primitives that need structured
|
|
627
|
-
* output (semantic concept judge, reviewer directives, critic scores). Primitives
|
|
628
|
-
* that need free-form text use `callLlm` and parse output themselves.
|
|
629
|
-
*/
|
|
630
|
-
|
|
631
|
-
interface LlmClientOptions {
|
|
632
|
-
/** Base URL (without trailing slash). Must end at the `/v1` prefix. */
|
|
633
|
-
baseUrl?: string;
|
|
634
|
-
/** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
|
|
635
|
-
apiKey?: string;
|
|
636
|
-
bearer?: string;
|
|
637
|
-
/** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
|
|
638
|
-
authHeader?: {
|
|
639
|
-
name: string;
|
|
640
|
-
value: string;
|
|
641
|
-
};
|
|
642
|
-
/** Stable provider idempotency key, reused across retries of this logical call. */
|
|
643
|
-
idempotencyKey?: string;
|
|
644
|
-
/** Default timeout in ms. Per-call can override. */
|
|
645
|
-
defaultTimeoutMs?: number;
|
|
646
|
-
/**
|
|
647
|
-
* Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
|
|
648
|
-
* each attempt's per-attempt timeout controller, so aborting it cancels
|
|
649
|
-
* the in-flight fetch. A caller abort is FATAL: it is not retried even
|
|
650
|
-
* though an AbortError otherwise matches the transient patterns.
|
|
651
|
-
*/
|
|
652
|
-
signal?: AbortSignal;
|
|
653
|
-
/**
|
|
654
|
-
* Cross-attempt wall-clock budget in ms, measured from the first attempt.
|
|
655
|
-
* Before launching each attempt the loop checks the remaining budget and
|
|
656
|
-
* stops retrying once it is exhausted, rather than waiting the full
|
|
657
|
-
* per-attempt timeout on every retry. Bounds total time independent of
|
|
658
|
-
* total attempts × `timeoutMs`.
|
|
659
|
-
*/
|
|
660
|
-
deadlineMs?: number;
|
|
661
|
-
/** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
|
|
662
|
-
maxRetries?: number;
|
|
663
|
-
/** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
|
|
664
|
-
fetch?: typeof fetch;
|
|
665
|
-
/**
|
|
666
|
-
* Optional raw HTTP capture sink. When provided, every request, response,
|
|
667
|
-
* and error (across all retry attempts) is recorded to the sink, with auth
|
|
668
|
-
* headers and credential-shaped body fields redacted by default. This is
|
|
669
|
-
* the layer-1 forensics primitive: structured `LlmSpan`s record intent,
|
|
670
|
-
* raw events record what actually crossed the wire.
|
|
671
|
-
*/
|
|
672
|
-
rawSink?: RawProviderSink;
|
|
673
|
-
/**
|
|
674
|
-
* Logical provider id attached to raw events. When omitted, derived from
|
|
675
|
-
* `baseUrl` via `providerFromBaseUrl`.
|
|
676
|
-
*/
|
|
677
|
-
provider?: string;
|
|
678
|
-
/** Trace context attached to raw events; populated by emitter-aware callers. */
|
|
679
|
-
traceContext?: {
|
|
680
|
-
runId?: string;
|
|
681
|
-
spanId?: string;
|
|
682
|
-
};
|
|
683
|
-
/** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
|
|
684
|
-
redactor?: ProviderRedactor;
|
|
685
|
-
}
|
|
686
|
-
|
|
687
|
-
declare const RubricDimensionSchema: z.ZodObject<{
|
|
688
|
-
id: z.ZodString;
|
|
689
|
-
description: z.ZodString;
|
|
690
|
-
weight: z.ZodDefault<z.ZodNumber>;
|
|
691
|
-
min: z.ZodDefault<z.ZodNumber>;
|
|
692
|
-
max: z.ZodDefault<z.ZodNumber>;
|
|
693
|
-
}, z.core.$strip>;
|
|
694
|
-
declare const FailureModeSchema: z.ZodObject<{
|
|
695
|
-
id: z.ZodString;
|
|
696
|
-
description: z.ZodString;
|
|
697
|
-
}, z.core.$strip>;
|
|
698
|
-
declare const RubricSchema: z.ZodObject<{
|
|
699
|
-
name: z.ZodString;
|
|
700
|
-
description: z.ZodString;
|
|
701
|
-
systemPrompt: z.ZodString;
|
|
702
|
-
dimensions: z.ZodArray<z.ZodObject<{
|
|
703
|
-
id: z.ZodString;
|
|
704
|
-
description: z.ZodString;
|
|
705
|
-
weight: z.ZodDefault<z.ZodNumber>;
|
|
706
|
-
min: z.ZodDefault<z.ZodNumber>;
|
|
707
|
-
max: z.ZodDefault<z.ZodNumber>;
|
|
708
|
-
}, z.core.$strip>>;
|
|
709
|
-
failureModes: z.ZodDefault<z.ZodArray<z.ZodObject<{
|
|
710
|
-
id: z.ZodString;
|
|
711
|
-
description: z.ZodString;
|
|
712
|
-
}, z.core.$strip>>>;
|
|
713
|
-
wins: z.ZodDefault<z.ZodArray<z.ZodObject<{
|
|
714
|
-
id: z.ZodString;
|
|
715
|
-
description: z.ZodString;
|
|
716
|
-
}, z.core.$strip>>>;
|
|
717
|
-
}, z.core.$strip>;
|
|
718
|
-
declare const JudgeRequestSchema: z.ZodObject<{
|
|
719
|
-
rubricName: z.ZodOptional<z.ZodString>;
|
|
720
|
-
rubric: z.ZodOptional<z.ZodObject<{
|
|
721
|
-
name: z.ZodString;
|
|
722
|
-
description: z.ZodString;
|
|
723
|
-
systemPrompt: z.ZodString;
|
|
724
|
-
dimensions: z.ZodArray<z.ZodObject<{
|
|
725
|
-
id: z.ZodString;
|
|
726
|
-
description: z.ZodString;
|
|
727
|
-
weight: z.ZodDefault<z.ZodNumber>;
|
|
728
|
-
min: z.ZodDefault<z.ZodNumber>;
|
|
729
|
-
max: z.ZodDefault<z.ZodNumber>;
|
|
730
|
-
}, z.core.$strip>>;
|
|
731
|
-
failureModes: z.ZodDefault<z.ZodArray<z.ZodObject<{
|
|
732
|
-
id: z.ZodString;
|
|
733
|
-
description: z.ZodString;
|
|
734
|
-
}, z.core.$strip>>>;
|
|
735
|
-
wins: z.ZodDefault<z.ZodArray<z.ZodObject<{
|
|
736
|
-
id: z.ZodString;
|
|
737
|
-
description: z.ZodString;
|
|
738
|
-
}, z.core.$strip>>>;
|
|
739
|
-
}, z.core.$strip>>;
|
|
740
|
-
content: z.ZodString;
|
|
741
|
-
context: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
742
|
-
model: z.ZodOptional<z.ZodString>;
|
|
743
|
-
}, z.core.$strip>;
|
|
744
|
-
declare const JudgeResultSchema: z.ZodObject<{
|
|
745
|
-
composite: z.ZodNumber;
|
|
746
|
-
dimensions: z.ZodRecord<z.ZodString, z.ZodNumber>;
|
|
747
|
-
failureModes: z.ZodDefault<z.ZodArray<z.ZodString>>;
|
|
748
|
-
wins: z.ZodDefault<z.ZodArray<z.ZodString>>;
|
|
749
|
-
rationale: z.ZodString;
|
|
750
|
-
rubricVersion: z.ZodString;
|
|
751
|
-
model: z.ZodString;
|
|
752
|
-
durationMs: z.ZodNumber;
|
|
753
|
-
}, z.core.$strip>;
|
|
754
|
-
declare const RubricInfoSchema: z.ZodObject<{
|
|
755
|
-
name: z.ZodString;
|
|
756
|
-
description: z.ZodString;
|
|
757
|
-
dimensions: z.ZodArray<z.ZodObject<{
|
|
758
|
-
id: z.ZodString;
|
|
759
|
-
description: z.ZodString;
|
|
760
|
-
weight: z.ZodNumber;
|
|
761
|
-
}, z.core.$strip>>;
|
|
762
|
-
failureModes: z.ZodDefault<z.ZodArray<z.ZodString>>;
|
|
763
|
-
rubricVersion: z.ZodString;
|
|
764
|
-
}, z.core.$strip>;
|
|
765
|
-
declare const ListRubricsResponseSchema: z.ZodObject<{
|
|
766
|
-
rubrics: z.ZodArray<z.ZodObject<{
|
|
767
|
-
name: z.ZodString;
|
|
768
|
-
description: z.ZodString;
|
|
769
|
-
dimensions: z.ZodArray<z.ZodObject<{
|
|
770
|
-
id: z.ZodString;
|
|
771
|
-
description: z.ZodString;
|
|
772
|
-
weight: z.ZodNumber;
|
|
773
|
-
}, z.core.$strip>>;
|
|
774
|
-
failureModes: z.ZodDefault<z.ZodArray<z.ZodString>>;
|
|
775
|
-
rubricVersion: z.ZodString;
|
|
776
|
-
}, z.core.$strip>>;
|
|
777
|
-
}, z.core.$strip>;
|
|
778
|
-
declare const VersionResponseSchema: z.ZodObject<{
|
|
779
|
-
package: z.ZodString;
|
|
780
|
-
version: z.ZodString;
|
|
781
|
-
wireVersion: z.ZodString;
|
|
782
|
-
apiSurface: z.ZodArray<z.ZodString>;
|
|
783
|
-
}, z.core.$strip>;
|
|
784
|
-
declare const HealthResponseSchema: z.ZodObject<{
|
|
785
|
-
status: z.ZodLiteral<"ok">;
|
|
786
|
-
uptimeSec: z.ZodNumber;
|
|
787
|
-
}, z.core.$strip>;
|
|
788
|
-
/**
|
|
789
|
-
* Minimal `TraceEvent` shape that the production runtime emits.
|
|
790
|
-
* Matches `trace/schema.ts` `TraceEvent` but is duplicated here as a
|
|
791
|
-
* wire schema so non-TypeScript clients can validate without depending
|
|
792
|
-
* on internal types.
|
|
793
|
-
*/
|
|
794
|
-
declare const TraceEventSchema: z.ZodObject<{
|
|
795
|
-
eventId: z.ZodString;
|
|
796
|
-
runId: z.ZodString;
|
|
797
|
-
spanId: z.ZodOptional<z.ZodString>;
|
|
798
|
-
kind: z.ZodEnum<{
|
|
799
|
-
error: "error";
|
|
800
|
-
custom: "custom";
|
|
801
|
-
policy_violation: "policy_violation";
|
|
802
|
-
log: "log";
|
|
803
|
-
budget_decrement: "budget_decrement";
|
|
804
|
-
budget_breach: "budget_breach";
|
|
805
|
-
state_mutation: "state_mutation";
|
|
806
|
-
redaction_applied: "redaction_applied";
|
|
807
|
-
}>;
|
|
808
|
-
timestamp: z.ZodNumber;
|
|
809
|
-
payload: z.ZodRecord<z.ZodString, z.ZodUnknown>;
|
|
810
|
-
}, z.core.$strip>;
|
|
811
|
-
declare const TracesIngestRequestSchema: z.ZodObject<{
|
|
812
|
-
events: z.ZodArray<z.ZodObject<{
|
|
813
|
-
eventId: z.ZodString;
|
|
814
|
-
runId: z.ZodString;
|
|
815
|
-
spanId: z.ZodOptional<z.ZodString>;
|
|
816
|
-
kind: z.ZodEnum<{
|
|
817
|
-
error: "error";
|
|
818
|
-
custom: "custom";
|
|
819
|
-
policy_violation: "policy_violation";
|
|
820
|
-
log: "log";
|
|
821
|
-
budget_decrement: "budget_decrement";
|
|
822
|
-
budget_breach: "budget_breach";
|
|
823
|
-
state_mutation: "state_mutation";
|
|
824
|
-
redaction_applied: "redaction_applied";
|
|
825
|
-
}>;
|
|
826
|
-
timestamp: z.ZodNumber;
|
|
827
|
-
payload: z.ZodRecord<z.ZodString, z.ZodUnknown>;
|
|
828
|
-
}, z.core.$strip>>;
|
|
829
|
-
}, z.core.$strip>;
|
|
830
|
-
declare const TracesIngestResponseSchema: z.ZodObject<{
|
|
831
|
-
accepted: z.ZodNumber;
|
|
832
|
-
rejected: z.ZodNumber;
|
|
833
|
-
errors: z.ZodDefault<z.ZodArray<z.ZodObject<{
|
|
834
|
-
eventId: z.ZodString;
|
|
835
|
-
message: z.ZodString;
|
|
836
|
-
}, z.core.$strip>>>;
|
|
837
|
-
}, z.core.$strip>;
|
|
838
|
-
declare const FeedbackLabelSchema: z.ZodObject<{
|
|
839
|
-
id: z.ZodOptional<z.ZodString>;
|
|
840
|
-
source: z.ZodEnum<{
|
|
841
|
-
judge: "judge";
|
|
842
|
-
user: "user";
|
|
843
|
-
system: "system";
|
|
844
|
-
policy: "policy";
|
|
845
|
-
environment: "environment";
|
|
846
|
-
metric: "metric";
|
|
847
|
-
}>;
|
|
848
|
-
kind: z.ZodEnum<{
|
|
849
|
-
approve: "approve";
|
|
850
|
-
reject: "reject";
|
|
851
|
-
select: "select";
|
|
852
|
-
edit: "edit";
|
|
853
|
-
rank: "rank";
|
|
854
|
-
rate: "rate";
|
|
855
|
-
comment: "comment";
|
|
856
|
-
metric_outcome: "metric_outcome";
|
|
857
|
-
policy_block: "policy_block";
|
|
858
|
-
revision_request: "revision_request";
|
|
859
|
-
}>;
|
|
860
|
-
value: z.ZodUnknown;
|
|
861
|
-
reason: z.ZodOptional<z.ZodString>;
|
|
862
|
-
severity: z.ZodOptional<z.ZodEnum<{
|
|
863
|
-
error: "error";
|
|
864
|
-
info: "info";
|
|
865
|
-
critical: "critical";
|
|
866
|
-
warning: "warning";
|
|
867
|
-
}>>;
|
|
868
|
-
createdAt: z.ZodString;
|
|
869
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
870
|
-
}, z.core.$strip>;
|
|
871
|
-
declare const FeedbackAttemptSchema: z.ZodObject<{
|
|
872
|
-
id: z.ZodString;
|
|
873
|
-
stepIndex: z.ZodNumber;
|
|
874
|
-
artifactType: z.ZodEnum<{
|
|
875
|
-
text: "text";
|
|
876
|
-
code: "code";
|
|
877
|
-
action: "action";
|
|
878
|
-
decision: "decision";
|
|
879
|
-
plan: "plan";
|
|
880
|
-
research: "research";
|
|
881
|
-
ui: "ui";
|
|
882
|
-
data: "data";
|
|
883
|
-
other: "other";
|
|
884
|
-
}>;
|
|
885
|
-
artifact: z.ZodUnknown;
|
|
886
|
-
options: z.ZodOptional<z.ZodArray<z.ZodUnknown>>;
|
|
887
|
-
proposedAction: z.ZodOptional<z.ZodObject<{
|
|
888
|
-
type: z.ZodString;
|
|
889
|
-
risk: z.ZodOptional<z.ZodEnum<{
|
|
890
|
-
medium: "medium";
|
|
891
|
-
low: "low";
|
|
892
|
-
high: "high";
|
|
893
|
-
}>>;
|
|
894
|
-
costUsd: z.ZodOptional<z.ZodNumber>;
|
|
895
|
-
externalSideEffect: z.ZodOptional<z.ZodBoolean>;
|
|
896
|
-
requiresApproval: z.ZodOptional<z.ZodBoolean>;
|
|
897
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
898
|
-
}, z.core.$strip>>;
|
|
899
|
-
feedback: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
900
|
-
id: z.ZodOptional<z.ZodString>;
|
|
901
|
-
source: z.ZodEnum<{
|
|
902
|
-
judge: "judge";
|
|
903
|
-
user: "user";
|
|
904
|
-
system: "system";
|
|
905
|
-
policy: "policy";
|
|
906
|
-
environment: "environment";
|
|
907
|
-
metric: "metric";
|
|
908
|
-
}>;
|
|
909
|
-
kind: z.ZodEnum<{
|
|
910
|
-
approve: "approve";
|
|
911
|
-
reject: "reject";
|
|
912
|
-
select: "select";
|
|
913
|
-
edit: "edit";
|
|
914
|
-
rank: "rank";
|
|
915
|
-
rate: "rate";
|
|
916
|
-
comment: "comment";
|
|
917
|
-
metric_outcome: "metric_outcome";
|
|
918
|
-
policy_block: "policy_block";
|
|
919
|
-
revision_request: "revision_request";
|
|
920
|
-
}>;
|
|
921
|
-
value: z.ZodUnknown;
|
|
922
|
-
reason: z.ZodOptional<z.ZodString>;
|
|
923
|
-
severity: z.ZodOptional<z.ZodEnum<{
|
|
924
|
-
error: "error";
|
|
925
|
-
info: "info";
|
|
926
|
-
critical: "critical";
|
|
927
|
-
warning: "warning";
|
|
928
|
-
}>>;
|
|
929
|
-
createdAt: z.ZodString;
|
|
930
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
931
|
-
}, z.core.$strip>>>;
|
|
932
|
-
createdAt: z.ZodString;
|
|
933
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
934
|
-
}, z.core.$strip>;
|
|
935
|
-
declare const FeedbackTrajectorySchema: z.ZodObject<{
|
|
936
|
-
id: z.ZodString;
|
|
937
|
-
projectId: z.ZodOptional<z.ZodString>;
|
|
938
|
-
scenarioId: z.ZodOptional<z.ZodString>;
|
|
939
|
-
task: z.ZodObject<{
|
|
940
|
-
intent: z.ZodString;
|
|
941
|
-
context: z.ZodOptional<z.ZodUnknown>;
|
|
942
|
-
}, z.core.$strip>;
|
|
943
|
-
attempts: z.ZodDefault<z.ZodArray<z.ZodObject<{
|
|
944
|
-
id: z.ZodString;
|
|
945
|
-
stepIndex: z.ZodNumber;
|
|
946
|
-
artifactType: z.ZodEnum<{
|
|
947
|
-
text: "text";
|
|
948
|
-
code: "code";
|
|
949
|
-
action: "action";
|
|
950
|
-
decision: "decision";
|
|
951
|
-
plan: "plan";
|
|
952
|
-
research: "research";
|
|
953
|
-
ui: "ui";
|
|
954
|
-
data: "data";
|
|
955
|
-
other: "other";
|
|
956
|
-
}>;
|
|
957
|
-
artifact: z.ZodUnknown;
|
|
958
|
-
options: z.ZodOptional<z.ZodArray<z.ZodUnknown>>;
|
|
959
|
-
proposedAction: z.ZodOptional<z.ZodObject<{
|
|
960
|
-
type: z.ZodString;
|
|
961
|
-
risk: z.ZodOptional<z.ZodEnum<{
|
|
962
|
-
medium: "medium";
|
|
963
|
-
low: "low";
|
|
964
|
-
high: "high";
|
|
965
|
-
}>>;
|
|
966
|
-
costUsd: z.ZodOptional<z.ZodNumber>;
|
|
967
|
-
externalSideEffect: z.ZodOptional<z.ZodBoolean>;
|
|
968
|
-
requiresApproval: z.ZodOptional<z.ZodBoolean>;
|
|
969
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
970
|
-
}, z.core.$strip>>;
|
|
971
|
-
feedback: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
972
|
-
id: z.ZodOptional<z.ZodString>;
|
|
973
|
-
source: z.ZodEnum<{
|
|
974
|
-
judge: "judge";
|
|
975
|
-
user: "user";
|
|
976
|
-
system: "system";
|
|
977
|
-
policy: "policy";
|
|
978
|
-
environment: "environment";
|
|
979
|
-
metric: "metric";
|
|
980
|
-
}>;
|
|
981
|
-
kind: z.ZodEnum<{
|
|
982
|
-
approve: "approve";
|
|
983
|
-
reject: "reject";
|
|
984
|
-
select: "select";
|
|
985
|
-
edit: "edit";
|
|
986
|
-
rank: "rank";
|
|
987
|
-
rate: "rate";
|
|
988
|
-
comment: "comment";
|
|
989
|
-
metric_outcome: "metric_outcome";
|
|
990
|
-
policy_block: "policy_block";
|
|
991
|
-
revision_request: "revision_request";
|
|
992
|
-
}>;
|
|
993
|
-
value: z.ZodUnknown;
|
|
994
|
-
reason: z.ZodOptional<z.ZodString>;
|
|
995
|
-
severity: z.ZodOptional<z.ZodEnum<{
|
|
996
|
-
error: "error";
|
|
997
|
-
info: "info";
|
|
998
|
-
critical: "critical";
|
|
999
|
-
warning: "warning";
|
|
1000
|
-
}>>;
|
|
1001
|
-
createdAt: z.ZodString;
|
|
1002
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
1003
|
-
}, z.core.$strip>>>;
|
|
1004
|
-
createdAt: z.ZodString;
|
|
1005
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
1006
|
-
}, z.core.$strip>>>;
|
|
1007
|
-
labels: z.ZodDefault<z.ZodArray<z.ZodObject<{
|
|
1008
|
-
id: z.ZodOptional<z.ZodString>;
|
|
1009
|
-
source: z.ZodEnum<{
|
|
1010
|
-
judge: "judge";
|
|
1011
|
-
user: "user";
|
|
1012
|
-
system: "system";
|
|
1013
|
-
policy: "policy";
|
|
1014
|
-
environment: "environment";
|
|
1015
|
-
metric: "metric";
|
|
1016
|
-
}>;
|
|
1017
|
-
kind: z.ZodEnum<{
|
|
1018
|
-
approve: "approve";
|
|
1019
|
-
reject: "reject";
|
|
1020
|
-
select: "select";
|
|
1021
|
-
edit: "edit";
|
|
1022
|
-
rank: "rank";
|
|
1023
|
-
rate: "rate";
|
|
1024
|
-
comment: "comment";
|
|
1025
|
-
metric_outcome: "metric_outcome";
|
|
1026
|
-
policy_block: "policy_block";
|
|
1027
|
-
revision_request: "revision_request";
|
|
1028
|
-
}>;
|
|
1029
|
-
value: z.ZodUnknown;
|
|
1030
|
-
reason: z.ZodOptional<z.ZodString>;
|
|
1031
|
-
severity: z.ZodOptional<z.ZodEnum<{
|
|
1032
|
-
error: "error";
|
|
1033
|
-
info: "info";
|
|
1034
|
-
critical: "critical";
|
|
1035
|
-
warning: "warning";
|
|
1036
|
-
}>>;
|
|
1037
|
-
createdAt: z.ZodString;
|
|
1038
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
1039
|
-
}, z.core.$strip>>>;
|
|
1040
|
-
outcome: z.ZodOptional<z.ZodObject<{
|
|
1041
|
-
success: z.ZodOptional<z.ZodBoolean>;
|
|
1042
|
-
score: z.ZodOptional<z.ZodNumber>;
|
|
1043
|
-
metrics: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodNumber>>;
|
|
1044
|
-
costUsd: z.ZodOptional<z.ZodNumber>;
|
|
1045
|
-
detail: z.ZodOptional<z.ZodString>;
|
|
1046
|
-
observedAt: z.ZodOptional<z.ZodString>;
|
|
1047
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
1048
|
-
}, z.core.$strip>>;
|
|
1049
|
-
split: z.ZodOptional<z.ZodEnum<{
|
|
1050
|
-
train: "train";
|
|
1051
|
-
dev: "dev";
|
|
1052
|
-
test: "test";
|
|
1053
|
-
holdout: "holdout";
|
|
1054
|
-
}>>;
|
|
1055
|
-
tags: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
1056
|
-
createdAt: z.ZodString;
|
|
1057
|
-
updatedAt: z.ZodOptional<z.ZodString>;
|
|
1058
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
1059
|
-
}, z.core.$strip>;
|
|
1060
|
-
declare const FeedbackIngestResponseSchema: z.ZodObject<{
|
|
1061
|
-
id: z.ZodString;
|
|
1062
|
-
persisted: z.ZodBoolean;
|
|
1063
|
-
}, z.core.$strip>;
|
|
1064
|
-
type TraceEvent = z.infer<typeof TraceEventSchema>;
|
|
1065
|
-
type TracesIngestRequest = z.infer<typeof TracesIngestRequestSchema>;
|
|
1066
|
-
type TracesIngestResponse = z.infer<typeof TracesIngestResponseSchema>;
|
|
1067
|
-
type FeedbackTrajectory = z.infer<typeof FeedbackTrajectorySchema>;
|
|
1068
|
-
type FeedbackIngestResponse = z.infer<typeof FeedbackIngestResponseSchema>;
|
|
1069
|
-
declare const ErrorResponseSchema: z.ZodObject<{
|
|
1070
|
-
error: z.ZodObject<{
|
|
1071
|
-
code: z.ZodString;
|
|
1072
|
-
message: z.ZodString;
|
|
1073
|
-
details: z.ZodOptional<z.ZodUnknown>;
|
|
1074
|
-
}, z.core.$strip>;
|
|
1075
|
-
}, z.core.$strip>;
|
|
1076
|
-
type RubricDimension = z.infer<typeof RubricDimensionSchema>;
|
|
1077
|
-
type FailureMode = z.infer<typeof FailureModeSchema>;
|
|
1078
|
-
type Rubric = z.infer<typeof RubricSchema>;
|
|
1079
|
-
type JudgeRequest = z.infer<typeof JudgeRequestSchema>;
|
|
1080
|
-
type JudgeResult = z.infer<typeof JudgeResultSchema>;
|
|
1081
|
-
type RubricInfo = z.infer<typeof RubricInfoSchema>;
|
|
1082
|
-
type ListRubricsResponse = z.infer<typeof ListRubricsResponseSchema>;
|
|
1083
|
-
type VersionResponse = z.infer<typeof VersionResponseSchema>;
|
|
1084
|
-
type ErrorResponse = z.infer<typeof ErrorResponseSchema>;
|
|
1085
|
-
/**
|
|
1086
|
-
* Bump on any breaking change to a request/response schema.
|
|
1087
|
-
* Non-breaking (additive) changes don't require a bump.
|
|
1088
|
-
*/
|
|
1089
|
-
declare const WIRE_VERSION = "1.0.0";
|
|
1090
|
-
/**
|
|
1091
|
-
* Stable hash of a rubric. Used to make scores comparable across runs:
|
|
1092
|
-
* if the rubricVersion matches, the rubric was identical.
|
|
1093
|
-
*/
|
|
1094
|
-
declare function hashRubric(rubric: Rubric): string;
|
|
1095
|
-
|
|
1096
|
-
/**
|
|
1097
|
-
* Pure handler functions — the "business logic" behind every wire-protocol
|
|
1098
|
-
* method. The HTTP server (`server.ts`) and the stdio RPC (`rpc.ts`) both
|
|
1099
|
-
* call these. Tests call these directly without spinning a server.
|
|
1100
|
-
*
|
|
1101
|
-
* Each handler:
|
|
1102
|
-
* - Takes a parsed request (already Zod-validated by the transport).
|
|
1103
|
-
* - Returns a result that matches the response schema.
|
|
1104
|
-
* - Throws `WireError` for caller-fixable errors (404, 400, 422).
|
|
1105
|
-
* - Lets unexpected errors bubble — the transport maps them to 500.
|
|
1106
|
-
*/
|
|
1107
|
-
|
|
1108
|
-
/** Caller-fixable error. The transport renders this to 4xx + ErrorResponse. */
|
|
1109
|
-
declare class WireError extends Error {
|
|
1110
|
-
readonly code: string;
|
|
1111
|
-
readonly status: number;
|
|
1112
|
-
readonly details?: unknown | undefined;
|
|
1113
|
-
constructor(code: string, message: string, status?: number, details?: unknown | undefined);
|
|
1114
|
-
}
|
|
1115
|
-
interface HandleJudgeOptions {
|
|
1116
|
-
costLedger?: CostLedgerHandle;
|
|
1117
|
-
costPhase?: string;
|
|
1118
|
-
llm?: LlmClientOptions;
|
|
1119
|
-
signal?: AbortSignal;
|
|
1120
|
-
}
|
|
1121
|
-
declare function handleJudge(req: JudgeRequest, options?: HandleJudgeOptions): Promise<JudgeResult>;
|
|
1122
|
-
declare function handleListRubrics(): ListRubricsResponse;
|
|
1123
|
-
declare function handleVersion(): VersionResponse;
|
|
1124
|
-
/**
|
|
1125
|
-
* Pluggable stores the wire layer routes ingestion writes into. Both
|
|
1126
|
-
* are optional — when omitted, the corresponding endpoint returns 503.
|
|
1127
|
-
*
|
|
1128
|
-
* Production deployments wire a `FileSystemTraceStore` and
|
|
1129
|
-
* `FileSystemFeedbackTrajectoryStore` here. Tests substitute in-memory
|
|
1130
|
-
* stores.
|
|
1131
|
-
*/
|
|
1132
|
-
interface IngestionStores {
|
|
1133
|
-
traceStore?: TraceStore;
|
|
1134
|
-
feedbackStore?: FeedbackTrajectoryStore;
|
|
1135
|
-
}
|
|
1136
|
-
/**
|
|
1137
|
-
* `POST /v1/traces/ingest` — accept a batch of `TraceEvent`s from the
|
|
1138
|
-
* production runtime. Best-effort: each event is appended independently;
|
|
1139
|
-
* one bad event does not poison the batch.
|
|
1140
|
-
*
|
|
1141
|
-
* Idempotency: the underlying store is append-only; consumers retrying
|
|
1142
|
-
* the same payload will get duplicate events. Consumers should
|
|
1143
|
-
* de-duplicate by `eventId` downstream — production traces frequently
|
|
1144
|
-
* land via at-least-once buses (Kafka, SQS) where dedup is unavoidable.
|
|
1145
|
-
*/
|
|
1146
|
-
declare function handleTracesIngest(req: TracesIngestRequest, stores: IngestionStores): Promise<TracesIngestResponse>;
|
|
1147
|
-
/**
|
|
1148
|
-
* `POST /v1/feedback` — accept a single `FeedbackTrajectory` from the
|
|
1149
|
-
* production runtime. Idempotent on `id`: re-posting the same trajectory
|
|
1150
|
-
* replaces the prior record.
|
|
1151
|
-
*/
|
|
1152
|
-
declare function handleFeedbackIngest(req: FeedbackTrajectory, stores: IngestionStores): Promise<FeedbackIngestResponse>;
|
|
1153
|
-
|
|
1154
|
-
declare function buildOpenApi(packageVersion: string): OpenAPIObject;
|
|
1155
|
-
|
|
1156
|
-
interface RpcRequest {
|
|
1157
|
-
method: 'judge' | 'listRubrics' | 'version';
|
|
1158
|
-
params?: unknown;
|
|
1159
|
-
}
|
|
1160
|
-
interface RpcSuccess {
|
|
1161
|
-
result: unknown;
|
|
1162
|
-
}
|
|
1163
|
-
interface RpcError {
|
|
1164
|
-
error: {
|
|
1165
|
-
code: string;
|
|
1166
|
-
message: string;
|
|
1167
|
-
details?: unknown;
|
|
1168
|
-
};
|
|
1169
|
-
}
|
|
1170
|
-
declare function dispatchRpc(req: RpcRequest): Promise<RpcSuccess | RpcError>;
|
|
1171
|
-
/** Read one JSON request from stdin, write one JSON response to stdout. */
|
|
1172
|
-
declare function runRpcOnce(method?: string): Promise<number>;
|
|
1173
|
-
/** Read JSONL requests from stdin, write JSONL responses to stdout. */
|
|
1174
|
-
declare function runRpcBatch(method?: string): Promise<number>;
|
|
1175
|
-
|
|
1176
|
-
/**
|
|
1177
|
-
* Built-in rubrics shipped with agent-eval.
|
|
1178
|
-
*
|
|
1179
|
-
* A rubric is a set of scoring axes plus a system prompt that tells the
|
|
1180
|
-
* judging LLM how to grade against those axes. Built-in rubrics are
|
|
1181
|
-
* curated for use cases that recur across Tangle projects — call them
|
|
1182
|
-
* by name from any client.
|
|
1183
|
-
*
|
|
1184
|
-
* Adding a rubric:
|
|
1185
|
-
* 1. Define the Rubric object below with a clear `description` and
|
|
1186
|
-
* named `dimensions`.
|
|
1187
|
-
* 2. Register it in `BUILTIN_RUBRICS` at the bottom.
|
|
1188
|
-
* 3. Add a test in `tests/wire/rubrics.test.ts`.
|
|
1189
|
-
*
|
|
1190
|
-
* Custom rubrics: callers pass `rubric` inline to /v1/judge instead of
|
|
1191
|
-
* `rubricName` — see schemas.ts.
|
|
1192
|
-
*/
|
|
1193
|
-
|
|
1194
|
-
declare const BUILTIN_RUBRICS: Record<string, Rubric>;
|
|
1195
|
-
/** Get a built-in rubric by name, or undefined. */
|
|
1196
|
-
declare function getBuiltinRubric(name: string): Rubric | undefined;
|
|
1197
|
-
/** List built-in rubrics with their stable versions. */
|
|
1198
|
-
declare function listBuiltinRubrics(): {
|
|
1199
|
-
name: string;
|
|
1200
|
-
description: string;
|
|
1201
|
-
dimensions: {
|
|
1202
|
-
id: string;
|
|
1203
|
-
description: string;
|
|
1204
|
-
weight: number;
|
|
1205
|
-
}[];
|
|
1206
|
-
failureModes: string[];
|
|
1207
|
-
rubricVersion: string;
|
|
1208
|
-
}[];
|
|
1209
|
-
|
|
1210
|
-
interface CreateAppOptions {
|
|
1211
|
-
/** Stores wired to the ingestion endpoints. */
|
|
1212
|
-
stores?: IngestionStores;
|
|
1213
|
-
/**
|
|
1214
|
-
* Bearer-token auth. When provided, every endpoint EXCEPT `/healthz`
|
|
1215
|
-
* and `/v1/version` requires `Authorization: Bearer <token>`. The
|
|
1216
|
-
* token may be a static string OR a function for time-bounded /
|
|
1217
|
-
* rotating tokens.
|
|
1218
|
-
*
|
|
1219
|
-
* Recommended for any server that accepts ingestion writes from the
|
|
1220
|
-
* public internet. Read-only deployments may omit it.
|
|
1221
|
-
*/
|
|
1222
|
-
auth?: {
|
|
1223
|
-
bearer: string | ((token: string) => boolean | Promise<boolean>);
|
|
1224
|
-
};
|
|
1225
|
-
}
|
|
1226
|
-
declare function createApp(opts?: CreateAppOptions): Hono<hono_types.BlankEnv, hono_types.BlankSchema, "/">;
|
|
1227
|
-
interface ServeOptions extends CreateAppOptions {
|
|
1228
|
-
/** Default 5005. */
|
|
1229
|
-
port?: number;
|
|
1230
|
-
/** Default '127.0.0.1'. Set to '0.0.0.0' to listen on all interfaces. */
|
|
1231
|
-
host?: string;
|
|
1232
|
-
}
|
|
1233
|
-
declare function startServer(opts?: ServeOptions): ServerType;
|
|
1234
|
-
interface StartedServer {
|
|
1235
|
-
server: ServerType;
|
|
1236
|
-
/** The OS-assigned port. When opts.port was 0, this is the actual port the
|
|
1237
|
-
* kernel bound — callers that need to dial back (smoke tests, sidecars
|
|
1238
|
-
* registering with a parent) read this rather than guessing a free port. */
|
|
1239
|
-
port: number;
|
|
1240
|
-
/** Resolved host the server bound to (defaults to 127.0.0.1). */
|
|
1241
|
-
host: string;
|
|
1242
|
-
/** Close the server. Resolves once active connections have drained. */
|
|
1243
|
-
close(): Promise<void>;
|
|
1244
|
-
}
|
|
1245
|
-
/**
|
|
1246
|
-
* Promise-returning variant of `startServer` that resolves once the server is
|
|
1247
|
-
* listening and surfaces the resolved bound port. Use this from smoke tests
|
|
1248
|
-
* (`startServerAsync({ port: 0 })`) and any caller that needs to dial back.
|
|
1249
|
-
*/
|
|
1250
|
-
declare function startServerAsync(opts?: ServeOptions): Promise<StartedServer>;
|
|
1251
|
-
|
|
1252
|
-
export { BUILTIN_RUBRICS, type ErrorResponse, ErrorResponseSchema, type FailureMode, FailureModeSchema, FeedbackAttemptSchema, type FeedbackIngestResponse, FeedbackIngestResponseSchema, FeedbackLabelSchema, type FeedbackTrajectory, FeedbackTrajectorySchema, type HandleJudgeOptions, HealthResponseSchema, type IngestionStores, type JudgeRequest, JudgeRequestSchema, type JudgeResult, JudgeResultSchema, type ListRubricsResponse, ListRubricsResponseSchema, type Rubric, type RubricDimension, RubricDimensionSchema, type RubricInfo, RubricInfoSchema, RubricSchema, type ServeOptions, type StartedServer, type TraceEvent, TraceEventSchema, type TracesIngestRequest, TracesIngestRequestSchema, type TracesIngestResponse, TracesIngestResponseSchema, type VersionResponse, VersionResponseSchema, WIRE_VERSION, WireError, buildOpenApi, createApp, dispatchRpc, getBuiltinRubric, handleFeedbackIngest, handleJudge, handleListRubrics, handleTracesIngest, handleVersion, hashRubric, listBuiltinRubrics, runRpcBatch, runRpcOnce, startServer, startServerAsync };
|