@tangle-network/agent-runtime 0.229.0 → 0.231.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/dist/{activation-o9SrkJqs.js → activation-DVqSBfe1.js} +2 -2
  2. package/dist/{activation-o9SrkJqs.js.map → activation-DVqSBfe1.js.map} +1 -1
  3. package/dist/{activation-DFSQ90A8.d.ts → activation-Dltkq5S8.d.ts} +2 -2
  4. package/dist/agent.d.ts +2 -2
  5. package/dist/agent.js +2 -2
  6. package/dist/{coordination-driver-C6rde-Rv.js → coordination-driver-CUdEC2Qh.js} +4 -3
  7. package/dist/{coordination-driver-C6rde-Rv.js.map → coordination-driver-CUdEC2Qh.js.map} +1 -1
  8. package/dist/{delegate-Cmd-qlWH.js → delegate-BN0cSSUT.js} +2 -2
  9. package/dist/{delegate-Cmd-qlWH.js.map → delegate-BN0cSSUT.js.map} +1 -1
  10. package/dist/durable.d.ts +2 -2
  11. package/dist/durable.js +2 -2
  12. package/dist/{graph-BK05MhFv.js → graph-D4wgK3ns.js} +3 -3
  13. package/dist/{graph-BK05MhFv.js.map → graph-D4wgK3ns.js.map} +1 -1
  14. package/dist/{improvement-cycle-DHl4SxbQ.js → improvement-cycle-CjIeWuyB.js} +3 -3
  15. package/dist/{improvement-cycle-DHl4SxbQ.js.map → improvement-cycle-CjIeWuyB.js.map} +1 -1
  16. package/dist/{index-B6xSpuPo.d.ts → index-BW6lJXnH.d.ts} +11 -6
  17. package/dist/index.d.ts +6 -6
  18. package/dist/index.js +7 -7
  19. package/dist/intelligence.d.ts +3 -3
  20. package/dist/intelligence.js +3 -3
  21. package/dist/kernel.d.ts +3 -3
  22. package/dist/kernel.js +8 -8
  23. package/dist/{loop-runner-bin-D5hutINM.d.ts → loop-runner-bin-BcqlGNW-.d.ts} +3 -3
  24. package/dist/{loop-runner-bin-CU2bdlr1.js → loop-runner-bin-BodmUPOf.js} +3 -3
  25. package/dist/{loop-runner-bin-CU2bdlr1.js.map → loop-runner-bin-BodmUPOf.js.map} +1 -1
  26. package/dist/loop-runner-bin.d.ts +1 -1
  27. package/dist/loop-runner-bin.js +1 -1
  28. package/dist/mcp/bin.js +3 -3
  29. package/dist/mcp/index.d.ts +5 -5
  30. package/dist/mcp/index.js +6 -6
  31. package/dist/mcp/memory-bin.js +1 -1
  32. package/dist/{memory-server-BR310Weg.js → memory-server-Cn8LbAxO.js} +2 -2
  33. package/dist/{memory-server-BR310Weg.js.map → memory-server-Cn8LbAxO.js.map} +1 -1
  34. package/dist/{openai-tools-HhRT1L1W.d.ts → openai-tools-avHT8eq3.d.ts} +2 -2
  35. package/dist/profiles.d.ts +2 -2
  36. package/dist/{provision-supervisor-C2rRWXG6.js → provision-supervisor-DyNDlA2I.js} +3 -3
  37. package/dist/{provision-supervisor-C2rRWXG6.js.map → provision-supervisor-DyNDlA2I.js.map} +1 -1
  38. package/dist/{runtime-BBcrtrkj.js → runtime-BxZHapmG.js} +8 -8
  39. package/dist/{runtime-BBcrtrkj.js.map → runtime-BxZHapmG.js.map} +1 -1
  40. package/dist/{server-CWIl-a3C.js → server-C9pa2tSq.js} +4 -4
  41. package/dist/{server-CWIl-a3C.js.map → server-C9pa2tSq.js.map} +1 -1
  42. package/dist/stream-agent-turn-hiU4vgGX.d.ts +2058 -0
  43. package/dist/{structural-rollout-zl02ztZ4.js → structural-rollout-DQoLyth9.js} +2 -2
  44. package/dist/{structural-rollout-zl02ztZ4.js.map → structural-rollout-DQoLyth9.js.map} +1 -1
  45. package/dist/{substrate-sWW5cDIe.d.ts → substrate-CBo6TWOl.d.ts} +2 -2
  46. package/dist/{supervise-DbNMYx7z.js → supervise-Cl_-IsVc.js} +134 -21
  47. package/dist/supervise-Cl_-IsVc.js.map +1 -0
  48. package/dist/{supervisor-DAHG98Rs.js → supervisor-DFsXULnE.js} +232 -96
  49. package/dist/supervisor-DFsXULnE.js.map +1 -0
  50. package/dist/testing.d.ts +2 -2
  51. package/dist/testing.js +12 -12
  52. package/dist/{tool-server-BJbCPhoW.js → tool-server-x9NEhvHZ.js} +2 -2
  53. package/dist/tool-server-x9NEhvHZ.js.map +1 -0
  54. package/dist/tui/index.d.ts +1 -1
  55. package/dist/tui/index.js +1 -1
  56. package/dist/{stream-agent-turn-jPdnzI_5.d.ts → types-DBovefLD.d.ts} +4015 -4770
  57. package/package.json +4 -4
  58. package/dist/supervise-DbNMYx7z.js.map +0 -1
  59. package/dist/supervisor-DAHG98Rs.js.map +0 -1
  60. package/dist/tool-server-BJbCPhoW.js.map +0 -1
  61. package/dist/types-Ba5mJkyd.d.ts +0 -1293
@@ -1,1293 +0,0 @@
1
- import { l as RuntimeHooks } from "./runtime-hooks-Bj6wJHlH.js";
2
- import { AgentProfile, InputPart, RequestedInteractions, StreamEvent } from "@tangle-network/agent-interface";
3
- import { ControlBudget, ControlDecision, ControlEvalResult, ControlRunResult, ControlStep, DataAcquisitionPlan, DefaultVerdict, KnowledgeReadinessReport, KnowledgeRequirement, RunRecord, TraceStore, UserQuestion } from "@tangle-network/agent-eval";
4
- import { AgentRunOutcome } from "@tangle-network/sandbox/runtime";
5
- import { CreateRequestOptions, CreateSandboxOptions, PromptOptions, SandboxEvent, SandboxInstance } from "@tangle-network/sandbox";
6
- //#region src/types.d.ts
7
- /** @stable */
8
- interface AgentTaskSpec {
9
- id: string;
10
- intent: string;
11
- /** Domain is metadata, not an architectural boundary: tax, legal, gtm, creative, blueprint, redteam, etc. */
12
- domain?: string;
13
- inputs?: Record<string, unknown>;
14
- requiredKnowledge?: KnowledgeRequirement[];
15
- budget?: Partial<ControlBudget>;
16
- metadata?: Record<string, unknown>;
17
- }
18
- /** @stable */
19
- interface AgentKnowledgeProvider {
20
- buildReadiness?(task: AgentTaskSpec): Promise<KnowledgeReadinessReport> | KnowledgeReadinessReport;
21
- answerQuestions?(questions: UserQuestion[], task: AgentTaskSpec): Promise<Record<string, string>> | Record<string, string>;
22
- executeAcquisitionPlans?(plans: DataAcquisitionPlan[], task: AgentTaskSpec): Promise<string[]> | string[];
23
- refreshReadiness?(input: {
24
- task: AgentTaskSpec;
25
- previous: KnowledgeReadinessReport;
26
- userAnswers: Record<string, string>;
27
- acquiredEvidenceIds: string[];
28
- }): Promise<KnowledgeReadinessReport> | KnowledgeReadinessReport;
29
- }
30
- /** @stable */
31
- interface AgentTaskContext<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult> {
32
- task: AgentTaskSpec;
33
- knowledge: KnowledgeReadinessReport;
34
- state: TState;
35
- evals: TEval[];
36
- history: ControlStep<TState, TAction, TActionResult, TEval>[];
37
- budget: ControlBudget;
38
- stepIndex: number;
39
- wallMs: number;
40
- spentCostUsd: number;
41
- remainingCostUsd?: number;
42
- abortSignal: AbortSignal;
43
- }
44
- /** @stable */
45
- interface AgentAdapter<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult> {
46
- observe(ctx: {
47
- task: AgentTaskSpec;
48
- knowledge: KnowledgeReadinessReport;
49
- history: ControlStep<TState, TAction, TActionResult, TEval>[];
50
- abortSignal: AbortSignal;
51
- }): Promise<TState> | TState;
52
- validate(ctx: {
53
- task: AgentTaskSpec;
54
- knowledge: KnowledgeReadinessReport;
55
- state: TState;
56
- history: ControlStep<TState, TAction, TActionResult, TEval>[];
57
- abortSignal: AbortSignal;
58
- }): Promise<TEval[]> | TEval[];
59
- decide(ctx: AgentTaskContext<TState, TAction, TActionResult, TEval>): Promise<ControlDecision<TAction>> | ControlDecision<TAction>;
60
- act(action: TAction, ctx: AgentTaskContext<TState, TAction, TActionResult, TEval>): Promise<TActionResult> | TActionResult;
61
- shouldStop?(ctx: AgentTaskContext<TState, TAction, TActionResult, TEval>): Promise<{
62
- stop: boolean;
63
- pass: boolean;
64
- reason: string;
65
- score?: number;
66
- }> | {
67
- stop: boolean;
68
- pass: boolean;
69
- reason: string;
70
- score?: number;
71
- };
72
- onKnowledgeBlocked?(ctx: {
73
- task: AgentTaskSpec;
74
- knowledge: KnowledgeReadinessReport;
75
- questions: UserQuestion[];
76
- acquisitionPlans: DataAcquisitionPlan[];
77
- }): Promise<ControlDecision<TAction>> | ControlDecision<TAction>;
78
- getActionCostUsd?(ctx: {
79
- action: TAction;
80
- result: TActionResult;
81
- task: AgentTaskSpec;
82
- state: TState;
83
- evals: TEval[];
84
- history: ControlStep<TState, TAction, TActionResult, TEval>[];
85
- }): number | undefined;
86
- projectRunRecords?(result: ControlRunResult<TState, TAction, TActionResult, TEval>, task: AgentTaskSpec): RunRecord[];
87
- }
88
- /** @stable */
89
- type AgentTaskStatus = 'completed' | 'blocked' | 'failed' | 'aborted';
90
- /** @stable */
91
- type AgentRuntimeEvent<TState = unknown, TAction = unknown, TActionResult = unknown, TEval extends ControlEvalResult = ControlEvalResult> = {
92
- type: 'task_start';
93
- task: AgentTaskSpec;
94
- } | {
95
- type: 'readiness_start';
96
- task: AgentTaskSpec;
97
- } | {
98
- type: 'readiness_end';
99
- task: AgentTaskSpec;
100
- knowledge: KnowledgeReadinessReport;
101
- } | {
102
- type: 'questions_start';
103
- task: AgentTaskSpec;
104
- questions: UserQuestion[];
105
- } | {
106
- type: 'questions_end';
107
- task: AgentTaskSpec;
108
- questions: UserQuestion[];
109
- userAnswers: Record<string, string>;
110
- } | {
111
- type: 'acquisition_start';
112
- task: AgentTaskSpec;
113
- acquisitionPlans: DataAcquisitionPlan[];
114
- } | {
115
- type: 'acquisition_end';
116
- task: AgentTaskSpec;
117
- acquisitionPlans: DataAcquisitionPlan[];
118
- acquiredEvidenceIds: string[];
119
- } | {
120
- type: 'control_start';
121
- task: AgentTaskSpec;
122
- knowledge: KnowledgeReadinessReport;
123
- } | {
124
- type: 'control_step';
125
- task: AgentTaskSpec;
126
- step: ControlStep<TState, TAction, TActionResult, TEval>;
127
- } | {
128
- type: 'control_end';
129
- task: AgentTaskSpec;
130
- control: ControlRunResult<TState, TAction, TActionResult, TEval>;
131
- } | {
132
- type: 'task_end';
133
- task: AgentTaskSpec;
134
- status: AgentTaskStatus;
135
- reason: string;
136
- };
137
- /** @stable */
138
- type AgentRuntimeEventSink<TState = unknown, TAction = unknown, TActionResult = unknown, TEval extends ControlEvalResult = ControlEvalResult> = (event: AgentRuntimeEvent<TState, TAction, TActionResult, TEval>) => Promise<void> | void;
139
- /**
140
- *
141
- * Typed transport / backend failure detail. Carried on `backend_error` and
142
- * `final` events when the backend's stream throws or the upstream HTTP call
143
- * returns a non-success status. Lets consumers (a) distinguish "stream
144
- * completed with no text" from "stream never reached the model" and
145
- * (b) reconstruct the precise upstream signal (status + truncated body) when
146
- * building a `RunRecord.error`.
147
- *
148
- * `body` is truncated to 2 KiB by the backend so an HTML error page from a
149
- * misconfigured proxy never bloats event payloads or logs. Consumers needing
150
- * the full body should inspect the underlying `BackendTransportError.body`
151
- * via a custom `mapEvent` or backend wrapper.
152
- *
153
- * @stable
154
- */
155
- interface BackendErrorDetail {
156
- /**
157
- * `'transport'` — upstream HTTP / network failure with optional status code.
158
- * `'backend'` — the backend's `stream()` generator threw for a non-transport
159
- * reason (e.g. a custom adapter error, sandbox crash).
160
- */
161
- kind: 'transport' | 'backend';
162
- message: string;
163
- /** Upstream HTTP status when known. `0` for connection / abort errors. */
164
- status?: number;
165
- /** Truncated response body (≤2 KiB). Diagnostic only — never machine-parsed. */
166
- body?: string;
167
- }
168
- /**
169
- *
170
- * OpenAI Chat Completions tool descriptor. The shape mirrors the
171
- * `/v1/chat/completions` `tools[]` parameter so caller-owned compatible
172
- * transports can pass tool definitions without translation. A router can
173
- * proxy this shape to Anthropic
174
- * (translated server-side), DeepSeek, Groq, OpenAI, and Gemini — every model
175
- * that the eval surface targets.
176
- *
177
- * Callers that build their tool list from MCP servers should run a one-shot
178
- * MCP `tools/list` at config time and project the result into this shape. The
179
- * runtime intentionally does NOT depend on `@modelcontextprotocol/sdk` —
180
- * keeping the backend transport thin lets domain repos own MCP plumbing.
181
- *
182
- * @stable
183
- */
184
- interface OpenAIChatTool {
185
- type: 'function';
186
- function: {
187
- name: string;
188
- description?: string;
189
- parameters?: Record<string, unknown>;
190
- };
191
- }
192
- /**
193
- *
194
- * `tool_choice` parameter for OpenAI-compat chat. Same shape as the OpenAI
195
- * spec: `'auto'` (default — model decides), `'none'` (disable tool calling
196
- * for this turn), `'required'` (force a tool call), or a specific function
197
- * pin `{ type: 'function', function: { name } }`.
198
- *
199
- * @stable
200
- */
201
- type OpenAIChatToolChoice = 'auto' | 'none' | 'required' | {
202
- type: 'function';
203
- function: {
204
- name: string;
205
- };
206
- };
207
- /**
208
- *
209
- * `response_format` parameter for OpenAI-compatible chat endpoints. Use
210
- * `json_object` when the caller needs syntactically valid JSON, or
211
- * `json_schema` when the upstream provider supports schema-constrained JSON.
212
- *
213
- * @stable
214
- */
215
- type OpenAIChatResponseFormat = {
216
- type: 'text';
217
- } | {
218
- type: 'json_object';
219
- } | {
220
- type: 'json_schema';
221
- json_schema: Record<string, unknown>;
222
- };
223
- /** Agent Interface events that do not belong to Runtime's task vocabulary. */
224
- type RuntimeCanonicalStreamEvent = StreamEvent & {
225
- task?: AgentTaskSpec;
226
- session?: RuntimeSession;
227
- timestamp?: string;
228
- };
229
- /** @stable */
230
- type RuntimeStreamEvent = RuntimeCanonicalStreamEvent | {
231
- type: 'task_start';
232
- task: AgentTaskSpec;
233
- timestamp: string;
234
- } | {
235
- type: 'readiness_start';
236
- task: AgentTaskSpec;
237
- timestamp: string;
238
- } | {
239
- type: 'readiness_end';
240
- task: AgentTaskSpec;
241
- knowledge: KnowledgeReadinessReport;
242
- decision: KnowledgeReadinessDecision;
243
- timestamp: string;
244
- } | {
245
- type: 'questions_start';
246
- task: AgentTaskSpec;
247
- questions: UserQuestion[];
248
- timestamp: string;
249
- } | {
250
- type: 'questions_end';
251
- task: AgentTaskSpec;
252
- questions: UserQuestion[];
253
- userAnswers: Record<string, string>;
254
- timestamp: string;
255
- } | {
256
- type: 'acquisition_start';
257
- task: AgentTaskSpec;
258
- acquisitionPlans: DataAcquisitionPlan[];
259
- timestamp: string;
260
- } | {
261
- type: 'acquisition_end';
262
- task: AgentTaskSpec;
263
- acquisitionPlans: DataAcquisitionPlan[];
264
- acquiredEvidenceIds: string[];
265
- timestamp: string;
266
- } | {
267
- type: 'session_created';
268
- task: AgentTaskSpec;
269
- session: RuntimeSession;
270
- timestamp: string;
271
- } | {
272
- type: 'session_resumed';
273
- task: AgentTaskSpec;
274
- session: RuntimeSession;
275
- timestamp: string;
276
- } | {
277
- type: 'backend_start';
278
- task: AgentTaskSpec;
279
- session: RuntimeSession;
280
- backend: string;
281
- /** Canonical execution identity and materialization evidence for this turn, when Runtime
282
- * owns the selected executor. Generic metadata keeps the event vocabulary open while the
283
- * values use Runtime's existing identity/materialization receipt shapes. */
284
- metadata?: Record<string, unknown>;
285
- timestamp: string;
286
- } | {
287
- type: 'text_delta';
288
- task?: AgentTaskSpec;
289
- session?: RuntimeSession;
290
- text: string;
291
- timestamp?: string;
292
- } | {
293
- type: 'reasoning_delta';
294
- task?: AgentTaskSpec;
295
- session?: RuntimeSession;
296
- text: string;
297
- timestamp?: string;
298
- } | {
299
- type: 'tool_call';
300
- task?: AgentTaskSpec;
301
- session?: RuntimeSession;
302
- toolName: string;
303
- toolCallId?: string;
304
- args?: unknown;
305
- timestamp?: string;
306
- } | {
307
- type: 'tool_result';
308
- task?: AgentTaskSpec;
309
- session?: RuntimeSession;
310
- toolName: string;
311
- toolCallId?: string;
312
- result?: unknown;
313
- timestamp?: string;
314
- } | {
315
- type: 'llm_call';
316
- task?: AgentTaskSpec;
317
- session?: RuntimeSession;
318
- model: string;
319
- tokensIn?: number;
320
- tokensOut?: number;
321
- /** False when the numeric token subtotal is incomplete or absent. */
322
- tokensKnown?: false;
323
- /** Why `tokensKnown` is false when a harness receipt was present but unreadable. */
324
- tokensUnknownReason?: string;
325
- costUsd?: number;
326
- /** False when `costUsd` is only an observed floor, estimate, or absent. */
327
- usdKnown?: false;
328
- /** Separately-labelled local/catalog estimate; never billed spend. */
329
- estimatedCostUsd?: number;
330
- /** Provider-reported prompt-cache fields; absent fields remain unknown. */
331
- promptCache?: Readonly<Record<string, number | string>>;
332
- latencyMs?: number;
333
- finishReason?: string;
334
- timestamp?: string;
335
- } | {
336
- type: 'artifact';
337
- task?: AgentTaskSpec;
338
- session?: RuntimeSession;
339
- artifactId: string;
340
- name?: string;
341
- mimeType?: string;
342
- uri?: string;
343
- content?: string;
344
- metadata?: Record<string, unknown>;
345
- timestamp?: string;
346
- } | {
347
- type: 'proposal_created';
348
- task?: AgentTaskSpec;
349
- session?: RuntimeSession;
350
- proposalId: string;
351
- title: string;
352
- status?: 'pending' | 'approved' | 'rejected';
353
- content?: string;
354
- timestamp?: string;
355
- } | {
356
- type: 'backend_error';
357
- task: AgentTaskSpec;
358
- session?: RuntimeSession;
359
- backend: string;
360
- message: string;
361
- recoverable: boolean;
362
- /**
363
- * Typed transport diagnostic. Present when the upstream returned a
364
- * non-success HTTP status or every retry attempt threw. Consumers MUST
365
- * surface this onto their `RunRecord.error` — silently treating a
366
- * `backend_error` as "no output" hides credit exhaustion, auth failure,
367
- * and upstream outages from operators.
368
- * - `kind: 'transport'` — HTTP / network failure with optional `status`
369
- * + truncated response `body`.
370
- * - `kind: 'backend'` — the backend's `stream()` generator threw for a
371
- * reason that isn't a recognized transport failure.
372
- */
373
- error?: BackendErrorDetail;
374
- timestamp: string;
375
- } | {
376
- type: 'backend_end';
377
- task: AgentTaskSpec;
378
- session: RuntimeSession;
379
- backend: string;
380
- timestamp: string;
381
- } | {
382
- type: 'task_end';
383
- task: AgentTaskSpec;
384
- status: AgentTaskStatus;
385
- reason: string;
386
- timestamp: string;
387
- } | {
388
- type: 'final';
389
- task: AgentTaskSpec;
390
- session?: RuntimeSession;
391
- status: AgentTaskStatus;
392
- reason: string;
393
- text?: string;
394
- metadata?: Record<string, unknown>;
395
- /**
396
- * Typed terminal-error diagnostic. Mirrors the `backend_error.error`
397
- * shape so a consumer that only listens for `final` still receives a
398
- * loud, structured failure when the backend never produced output. Only
399
- * set when `status !== 'completed'`. Consumers building a `RunRecord`
400
- * MUST map this to `RunRecord.error` rather than recording silent
401
- * `error: null` with empty `finalText`.
402
- */
403
- error?: BackendErrorDetail;
404
- timestamp: string;
405
- };
406
- /** @stable */
407
- interface RuntimeSession {
408
- id: string;
409
- backend: string;
410
- status: 'active' | 'completed' | 'failed' | 'aborted';
411
- resumeToken?: string;
412
- createdAt: string;
413
- updatedAt: string;
414
- metadata?: Record<string, unknown>;
415
- }
416
- /** @stable */
417
- interface RuntimeSessionStore {
418
- get(sessionId: string): Promise<RuntimeSession | undefined> | RuntimeSession | undefined;
419
- put(session: RuntimeSession): Promise<void> | void;
420
- appendEvent?(sessionId: string, event: RuntimeStreamEvent): Promise<void> | void;
421
- listEvents?(sessionId: string): Promise<RuntimeStreamEvent[]> | RuntimeStreamEvent[];
422
- }
423
- /** @stable */
424
- interface AgentBackendInput {
425
- task: AgentTaskSpec;
426
- message?: string;
427
- messages?: Array<{
428
- role: string;
429
- content: string;
430
- }>;
431
- parts?: InputPart[];
432
- interactions?: RequestedInteractions;
433
- providerOptions?: Record<string, unknown>;
434
- inputs?: Record<string, unknown>;
435
- }
436
- /** @stable */
437
- interface AgentBackendContext {
438
- task: AgentTaskSpec;
439
- knowledge: KnowledgeReadinessReport;
440
- session: RuntimeSession;
441
- signal?: AbortSignal;
442
- /**
443
- * Conversation/run identifier when this call is part of a multi-agent run.
444
- * Backends should stamp it into any trace/log emission so cross-participant
445
- * events correlate. Absent when the call is a stand-alone `runAgentTask`.
446
- */
447
- runId?: string;
448
- /**
449
- * Deterministic turn id for this single call. Stable across retries of the
450
- * same logical turn so a caching gateway / idempotent backend can dedupe.
451
- */
452
- turnId?: string;
453
- /**
454
- * If this call is itself nested inside a higher-order conversation
455
- * (recursion via `createConversationBackend`), the enclosing turn's id.
456
- * Used for trace stitching across nested orchestration.
457
- */
458
- parentTurnId?: string;
459
- /**
460
- * Headers to forward verbatim to any outbound HTTP the backend issues:
461
- * `X-Tangle-Forwarded-Authorization`, `X-Tangle-Forwarded-Depth`,
462
- * run/turn correlation. Backends that issue HTTP MUST merge these into
463
- * the outbound request; backends that don't issue HTTP may ignore them.
464
- */
465
- propagatedHeaders?: Readonly<Record<string, string>>;
466
- }
467
- /** @stable */
468
- interface AgentExecutionBackend<TInput extends AgentBackendInput = AgentBackendInput> {
469
- kind: string;
470
- start?(input: TInput, context: Omit<AgentBackendContext, 'session'> & {
471
- requestedSessionId?: string;
472
- }): Promise<RuntimeSession> | RuntimeSession;
473
- resume?(session: RuntimeSession, input: TInput, context: Omit<AgentBackendContext, 'session'>): Promise<RuntimeSession> | RuntimeSession;
474
- stream(input: TInput, context: AgentBackendContext): AsyncIterable<RuntimeStreamEvent>;
475
- stop?(session: RuntimeSession, reason: string): Promise<void> | void;
476
- }
477
- /** @stable */
478
- interface RunAgentTaskStreamOptions<TInput extends AgentBackendInput = AgentBackendInput> {
479
- task: AgentTaskSpec;
480
- backend: AgentExecutionBackend<TInput>;
481
- input?: Omit<TInput, 'task'>;
482
- knowledge?: AgentKnowledgeProvider;
483
- sessionStore?: RuntimeSessionStore;
484
- sessionId?: string;
485
- resume?: boolean;
486
- signal?: AbortSignal;
487
- minimumReadinessScore?: number;
488
- }
489
- /** @stable */
490
- interface RunAgentTaskOptions<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult> {
491
- task: AgentTaskSpec;
492
- adapter: AgentAdapter<TState, TAction, TActionResult, TEval>;
493
- knowledge?: AgentKnowledgeProvider;
494
- onEvent?: AgentRuntimeEventSink<TState, TAction, TActionResult, TEval>;
495
- store?: TraceStore;
496
- signal?: AbortSignal;
497
- scenarioId?: string;
498
- projectId?: string;
499
- variantId?: string;
500
- minimumReadinessScore?: number;
501
- }
502
- /** @stable */
503
- interface AgentTaskRunResult<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult> {
504
- task: AgentTaskSpec;
505
- status: AgentTaskStatus;
506
- knowledge: KnowledgeReadinessReport;
507
- questions: UserQuestion[];
508
- acquisitionPlans: DataAcquisitionPlan[];
509
- userAnswers: Record<string, string>;
510
- acquiredEvidenceIds: string[];
511
- control: ControlRunResult<TState, TAction, TActionResult, TEval>;
512
- runRecords: RunRecord[];
513
- }
514
- /** @stable */
515
- interface KnowledgeReadinessDecision {
516
- passed: boolean;
517
- status: 'ready' | 'blocked' | 'caveat';
518
- reason: string;
519
- readinessScore: number;
520
- recommendedAction: KnowledgeReadinessReport['recommendedAction'];
521
- severity: KnowledgeReadinessReport['severity'];
522
- blockingGapIds: string[];
523
- nonBlockingGapIds: string[];
524
- }
525
- //#endregion
526
- //#region src/runtime-run.d.ts
527
- /** @stable */
528
- type RuntimeRunStatus = 'running' | 'completed' | 'failed' | 'cancelled';
529
- /** @stable */
530
- interface RuntimeRunCost {
531
- /** Cumulative input tokens across every observed `llm_call` event. */
532
- tokensIn: number;
533
- /** Cumulative output tokens across every observed `llm_call` event. */
534
- tokensOut: number;
535
- /** Sum of `costUsd` from every observed `llm_call` event. */
536
- costUsd: number;
537
- /** Wall time from `startRuntimeRun()` to `complete()` (or `now()` if not yet completed). */
538
- wallMs: number;
539
- /** Count of `llm_call` events observed during the run. */
540
- llmCalls: number;
541
- }
542
- /** @stable */
543
- interface RuntimeRunCompleteInput {
544
- status: Exclude<RuntimeRunStatus, 'running'>;
545
- resultSummary?: string;
546
- /** Optional explicit cost override; if omitted, the accumulated ledger is used. */
547
- cost?: Partial<RuntimeRunCost>;
548
- /** Stable error message when `status === 'failed'`. */
549
- error?: string;
550
- /** Additional adapter-specific fields merged into the persisted row. */
551
- metadata?: Record<string, unknown>;
552
- }
553
- /** @stable */
554
- interface RuntimeRunRow {
555
- /** Stable runtime-side identifier. Adapters may translate to their own primary key. */
556
- id: string;
557
- workspaceId: string;
558
- sessionId?: string;
559
- agentId?: string;
560
- domain?: string;
561
- taskId: string;
562
- scenarioId?: string;
563
- status: RuntimeRunStatus;
564
- resultSummary?: string;
565
- error?: string;
566
- cost: RuntimeRunCost;
567
- startedAt: string;
568
- completedAt?: string;
569
- metadata?: Record<string, unknown>;
570
- }
571
- /** @stable */
572
- interface RuntimeRunPersistenceAdapter {
573
- /**
574
- * Called once when `handle.persist()` runs. Implementations write `row` to
575
- * their durable store (D1, postgres, KV) and return whatever the consumer
576
- * wants the caller to see (often the storage-side row id). Errors thrown
577
- * here propagate out of `persist()` so the caller can decide whether to
578
- * retry or log-and-continue.
579
- */
580
- upsert(row: RuntimeRunRow): Promise<void> | void;
581
- }
582
- /** @stable */
583
- interface RuntimeRunOptions {
584
- workspaceId: string;
585
- sessionId?: string;
586
- agentId?: string;
587
- taskSpec: AgentTaskSpec;
588
- scenarioId?: string;
589
- /** Optional persistence adapter; if omitted, `persist()` is a no-op. */
590
- adapter?: RuntimeRunPersistenceAdapter;
591
- /** Override the row id; default = `${taskSpec.id}:${random suffix}`. */
592
- id?: string;
593
- /** Override the clock; default = `Date.now()`. Useful for deterministic tests. */
594
- now?: () => number;
595
- }
596
- /** @stable */
597
- interface RuntimeRunHandle {
598
- /** Stable id assigned at start. */
599
- readonly id: string;
600
- readonly workspaceId: string;
601
- readonly sessionId: string | undefined;
602
- readonly taskSpec: AgentTaskSpec;
603
- readonly status: RuntimeRunStatus;
604
- /**
605
- * Observe a single `RuntimeStreamEvent`. The handle ignores non-cost events
606
- * (text deltas, tool calls) silently so consumers can pipe the whole stream
607
- * through `handle.observe`. `llm_call` events update the ledger.
608
- */
609
- observe(event: RuntimeStreamEvent): void;
610
- /** Snapshot of the current cost ledger. Safe to call at any time. */
611
- cost(): RuntimeRunCost;
612
- /**
613
- * Transition to a terminal state. Idempotent for the same status; throws
614
- * `RuntimeRunStateError` for a different terminal status (state machines
615
- * don't time-travel).
616
- */
617
- complete(input: RuntimeRunCompleteInput): void;
618
- /** Build the current row without writing it. Useful for tests + dry runs. */
619
- toRow(metadata?: Record<string, unknown>): RuntimeRunRow;
620
- /**
621
- * Persist the current row via the configured adapter. Must be called after
622
- * `complete()`. Idempotent for the same terminal state (the adapter sees
623
- * the same row on retry).
624
- */
625
- persist(metadata?: Record<string, unknown>): Promise<void>;
626
- }
627
- /**
628
- *
629
- * Construct a runtime-run handle. The returned handle is mutable across its
630
- * lifetime; consumers should not share it across requests.
631
- *
632
- * @stable
633
- */
634
- declare function startRuntimeRun(options: RuntimeRunOptions): RuntimeRunHandle;
635
- //#endregion
636
- //#region src/runtime/types.d.ts
637
- /** @stable */
638
- interface ValidationCtx {
639
- /** Iteration index this output came from (0-based). */
640
- iteration: number;
641
- /**
642
- * Live sandbox for this iteration. Validators that need execution-grounded
643
- * evidence can inspect files or run commands here instead of forcing callers
644
- * to bypass the loop kernel with raw Sandbox SDK orchestration.
645
- */
646
- box?: SandboxInstance;
647
- /** Cooperative cancellation channel. */
648
- signal: AbortSignal;
649
- /**
650
- * Optional trace emitter. When set, validator implementations that make
651
- * LLM calls (e.g. an LLM-judge reviewer) emit spans into it.
652
- * The kernel passes `ctx.traceEmitter` from `ExecCtx` when available.
653
- */
654
- traceEmitter?: LoopTraceEmitter;
655
- }
656
- /** @stable */
657
- interface Validator<Output, Verdict = DefaultVerdict> {
658
- validate(output: Output, ctx: ValidationCtx): Promise<Verdict>;
659
- }
660
- /**
661
- * Sandbox-SDK-shaped agent specification.
662
- *
663
- * The kernel uses `profile` to instantiate a sandbox per iteration, formats
664
- * `task` into a prompt via `taskToPrompt`, and merges `sandboxOverrides` into
665
- * the `CreateSandboxOptions` it passes to `client.create`. Heterogeneous
666
- * fanout supplies multiple `AgentRunSpec`s and the kernel round-robins
667
- * through them when the driver plans N tasks.
668
- *
669
- * @stable
670
- */
671
- interface AgentRunSpec<Task> {
672
- /** Sandbox SDK profile — what kind of agent runs the task. */
673
- profile: AgentProfile;
674
- /** Task → prompt formatter. Pure and deterministic. */
675
- taskToPrompt: (task: Task) => string;
676
- /**
677
- * Optional pre-prompt sandbox provisioner. Runs after the sandbox is acquired
678
- * and before the first prompt is streamed into that box. Use this for
679
- * domain-agnostic setup such as repo snapshots, benchmark fixtures, policy
680
- * files, or seed datasets. The hook is part of the runtime surface so loop
681
- * consumers do not hand-roll Sandbox SDK orchestration just to prepare a
682
- * workspace before the agent sees it.
683
- *
684
- * `ctx.recordMount` records what was placed into the box so the run carries a
685
- * provenance manifest (`LoopResult.provenance.mounts`). It is optional and
686
- * provenance-only — the kernel never reads box contents and attaches no
687
- * meaning to the entries; not calling it simply leaves the manifest empty.
688
- */
689
- prepareBox?: (box: SandboxInstance, ctx: {
690
- signal: AbortSignal;
691
- recordMount: MountRecorder;
692
- }) => Promise<void> | void;
693
- /**
694
- * Per-spec stable name. Surfaced in trace events and the default winner
695
- * selector tiebreak. Falls back to `profile.name ?? 'agent'`.
696
- */
697
- name?: string;
698
- /**
699
- * Optional sandbox-SDK `CreateSandboxOptions` overrides merged on top of
700
- * the kernel's defaults. `backend.profile` is set to `profile` by the
701
- * kernel and cannot be overridden here — use `profile` itself for that.
702
- */
703
- sandboxOverrides?: Partial<Omit<CreateSandboxOptions, 'backend'>> & {
704
- backend?: Omit<NonNullable<CreateSandboxOptions['backend']>, 'profile'>;
705
- };
706
- }
707
- /**
708
- * Stream of `SandboxEvent`s → typed `Output`.
709
- *
710
- * Adapters are pure functions over the already-collected event array; they
711
- * do not receive the live AsyncIterable so they can be replayed against
712
- * persisted streams during tests / replays.
713
- *
714
- * @stable
715
- */
716
- interface OutputAdapter<Output> {
717
- parse(events: SandboxEvent[]): Output;
718
- }
719
- /** LLM token usage. Structurally maps into agent-eval's paid-call receipt so a
720
- * campaign dispatch settles real usage instead of appearing as a stub. */
721
- interface LoopTokenUsage {
722
- /** Total provider-reported prompt tokens. Budgets always use this total. */
723
- input: number;
724
- output: number;
725
- /** False when the subtotal is incomplete. */
726
- tokensKnown?: false;
727
- /** Prompt tokens newly processed by the provider, when every prompt class is known. */
728
- freshInput?: number;
729
- /** Prompt tokens the provider reported serving from its cache. */
730
- cacheRead?: number;
731
- /** Prompt tokens the provider reported writing to its cache. */
732
- cacheWrite?: number;
733
- /**
734
- * False when any positive-input observation omitted or contradicted the prompt-cache split.
735
- * This marker is sticky during aggregation. Missing cache fields must never become zero.
736
- */
737
- cacheBreakdownKnown?: false;
738
- }
739
- /**
740
- * One mounted resource recorded during box preparation — a pure provenance
741
- * record of what the caller placed into a box before the agent saw it. The
742
- * kernel never reads box contents itself (it does not know what was mounted);
743
- * the caller, which owns the bytes inside `prepareBox`, supplies each entry via
744
- * `recordMount`. Carries no domain semantics — just where the resource landed,
745
- * its content fingerprint, its size, and where it came from — so a run is
746
- * auditable after the fact ("what exactly was this agent given?").
747
- *
748
- * @stable
749
- */
750
- interface MountManifestEntry {
751
- /** Destination path inside the box where the resource was placed. */
752
- path: string;
753
- /** Hex SHA-256 of the mounted bytes. The caller computes it from the bytes
754
- * it wrote — the kernel does not hash box contents. */
755
- sha256: string;
756
- /** Size of the mounted resource in bytes. */
757
- bytes: number;
758
- /** Free-form origin of the resource (e.g. a repo ref, a corpus id, a local
759
- * path, a URL). Provenance only — the kernel attaches no meaning to it. */
760
- source: string;
761
- }
762
- /**
763
- * A record of one candidate-selection decision: which iteration the selector
764
- * picked (or rejected) and why. Pure audit trail of the SELECTOR role — it
765
- * carries the selector's identity, the candidate's score, and an optional
766
- * human-readable reason, with no domain semantics. The kernel emits one receipt
767
- * per scored candidate at finalize so a run answers "why did THIS one win?".
768
- *
769
- * @stable
770
- */
771
- interface SelectionReceipt {
772
- /** Iteration index this receipt is about. */
773
- candidateIndex: number;
774
- /** True for the iteration the selector chose as winner; false otherwise. */
775
- selected: boolean;
776
- /** The candidate's verdict score, when it has one. */
777
- score?: number;
778
- /** Why this candidate was (or was not) selected, when the selector states it. */
779
- reason?: string;
780
- /** Identity of the selector that produced this receipt — `'caller'` (an
781
- * explicit `selectWinner`), `'driver'` (a driver-authored winner), or
782
- * `'default'` (the kernel's best-valid-score argmax). */
783
- selector: 'caller' | 'driver' | 'default';
784
- }
785
- /**
786
- * Domain-free run provenance: a manifest of what was mounted into the run's
787
- * boxes and the receipts for how the winner was selected. Surfaced on
788
- * `LoopResult` purely for run auditability — nothing in the kernel branches on
789
- * it. Empty arrays when the caller recorded no mounts and there was no
790
- * candidate to select.
791
- *
792
- * @stable
793
- */
794
- interface RunProvenance {
795
- /** Every resource recorded via `prepareBox`'s `recordMount`, in record order. */
796
- mounts: MountManifestEntry[];
797
- /** One receipt per scored candidate at finalize, in iteration order. */
798
- selectionReceipts: SelectionReceipt[];
799
- }
800
- /**
801
- * Records a mounted resource into the run's provenance manifest. Passed to
802
- * `prepareBox` so the caller — which owns the bytes it writes into the box —
803
- * declares what it mounted without the kernel having to inspect box contents.
804
- *
805
- * @stable
806
- */
807
- type MountRecorder = (entry: MountManifestEntry) => void;
808
- /** @stable */
809
- interface Iteration<Task, Output> {
810
- /** 0-based iteration index assigned by the kernel. */
811
- index: number;
812
- task: Task;
813
- /** Stable name of the `AgentRunSpec` that produced this iteration. */
814
- agentRunName: string;
815
- output?: Output;
816
- verdict?: DefaultVerdict;
817
- error?: Error;
818
- /** Public Sandbox outcome settled after the complete event stream. */
819
- sandboxOutcome?: AgentRunOutcome;
820
- /** Raw sandbox event stream collected for this iteration. Present on a failed iteration too,
821
- * holding the events received before the failure — including the one that reported it. */
822
- events: SandboxEvent[];
823
- startedAt: number;
824
- endedAt: number;
825
- costUsd: number;
826
- /** False when `costUsd` is only the observed subtotal, not a complete bill. */
827
- costUsdKnown?: false;
828
- /**
829
- * The part of `costUsd` that came from calls carrying no billing receipt, summed per call.
830
- *
831
- * `costUsdKnown` is an AND over the iteration, so it cannot say HOW MUCH of the total is
832
- * unproven: one receiptless call marks the whole iteration unknown. This is the amount that
833
- * belongs on `Spend.usdEstimated`, which keeps `usd - usdEstimated` reading as billed money on
834
- * a settlement that mixed both kinds. Absent when every dollar here carried a receipt.
835
- */
836
- unprovenCostUsd?: number;
837
- /** Local/catalog estimates remain separate from billed spend. */
838
- estimatedCostUsd?: number;
839
- /** Provider-reported prompt-cache fields; absent fields remain unknown. */
840
- promptCache?: Record<string, number | string>;
841
- /** Summed LLM token usage across every `llm_call` event in this iteration. */
842
- tokenUsage: LoopTokenUsage;
843
- /**
844
- * Wall time this iteration's box was alive, in milliseconds: from the moment the loop acquired
845
- * the box to the moment the loop's own teardown returned.
846
- *
847
- * ABSENT when this iteration did not own the box's terminal. A lineage box and a same-sandbox
848
- * box are reaped at loop end, AFTER the loop has already built its result, so no iteration can
849
- * pair them. A missing lifetime is never a zero: a box nobody timed is a different fact from a
850
- * box that lived no time.
851
- */
852
- boxLiveMs?: number;
853
- /** False when `boxLiveMs` is a floor rather than the full lifetime: the delete was attempted and
854
- * never acknowledged, so the box may have outlived the number. */
855
- boxLiveMsKnown?: false;
856
- }
857
- /** @stable */
858
- interface Driver<Task, Output, Decision> {
859
- /**
860
- * Trace label surfaced in trace events. No behavioral effect: it never
861
- * selects a strategy or a decision path. Default `'driver'`.
862
- */
863
- readonly name?: string;
864
- /**
865
- * Tasks to issue this iteration. `[task]` → refine; N copies → fanout;
866
- * `[]` → no more work this round (kernel proceeds to `decide`).
867
- */
868
- plan(task: Task, history: ReadonlyArray<Iteration<Task, Output>>): Promise<Task[]>;
869
- /**
870
- * Inspect history and return the next state. The kernel terminates the
871
- * loop when `decide` returns a `TerminalDecision`
872
- * (`'stop' | 'pick-winner' | 'fail' | 'done'`, exported as
873
- * `TERMINAL_DECISIONS` with the `isTerminalDecision` guard), when
874
- * `maxIterations` is hit, or when the abort signal fires. Every other
875
- * value is caller vocabulary and continues the loop.
876
- */
877
- decide(history: ReadonlyArray<Iteration<Task, Output>>): Decision | Promise<Decision>;
878
- /**
879
- * Optional: describe the move `plan()` just produced, for trace emission.
880
- * The kernel calls this immediately after `plan()` and emits the result in
881
- * the `loop.plan` event so a topology viewer can render the agent's chosen
882
- * move + rationale (not just the inferred fan-width). Drivers whose topology
883
- * is a pure function of count (refine/fanout-vote) omit it — the kernel
884
- * infers `moveKind` from the planned-task count. A driver that authors its
885
- * own topology returns its chosen move's kind + rationale here.
886
- */
887
- describePlan?(): LoopPlanDescription | undefined;
888
- /**
889
- * Optional: the driver AUTHORS the winner instead of the kernel's argmax. The
890
- * kernel consults this at finalize ONLY when the caller did not pass an explicit
891
- * `selectWinner` to runAgentRounds. Return the driver-declared winner (e.g. from a
892
- * `select` topology move) or `undefined` to fall through to the default
893
- * (best-valid-score, earliest index). This is the SELECTOR role made
894
- * agent-authorable — the planner runs the selection, not the kernel.
895
- * @experimental
896
- */
897
- selectWinner?(history: ReadonlyArray<Iteration<Task, Output>>): LoopWinner<Task, Output> | undefined;
898
- }
899
- /** @stable Driver-supplied description of the just-planned move. */
900
- interface LoopPlanDescription {
901
- /** Topology move this round — e.g. `'refine' | 'fanout' | 'verify' | 'stop'`. */
902
- kind: string;
903
- /** Why the driver chose this move (the agent's rationale), when available. */
904
- rationale?: string;
905
- /**
906
- * Iteration index this round branches FROM, when the driver declares it.
907
- * Overrides the kernel's inferred branch point — lets a planner that
908
- * branches off a specific (non-winner) iteration emit faithful edge lineage.
909
- * Omit to keep the inferred (best-valid / latest) branch point.
910
- */
911
- parentIndex?: number;
912
- }
913
- /** @stable */
914
- interface LoopWinner<Task, Output> {
915
- task: Task;
916
- output: Output;
917
- verdict?: DefaultVerdict;
918
- iterationIndex: number;
919
- agentRunName: string;
920
- }
921
- /** @stable */
922
- interface LoopResult<Task, Output, Decision> {
923
- decision: Decision;
924
- iterations: Iteration<Task, Output>[];
925
- winner?: LoopWinner<Task, Output>;
926
- durationMs: number;
927
- /** Sum of every iteration's `costUsd`. */
928
- costUsd: number;
929
- /** False when `costUsd` is only the observed subtotal, not a complete bill. */
930
- costUsdKnown?: false;
931
- /** Sum of every iteration's `unprovenCostUsd` — the part of `costUsd` no billing receipt
932
- * covers. Absent when every dollar in the loop carried one. */
933
- unprovenCostUsd?: number;
934
- /** Sum of separately-labelled local/catalog estimates. */
935
- estimatedCostUsd?: number;
936
- /** Aggregated provider-reported prompt-cache fields. */
937
- promptCache?: Record<string, number | string>;
938
- /** Sum of every iteration's token usage. `loopDispatch` commits it through
939
- * the campaign's paid-call receipt. */
940
- tokenUsage: LoopTokenUsage;
941
- /** Sum of `Iteration.boxLiveMs` over the iterations that could pair a box acquire with its
942
- * teardown. ABSENT when none could — the run's box time went unmeasured, which is not a zero. */
943
- boxLiveMs?: number;
944
- /** False when at least one iteration ran a box whose lifetime the loop could not fully observe,
945
- * so `boxLiveMs` is a floor over the run rather than its total. */
946
- boxLiveMsKnown?: false;
947
- /** Domain-free run provenance for auditability: the mount manifest recorded
948
- * during `prepareBox` and the selection receipts for how the winner was
949
- * chosen. Always present; empty arrays when nothing was recorded. */
950
- provenance: RunProvenance;
951
- }
952
- /**
953
- * Minimal sandbox client surface the kernel calls. Satisfied structurally by
954
- * `new Sandbox({ apiKey, baseUrl })` — declared as a structural type so
955
- * tests can pass a stub without instantiating the SDK.
956
- *
957
- * `describePlacement` is optional. When present, the kernel calls it after
958
- * each `create()` so the `loop.iteration.dispatch` trace event carries fleet
959
- * coordinates (fleetId + machineId) instead of just the sibling sandboxId.
960
- * Fleet-aware adapters set this; the raw `Sandbox` SDK class does not, and
961
- * the kernel falls back to `{ placement: 'sibling', sandboxId: box.id }`.
962
- *
963
- * @stable
964
- */
965
- interface SandboxClient {
966
- create(options?: CreateSandboxOptions, requestOptions?: CreateRequestOptions): Promise<SandboxInstance>;
967
- describePlacement?(box: SandboxInstance): LoopSandboxPlacement;
968
- /**
969
- * Optional legacy CRIU capability probe. When present and it resolves
970
- * `{ available: true }`, the loop's `lineage.fork` seam may checkpoint and fork
971
- * a parent box when live `branch(count)` is unavailable. Current Sandbox boxes
972
- * expose live branching directly. The kernel reads this ONLY through the
973
- * capability probe — it never branches on backend kind.
974
- * The raw `Sandbox` SDK class satisfies it; the loop's test fakes omit it
975
- * (⇒ `canFork = false`).
976
- * @experimental
977
- */
978
- criuStatus?(): Promise<{
979
- available: boolean;
980
- criuVersion?: string;
981
- reason?: string;
982
- }>;
983
- }
984
- /**
985
- * Opt-in box-lineage controls for `runAgentRounds`. Default OFF — with both flags
986
- * unset the kernel's per-iteration behavior is byte-identical to acquiring a
987
- * fresh box, streaming once, and tearing it down. The independence of N fresh
988
- * boxes (e.g. `random@k`) is a compute-control invariant; these flags must
989
- * never apply to it. Enable them ONLY on a steered loop (refine / planner-driven
990
- * fanout) where reusing the parent's context is intended.
991
- *
992
- * Live-box footprint: the lineage keeps every box it starts or forks alive
993
- * across rounds so a later round can descend from it, and tears them down at
994
- * loop end. When the driver's branch point is kernel-inferred (no
995
- * `describePlan` — refine, fanout-vote), the kernel prunes boxes no future
996
- * round can reach after each round, so the live set tracks the active frontier.
997
- * When the driver authors its own branch point (`describePlan().parentIndex`),
998
- * it may descend from any prior
999
- * iteration, so no box is pruned and the live-box count rises to the total
1000
- * iterations across all rounds. Size `forkFanout` runs accordingly. Live branch
1001
- * children use copy-on-write, but each is still a live box until loop end.
1002
- *
1003
- * @experimental
1004
- */
1005
- interface LoopLineageOptions {
1006
- /**
1007
- * When true, a refine round (1 planned task) descending from a prior round
1008
- * CONTINUES the parent iteration's session on the SAME box
1009
- * (`streamPrompt({ sessionId })`) instead of acquiring a fresh box and
1010
- * re-injecting prior context as prompt text. Round 0 (no parent) always
1011
- * starts fresh. Usable on any single-task path, not just the refine driver.
1012
- *
1013
- * Requires a platform that honors a client-supplied `sessionId`. The lineage
1014
- * mints the id and `continue` asserts the session is still live
1015
- * (`box.session(id).status()`), failing loud if the platform dropped it — so a
1016
- * non-honoring platform errors instead of silently running contextless turns.
1017
- * Verify continuity against the live platform before enabling: the assertion
1018
- * proves the session EXISTS server-side, not that prior turns replay into it.
1019
- */
1020
- sessionContinuity?: boolean;
1021
- /**
1022
- * When true, a fanout round (N planned tasks) descending from a prior round
1023
- * branches the parent's live box so all N branches inherit its context prefix.
1024
- * If live branching is unavailable, the lineage uses legacy CRIU when its
1025
- * probe is positive. Otherwise it degrades to N fresh boxes with no prefix.
1026
- * Round 0 always starts fresh. NEVER set this for a `random@k` control arm —
1027
- * forking would couple the independent samples.
1028
- *
1029
- * A real fork inherits the parent's IMAGE/PROFILE: per-branch `AgentRunSpec`
1030
- * profiles are honored only on the degraded fresh-box path, so a
1031
- * heterogeneous-profile fanout silently homogenizes to the parent's profile
1032
- * when fork is available. Use this for same-profile branching; for
1033
- * different-per-branch profiles use the unforked fanout path.
1034
- */
1035
- forkFanout?: boolean;
1036
- /**
1037
- * Per-turn sandbox streaming mode. Default `'sse'` (live `streamPrompt` —
1038
- * low-latency, full per-token trace; best for interactive chat). `'poll'`
1039
- * fire-and-detaches via `dispatchPrompt` and awaits the terminal result by
1040
- * status-polling, so a long, quiet in-box turn (clone + build + test) never
1041
- * holds a live stream a proxy idle-timeout can drop mid-execution. Lower trace
1042
- * fidelity (one terminal event), so it is opt-in — intended for BATCH eval
1043
- * runs, which don't need live streaming and were losing long turns to the
1044
- * idle-drop. Applies to the default fresh-box path too, not only when
1045
- * `sessionContinuity`/`forkFanout` are on.
1046
- */
1047
- streaming?: 'sse' | 'poll';
1048
- }
1049
- /** @stable */
1050
- interface LoopSandboxPlacement {
1051
- /** `in-process` is a local harness CLI in the caller's own process tree — no sandbox, no fleet.
1052
- * It is a placement in its own right so a cost or latency breakdown split by placement does not
1053
- * count local runs in the sandbox bucket. */
1054
- kind: 'sibling' | 'fleet' | 'in-process';
1055
- sandboxId?: string;
1056
- fleetId?: string;
1057
- machineId?: string;
1058
- }
1059
- /** @stable */
1060
- interface LoopTraceEmitter {
1061
- emit(event: LoopTraceEvent): void | Promise<void>;
1062
- }
1063
- /** @stable */
1064
- type LoopTraceEvent = {
1065
- kind: 'loop.started';
1066
- runId: string;
1067
- timestamp: number;
1068
- payload: LoopStartedPayload;
1069
- } | {
1070
- kind: 'loop.plan';
1071
- runId: string;
1072
- timestamp: number;
1073
- payload: LoopPlanPayload;
1074
- } | {
1075
- kind: 'loop.iteration.started';
1076
- runId: string;
1077
- timestamp: number;
1078
- payload: LoopIterationStartedPayload;
1079
- } | {
1080
- kind: 'loop.iteration.dispatch';
1081
- runId: string;
1082
- timestamp: number;
1083
- payload: LoopIterationDispatchPayload;
1084
- } | {
1085
- kind: 'loop.iteration.ended';
1086
- runId: string;
1087
- timestamp: number;
1088
- payload: LoopIterationEndedPayload;
1089
- } | {
1090
- kind: 'loop.decision';
1091
- runId: string;
1092
- timestamp: number;
1093
- payload: LoopDecisionPayload;
1094
- } | {
1095
- kind: 'loop.ended';
1096
- runId: string;
1097
- timestamp: number;
1098
- payload: LoopEndedPayload;
1099
- } | {
1100
- kind: 'loop.teardown.failed';
1101
- runId: string;
1102
- timestamp: number;
1103
- payload: LoopTeardownFailedPayload;
1104
- };
1105
- /** @stable */
1106
- interface LoopStartedPayload {
1107
- driver: string;
1108
- agentRunNames: string[];
1109
- maxIterations: number;
1110
- maxConcurrency: number;
1111
- }
1112
- /**
1113
- * Emitted once per `plan()` round, immediately after the driver plans. Carries
1114
- * the topology move so a viewer renders WHAT the agent decided + WHY, not just
1115
- * the inferred fan-width. `moveKind` is the driver's `describePlan().kind` when
1116
- * provided, else inferred from `plannedCount` (0→stop, 1→refine, N→fanout).
1117
- *
1118
- * @stable
1119
- */
1120
- interface LoopPlanPayload {
1121
- /** 0-based plan round (one per `plan()` call). */
1122
- roundIndex: number;
1123
- /** Tasks the driver issued this round. */
1124
- plannedCount: number;
1125
- /** Topology move — `'refine' | 'fanout' | 'verify' | 'stop'` etc. */
1126
- moveKind: string;
1127
- /** Driver rationale for the move, when available. */
1128
- rationale?: string;
1129
- /**
1130
- * Iteration index this round branched FROM (the edge source). `undefined`
1131
- * for round 0 (root). Kernel-inferred branch point — the best-valid (else
1132
- * latest) iteration so far — unless a driver later declares it explicitly.
1133
- */
1134
- parentIndex?: number;
1135
- /** Iteration indices this round dispatched (the edge targets). */
1136
- childIndices: number[];
1137
- }
1138
- /** @stable */
1139
- interface LoopIterationStartedPayload {
1140
- iterationIndex: number;
1141
- agentRunName: string;
1142
- taskHash: string;
1143
- /** Plan round (== `LoopPlanPayload.roundIndex`) this iteration belongs to. */
1144
- groupId?: number;
1145
- /** Iteration this one was planned from; `undefined` ⇒ root. */
1146
- parentIndex?: number;
1147
- }
1148
- /**
1149
- * Where the iteration's worker was placed. `sibling` = a fresh sandbox the
1150
- * kernel created via `sandboxClient.create`. `fleet` = an existing machine in
1151
- * a shared-workspace fleet — workers see the caller's filesystem and any diff
1152
- * they write lands on it directly.
1153
- *
1154
- * @stable
1155
- */
1156
- interface LoopIterationDispatchPayload {
1157
- iterationIndex: number;
1158
- agentRunName: string;
1159
- placement: 'sibling' | 'fleet' | 'in-process';
1160
- /** Set on every placement. Lets analyst loops correlate per-iteration logs. */
1161
- sandboxId?: string;
1162
- /** Set only when `placement === 'fleet'`. */
1163
- fleetId?: string;
1164
- /** Set only when `placement === 'fleet'`. */
1165
- machineId?: string;
1166
- /** Plan round this iteration belongs to. */
1167
- groupId?: number;
1168
- /** Iteration this one was planned from; `undefined` ⇒ root. */
1169
- parentIndex?: number;
1170
- }
1171
- /** @stable */
1172
- interface LoopIterationEndedPayload {
1173
- iterationIndex: number;
1174
- agentRunName: string;
1175
- outputHash?: string;
1176
- verdict?: DefaultVerdict;
1177
- error?: string;
1178
- costUsd: number;
1179
- costUsdKnown?: false;
1180
- estimatedCostUsd?: number;
1181
- durationMs: number;
1182
- /** Summed LLM token usage for this iteration — maps to gen_ai.usage.* on the
1183
- * branch span. Omitted when no `llm_call` events carried token counts. */
1184
- tokenUsage?: LoopTokenUsage;
1185
- /** Plan round this iteration belongs to. */
1186
- groupId?: number;
1187
- /** Iteration this one was planned from; `undefined` ⇒ root. */
1188
- parentIndex?: number;
1189
- /** Truncated string preview of the parsed output — for a viewer's drawer.
1190
- * Bounded to ~280 chars; never the full payload. */
1191
- outputPreview?: string;
1192
- }
1193
- /** @stable */
1194
- interface LoopDecisionPayload {
1195
- decision: string;
1196
- historyLength: number;
1197
- }
1198
- /** @stable */
1199
- interface LoopEndedPayload {
1200
- winnerIterationIndex?: number;
1201
- totalCostUsd: number;
1202
- costUsdKnown?: false;
1203
- estimatedCostUsd?: number;
1204
- durationMs: number;
1205
- iterations: number;
1206
- }
1207
- /** Emitted when a box's `delete()` throws or times out during teardown — the
1208
- * loop swallows the failure (platform reaps on expiry) but surfaces it here so
1209
- * a real leak (e.g. mid-loop auth expiry) is observable. @stable */
1210
- interface LoopTeardownFailedPayload {
1211
- sandboxId?: string;
1212
- /** `'timeout'` or the delete error message. */
1213
- reason: string;
1214
- }
1215
- /**
1216
- * Execution context for `runAgentRounds`: the sandbox client the kernel creates boxes through, plus optional runtime hooks.
1217
- *
1218
- * @stable
1219
- */
1220
- interface ExecCtx {
1221
- /** Sandbox SDK client — the kernel calls `.create()` per iteration. */
1222
- sandboxClient: SandboxClient;
1223
- /**
1224
- * Per-prompt sandbox SDK options, forwarded verbatim into EVERY `streamPrompt`
1225
- * of every iteration and every turn of this run. The kernel owns `sessionId`
1226
- * and `signal`: both are removed from the supplied value and applied last, so
1227
- * only the kernel's own session id and abort signal reach the SDK. A value
1228
- * that is present but not an object is a `ValidationError`, raised before any
1229
- * box is created.
1230
- *
1231
- * Typical use: `backend.model` credentials (`authMode` / `authFiles`) so a
1232
- * session runs on a caller-supplied subscription credential, `timeoutMs` for a
1233
- * per-turn wall-clock ceiling, and `context` for platform-side metadata.
1234
- *
1235
- * The instrument keys — `backend` and `model` — need a real box. The no-box
1236
- * `SandboxClient` seams (`inlineSandboxClient`, `localSandboxClient`,
1237
- * `inProcessSandboxClient`, and the in-process MCP executor) run an in-process
1238
- * executor with nothing to reconfigure, so they refuse a `backend` or `model`
1239
- * with a `ValidationError` instead of running the turn on an instrument the
1240
- * caller did not ask for. They accept and ignore every other key. The one
1241
- * exception is `inProcessSandboxClient`, which surfaces the verbatim options
1242
- * to its `onPrompt` callback: that callback is the executor, so a per-turn
1243
- * `model` reaches it and only `backend` is refused.
1244
- */
1245
- promptOptions?: Omit<PromptOptions, 'signal' | 'sessionId'>;
1246
- /** Optional runtime hooks. Execution-scoped; never part of `AgentProfile`. */
1247
- hooks?: RuntimeHooks;
1248
- /** Optional trace emitter. When set, the kernel emits `loop.*` events. */
1249
- traceEmitter?: LoopTraceEmitter;
1250
- /**
1251
- * Optional per-event tee. When set, the kernel forwards EVERY raw event from
1252
- * each iteration's `streamPrompt` stream as it arrives, so a host can stream
1253
- * the agent's live output (tokens, tool calls) token-by-token. The observer
1254
- * receives a defensive copy of each event — mutating it cannot affect the
1255
- * run's own cost accounting or output parsing. Called synchronously in the hot
1256
- * stream loop and never awaited, so a slow or never-settling observer cannot
1257
- * stall the stream; keep it cheap. An async observer is fire-and-forget: its
1258
- * promise is not awaited, so events carry no ordering or backpressure
1259
- * guarantees (the next event may be observed before a prior async observer
1260
- * settles) — use it for side-effect telemetry, not sequential processing.
1261
- * Both a synchronous throw and a rejected returned promise are caught +
1262
- * ignored so the observer can never break the run — but prefer not to depend
1263
- * on that.
1264
- *
1265
- * @experimental
1266
- */
1267
- onSandboxEvent?: (event: SandboxEvent, meta: {
1268
- iterationIndex: number;
1269
- agentRunName: string;
1270
- }) => void | PromiseLike<void>;
1271
- /**
1272
- * Optional production-run handle. When set, every synthesized `llm_call`
1273
- * the kernel infers from a sandbox event stream is forwarded via
1274
- * `runHandle.observe` so per-run cost aggregates pick up loop spend.
1275
- */
1276
- runHandle?: RuntimeRunHandle;
1277
- /** Cooperative cancellation signal. */
1278
- signal?: AbortSignal;
1279
- /**
1280
- * Trace id for OTEL correlation. When set alongside `traceEmitter`, the
1281
- * exporter uses this as the parent trace for all emitted spans. Typically
1282
- * inherited from TRACE_ID env var in MCP subprocess mode.
1283
- */
1284
- traceId?: string;
1285
- /**
1286
- * Parent span id for OTEL correlation. Loop events become children of
1287
- * this span. Typically inherited from PARENT_SPAN_ID env var.
1288
- */
1289
- parentSpanId?: string;
1290
- }
1291
- //#endregion
1292
- export { OpenAIChatToolChoice as $, RuntimeRunCompleteInput as A, AgentBackendInput as B, MountRecorder as C, SelectionReceipt as D, SandboxClient as E, RuntimeRunRow as F, AgentTaskContext as G, AgentKnowledgeProvider as H, RuntimeRunStatus as I, AgentTaskStatus as J, AgentTaskRunResult as K, startRuntimeRun as L, RuntimeRunHandle as M, RuntimeRunOptions as N, ValidationCtx as O, RuntimeRunPersistenceAdapter as P, OpenAIChatTool as Q, AgentAdapter as R, MountManifestEntry as S, RunProvenance as T, AgentRuntimeEvent as U, AgentExecutionBackend as V, AgentRuntimeEventSink as W, KnowledgeReadinessDecision as X, BackendErrorDetail as Y, OpenAIChatResponseFormat as Z, LoopTeardownFailedPayload as _, Iteration as a, RuntimeStreamEvent as at, LoopTraceEvent as b, LoopIterationDispatchPayload as c, LoopLineageOptions as d, RunAgentTaskOptions as et, LoopPlanDescription as f, LoopStartedPayload as g, LoopSandboxPlacement as h, ExecCtx as i, RuntimeSessionStore as it, RuntimeRunCost as j, Validator as k, LoopIterationEndedPayload as l, LoopResult as m, DefaultVerdict as n, RuntimeCanonicalStreamEvent as nt, LoopDecisionPayload as o, LoopPlanPayload as p, AgentTaskSpec as q, Driver as r, RuntimeSession as rt, LoopEndedPayload as s, AgentRunSpec as t, RunAgentTaskStreamOptions as tt, LoopIterationStartedPayload as u, LoopTokenUsage as v, OutputAdapter as w, LoopWinner as x, LoopTraceEmitter as y, AgentBackendContext as z };
1293
- //# sourceMappingURL=types-Ba5mJkyd.d.ts.map