@tangle-network/agent-runtime 0.229.0 → 0.231.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{activation-o9SrkJqs.js → activation-DVqSBfe1.js} +2 -2
- package/dist/{activation-o9SrkJqs.js.map → activation-DVqSBfe1.js.map} +1 -1
- package/dist/{activation-DFSQ90A8.d.ts → activation-Dltkq5S8.d.ts} +2 -2
- package/dist/agent.d.ts +2 -2
- package/dist/agent.js +2 -2
- package/dist/{coordination-driver-C6rde-Rv.js → coordination-driver-CUdEC2Qh.js} +4 -3
- package/dist/{coordination-driver-C6rde-Rv.js.map → coordination-driver-CUdEC2Qh.js.map} +1 -1
- package/dist/{delegate-Cmd-qlWH.js → delegate-BN0cSSUT.js} +2 -2
- package/dist/{delegate-Cmd-qlWH.js.map → delegate-BN0cSSUT.js.map} +1 -1
- package/dist/durable.d.ts +2 -2
- package/dist/durable.js +2 -2
- package/dist/{graph-BK05MhFv.js → graph-D4wgK3ns.js} +3 -3
- package/dist/{graph-BK05MhFv.js.map → graph-D4wgK3ns.js.map} +1 -1
- package/dist/{improvement-cycle-DHl4SxbQ.js → improvement-cycle-CjIeWuyB.js} +3 -3
- package/dist/{improvement-cycle-DHl4SxbQ.js.map → improvement-cycle-CjIeWuyB.js.map} +1 -1
- package/dist/{index-B6xSpuPo.d.ts → index-BW6lJXnH.d.ts} +11 -6
- package/dist/index.d.ts +6 -6
- package/dist/index.js +7 -7
- package/dist/intelligence.d.ts +3 -3
- package/dist/intelligence.js +3 -3
- package/dist/kernel.d.ts +3 -3
- package/dist/kernel.js +8 -8
- package/dist/{loop-runner-bin-D5hutINM.d.ts → loop-runner-bin-BcqlGNW-.d.ts} +3 -3
- package/dist/{loop-runner-bin-CU2bdlr1.js → loop-runner-bin-BodmUPOf.js} +3 -3
- package/dist/{loop-runner-bin-CU2bdlr1.js.map → loop-runner-bin-BodmUPOf.js.map} +1 -1
- package/dist/loop-runner-bin.d.ts +1 -1
- package/dist/loop-runner-bin.js +1 -1
- package/dist/mcp/bin.js +3 -3
- package/dist/mcp/index.d.ts +5 -5
- package/dist/mcp/index.js +6 -6
- package/dist/mcp/memory-bin.js +1 -1
- package/dist/{memory-server-BR310Weg.js → memory-server-Cn8LbAxO.js} +2 -2
- package/dist/{memory-server-BR310Weg.js.map → memory-server-Cn8LbAxO.js.map} +1 -1
- package/dist/{openai-tools-HhRT1L1W.d.ts → openai-tools-avHT8eq3.d.ts} +2 -2
- package/dist/profiles.d.ts +2 -2
- package/dist/{provision-supervisor-C2rRWXG6.js → provision-supervisor-DyNDlA2I.js} +3 -3
- package/dist/{provision-supervisor-C2rRWXG6.js.map → provision-supervisor-DyNDlA2I.js.map} +1 -1
- package/dist/{runtime-BBcrtrkj.js → runtime-BxZHapmG.js} +8 -8
- package/dist/{runtime-BBcrtrkj.js.map → runtime-BxZHapmG.js.map} +1 -1
- package/dist/{server-CWIl-a3C.js → server-C9pa2tSq.js} +4 -4
- package/dist/{server-CWIl-a3C.js.map → server-C9pa2tSq.js.map} +1 -1
- package/dist/stream-agent-turn-hiU4vgGX.d.ts +2058 -0
- package/dist/{structural-rollout-zl02ztZ4.js → structural-rollout-DQoLyth9.js} +2 -2
- package/dist/{structural-rollout-zl02ztZ4.js.map → structural-rollout-DQoLyth9.js.map} +1 -1
- package/dist/{substrate-sWW5cDIe.d.ts → substrate-CBo6TWOl.d.ts} +2 -2
- package/dist/{supervise-DbNMYx7z.js → supervise-Cl_-IsVc.js} +134 -21
- package/dist/supervise-Cl_-IsVc.js.map +1 -0
- package/dist/{supervisor-DAHG98Rs.js → supervisor-DFsXULnE.js} +232 -96
- package/dist/supervisor-DFsXULnE.js.map +1 -0
- package/dist/testing.d.ts +2 -2
- package/dist/testing.js +12 -12
- package/dist/{tool-server-BJbCPhoW.js → tool-server-x9NEhvHZ.js} +2 -2
- package/dist/tool-server-x9NEhvHZ.js.map +1 -0
- package/dist/tui/index.d.ts +1 -1
- package/dist/tui/index.js +1 -1
- package/dist/{stream-agent-turn-jPdnzI_5.d.ts → types-DBovefLD.d.ts} +4015 -4770
- package/package.json +4 -4
- package/dist/supervise-DbNMYx7z.js.map +0 -1
- package/dist/supervisor-DAHG98Rs.js.map +0 -1
- package/dist/tool-server-BJbCPhoW.js.map +0 -1
- package/dist/types-Ba5mJkyd.d.ts +0 -1293
|
@@ -0,0 +1,2058 @@
|
|
|
1
|
+
import { $ as Spend, An as ExecutorProgress, C as ExecutorToolCall, Ci as BackendErrorDetail, Cn as TraceSource, G as Runtime, K as Scope, Kr as LoopTokenUsage, Mr as ExecCtx, Ni as RuntimeStreamEvent, Si as AgentTaskStatus, Un as NativeSessionEvidence, b as ExecutorRegistry, ei as SandboxClient, f as ExecutorCancellation, g as ExecutorFactory, p as ExecutorCancellationRequest, ri as Validator, s as DefaultVerdict, ut as UsageEvent, y as ExecutorProgressEvent } from "./types-DBovefLD.js";
|
|
2
|
+
import { AgentProfile, AgentProfileMcpServer, AgentProfileValidationResult, HarnessType, ReasoningEffort, StreamEvent } from "@tangle-network/agent-interface";
|
|
3
|
+
import { AgentRunOutcome } from "@tangle-network/sandbox/runtime";
|
|
4
|
+
import { ChildProcess } from "node:child_process";
|
|
5
|
+
import { WorkspacePlanReceipt } from "@tangle-network/agent-profile-materialize";
|
|
6
|
+
import { BackendType, CreateSandboxOptions, PromptOptions, Sandbox, SandboxEvent, SandboxInstance } from "@tangle-network/sandbox";
|
|
7
|
+
import { AgentEnvironment as AgentEnvironment$1, AgentEnvironmentCapabilities, AgentEnvironmentCapabilities as AgentEnvironmentCapabilities$1, AgentEnvironmentEvent, AgentEnvironmentEvent as AgentEnvironmentEvent$1, AgentEnvironmentProvider, AgentEnvironmentProvider as AgentEnvironmentProvider$1, AgentEnvironmentQuery, AgentEnvironmentStatus, AgentEnvironmentSummary, AgentProfileRef, AgentProfileRef as AgentProfileRef$1, AgentSession, AgentSessionRef, AgentSessionStatus as AgentSessionStatus$1, AgentTurnInput, AgentTurnInput as AgentTurnInput$1, AgentTurnResult as AgentTurnResult$2, CheckpointRef, CheckpointRequest, CreateAgentEnvironmentInput, CreateAgentEnvironmentInput as CreateAgentEnvironmentInput$1, ExecRequest, ExecResult, ForkRequest, PlacementInfo, ResourceRequest, WorkspaceRequest } from "@tangle-network/agent-interface/environment-provider";
|
|
8
|
+
//#region src/redact.d.ts
|
|
9
|
+
/**
|
|
10
|
+
*
|
|
11
|
+
* Redaction for values that may leave the Runtime process. The default scrubs
|
|
12
|
+
* common leak classes (API keys, bearer tokens, emails, private keys) from
|
|
13
|
+
* strings and walks nested objects and arrays. A customer with domain-specific
|
|
14
|
+
* PII supplies their own `redact` hook.
|
|
15
|
+
*
|
|
16
|
+
* This is intentionally narrower than `src/sanitize.ts` (which redacts the
|
|
17
|
+
* runtime's *event envelope* field-by-field): here the value is opaque
|
|
18
|
+
* customer payload, so the scrub is value-shaped, not schema-shaped.
|
|
19
|
+
*
|
|
20
|
+
* @experimental
|
|
21
|
+
*/
|
|
22
|
+
/** A redactor maps an arbitrary trace value to a safe-to-export value. Pure;
|
|
23
|
+
* must not throw on cyclic input (the default tolerates cycles). */
|
|
24
|
+
type Redactor = (value: unknown) => unknown;
|
|
25
|
+
/**
|
|
26
|
+
* The built-in redactor. Walks objects and arrays; replaces values under
|
|
27
|
+
* secret-bearing keys wholesale; scrubs in-value patterns from every string.
|
|
28
|
+
* Cycle-safe (a seen-set short-circuits self-referential payloads to
|
|
29
|
+
* `'[circular]'`), depth-bounded, and total — never throws on customer input.
|
|
30
|
+
*/
|
|
31
|
+
declare function defaultRedactor(value: unknown): unknown;
|
|
32
|
+
/**
|
|
33
|
+
* Resolve the redactor a client uses. A caller-supplied hook handles
|
|
34
|
+
* domain-specific values first, then the built-in scrubber still removes
|
|
35
|
+
* common credentials and email addresses. Returning `false` is the explicit
|
|
36
|
+
* opt-out for already-reviewed public values.
|
|
37
|
+
*/
|
|
38
|
+
declare function resolveRedactor(redact: Redactor | false | undefined): Redactor;
|
|
39
|
+
//#endregion
|
|
40
|
+
//#region src/mcp/protocol.d.ts
|
|
41
|
+
/**
|
|
42
|
+
* Shared wire contracts for the in-process stdio MCP servers.
|
|
43
|
+
*
|
|
44
|
+
* Keeping these types in one module prevents the delegation and generic tool
|
|
45
|
+
* servers from accepting subtly different JSON-RPC messages.
|
|
46
|
+
*
|
|
47
|
+
* @experimental
|
|
48
|
+
*/
|
|
49
|
+
/** A callable MCP tool exposed by either stdio server. @experimental */
|
|
50
|
+
interface McpToolDescriptor {
|
|
51
|
+
name: string;
|
|
52
|
+
description: string;
|
|
53
|
+
inputSchema: Record<string, unknown>;
|
|
54
|
+
handler: (raw: unknown) => Promise<unknown>;
|
|
55
|
+
}
|
|
56
|
+
/** Stdio-shaped transport used by the shared JSON-RPC server implementation. @experimental */
|
|
57
|
+
interface McpTransport {
|
|
58
|
+
input: NodeJS.ReadableStream;
|
|
59
|
+
output: NodeJS.WritableStream;
|
|
60
|
+
}
|
|
61
|
+
/** One JSON-RPC 2.0 request or notification. @experimental */
|
|
62
|
+
interface JsonRpcMessage {
|
|
63
|
+
jsonrpc: '2.0';
|
|
64
|
+
id?: number | string | null;
|
|
65
|
+
method: string;
|
|
66
|
+
params?: unknown;
|
|
67
|
+
}
|
|
68
|
+
/** One JSON-RPC 2.0 response. @experimental */
|
|
69
|
+
interface JsonRpcResponse {
|
|
70
|
+
jsonrpc: '2.0';
|
|
71
|
+
id: number | string | null;
|
|
72
|
+
result?: unknown;
|
|
73
|
+
error?: {
|
|
74
|
+
code: number;
|
|
75
|
+
message: string;
|
|
76
|
+
data?: unknown;
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
//#endregion
|
|
80
|
+
//#region src/runtime/supervise/peer-mail.d.ts
|
|
81
|
+
/**
|
|
82
|
+
* What one envelope IS, typed so a reader can act on it without parsing prose.
|
|
83
|
+
*
|
|
84
|
+
* - `ask` — request a fact the sender lacks; expects an `answer`.
|
|
85
|
+
* - `tell` — share a result; MUST carry evidence refs.
|
|
86
|
+
* - `challenge` — dispute a peer's claim; MUST cite the refs of the claim it disputes.
|
|
87
|
+
* - `answer` — reply to an `ask` or a `challenge`.
|
|
88
|
+
*/
|
|
89
|
+
type PeerMailKind = 'ask' | 'tell' | 'challenge' | 'answer';
|
|
90
|
+
/** One admitted peer message. `threadId` is the root mail's id; `depth` is 0 for a root mail and
|
|
91
|
+
* one more than its parent for a reply, which is what the reply-depth cap counts. */
|
|
92
|
+
interface PeerMailEnvelope {
|
|
93
|
+
readonly mailId: string;
|
|
94
|
+
readonly threadId: string;
|
|
95
|
+
readonly depth: number;
|
|
96
|
+
/** The bound sender — resolved from the capability, never from a tool argument. */
|
|
97
|
+
readonly from: string;
|
|
98
|
+
readonly to: string;
|
|
99
|
+
readonly kind: PeerMailKind;
|
|
100
|
+
readonly subject: string;
|
|
101
|
+
readonly body: string;
|
|
102
|
+
/** Evidence the receiver can re-check for itself. Required for `tell` and `challenge`. */
|
|
103
|
+
readonly evidenceRefs: ReadonlyArray<string>;
|
|
104
|
+
/** The mail id this replies to. Never a coordination question id — a peer cannot address the
|
|
105
|
+
* parent's answer channel. */
|
|
106
|
+
readonly replyTo?: string;
|
|
107
|
+
readonly at: number;
|
|
108
|
+
}
|
|
109
|
+
/** Why an attempt did not reach a sibling. Each value is a fact the sender can read and act on. */
|
|
110
|
+
type PeerMailRefusal = 'sender-unbound' | 'self-addressed' | 'send-quota-exhausted' | 'mailbox-full' | 'thread-depth-exceeded' | 'thread-stopped' | 'unknown-reply-target' | 'evidence-required' | 'subject-too-large' | 'body-too-large' | 'forged-authority' | 'unknown-worker' | 'already-settled' | 'worker-has-no-inbox' | 'scope-stopped' | 'runtime-error';
|
|
111
|
+
type PeerMailOutcome = 'delivered' | PeerMailRefusal;
|
|
112
|
+
/** The audit record for one attempt — published whether it delivered or was refused, because a
|
|
113
|
+
* refused attempt is exactly what a parent auditing a channel needs to see. */
|
|
114
|
+
interface PeerMailEvent {
|
|
115
|
+
readonly envelope: PeerMailEnvelope;
|
|
116
|
+
readonly delivered: boolean;
|
|
117
|
+
readonly outcome: PeerMailOutcome;
|
|
118
|
+
/** Canonical digest of the exact admitted body, so a later claim can name the bytes it read. */
|
|
119
|
+
readonly bodyDigest: string;
|
|
120
|
+
readonly error?: string;
|
|
121
|
+
}
|
|
122
|
+
/** Hard bounds. Every one fails closed with a refusal the sender can read. */
|
|
123
|
+
interface PeerMailLimits {
|
|
124
|
+
/** Mail one worker may attempt to send for the whole run. */
|
|
125
|
+
readonly maxSentPerWorker: number;
|
|
126
|
+
/** Mail one worker may receive for the whole run. */
|
|
127
|
+
readonly maxInboxPerWorker: number;
|
|
128
|
+
/** Total admitted body bytes one worker may receive for the whole run. */
|
|
129
|
+
readonly maxInboxBytesPerWorker: number;
|
|
130
|
+
/** Maximum reply depth; a root mail is depth 0, so `2` allows ask → answer → answer. */
|
|
131
|
+
readonly maxThreadDepth: number;
|
|
132
|
+
readonly maxBodyBytes: number;
|
|
133
|
+
readonly maxSubjectBytes: number;
|
|
134
|
+
}
|
|
135
|
+
/** Bounds chosen so a peer channel cannot become the dominant cost of a run: eight sends and
|
|
136
|
+
* sixteen receives per worker, 32 KiB of received body, and a reply chain that terminates. */
|
|
137
|
+
declare const DEFAULT_PEER_MAIL_LIMITS: PeerMailLimits;
|
|
138
|
+
/**
|
|
139
|
+
* Phrases that mark the run's AUTHORITY in a folded prompt. A peer that writes one of these is
|
|
140
|
+
* trying to speak as the supervisor, so intake refuses the envelope outright.
|
|
141
|
+
*
|
|
142
|
+
* The render-time fence in the inbox is the second half of this defence and neither half is
|
|
143
|
+
* sufficient alone: a fence loses to a body that closes it, and an intake filter loses to a body
|
|
144
|
+
* that invents a new authority phrase. Together they make forgery mechanically detectable and give
|
|
145
|
+
* the standing prompt one concrete boundary to bind to. Neither makes a model OBEY a boundary.
|
|
146
|
+
*/
|
|
147
|
+
declare const AUTHORITY_MARKERS: ReadonlyArray<string>;
|
|
148
|
+
/** The wire property carrying an envelope to a worker inbox. Deliberately its OWN discriminant:
|
|
149
|
+
* reusing `steer`/`answer` would let a peer mint a message on the parent's channels. */
|
|
150
|
+
declare const PEER_MAIL_WIRE_KEY = "mail";
|
|
151
|
+
/** The tool names a mail capability endpoint serves. It serves NOTHING else. */
|
|
152
|
+
declare const peerMailVerbNames: readonly ["send_mail", "read_mail"];
|
|
153
|
+
/** What a worker sees when it reads its own mailbox. */
|
|
154
|
+
interface PeerMailReadout {
|
|
155
|
+
/** The reading worker's own id, so a worker can address a reply correctly. */
|
|
156
|
+
readonly you: string;
|
|
157
|
+
/** Every envelope admitted to this worker so far, oldest first. */
|
|
158
|
+
readonly inbox: ReadonlyArray<PeerMailEnvelope>;
|
|
159
|
+
/** Live siblings this worker may write to (itself excluded). Without this a worker knows no
|
|
160
|
+
* peer's id and the channel is unusable. */
|
|
161
|
+
readonly peers: ReadonlyArray<{
|
|
162
|
+
readonly workerId: string;
|
|
163
|
+
readonly label: string;
|
|
164
|
+
}>;
|
|
165
|
+
readonly sent: number;
|
|
166
|
+
/** Sends still allowed, or `null` when this run set no send quota. */
|
|
167
|
+
readonly sendQuotaLeft: number | null;
|
|
168
|
+
readonly limits: PeerMailLimits;
|
|
169
|
+
}
|
|
170
|
+
interface PeerMailSendInput {
|
|
171
|
+
readonly to: unknown;
|
|
172
|
+
readonly kind: unknown;
|
|
173
|
+
readonly subject: unknown;
|
|
174
|
+
readonly body: unknown;
|
|
175
|
+
readonly evidenceRefs?: unknown;
|
|
176
|
+
readonly replyTo?: unknown;
|
|
177
|
+
}
|
|
178
|
+
interface PeerMailbox {
|
|
179
|
+
readonly limits: PeerMailLimits;
|
|
180
|
+
/**
|
|
181
|
+
* Publish the base URL of the capability listener once it has a port. Until it is set no spawn
|
|
182
|
+
* receives a mail endpoint: a capability nobody can reach is not worth handing out, and a URL
|
|
183
|
+
* built from an unassigned port would be a lie.
|
|
184
|
+
*/
|
|
185
|
+
setEndpoint(baseUrl: string): void;
|
|
186
|
+
/** Mint (idempotently, per assignment) the capability URL for one spawn. Undefined before the
|
|
187
|
+
* listener has published its endpoint. */
|
|
188
|
+
mintCapability(assignmentId: string): string | undefined;
|
|
189
|
+
/** Bind a minted capability to the concrete worker the spawn produced. Until this runs the
|
|
190
|
+
* capability can send nothing. */
|
|
191
|
+
bindCapability(assignmentId: string, workerId: string): void;
|
|
192
|
+
/** Resolve the capability path segment carried in a request URL. */
|
|
193
|
+
hasCapability(capabilityId: string): boolean;
|
|
194
|
+
/** The two tools a single capability serves, with the sender closed over. */
|
|
195
|
+
tools(capabilityId: string): McpToolDescriptor[];
|
|
196
|
+
send(capabilityId: string, input: PeerMailSendInput): Promise<PeerMailEvent>;
|
|
197
|
+
read(capabilityId: string): PeerMailReadout;
|
|
198
|
+
/** The parent's control: refuse every further mail on one thread. Returns false when the thread
|
|
199
|
+
* was already stopped. Mail already delivered is not recalled — this stops the next reply. */
|
|
200
|
+
stopThread(threadId: string): boolean;
|
|
201
|
+
/** Every attempt in order — delivered and refused alike. */
|
|
202
|
+
history(): ReadonlyArray<PeerMailEvent>;
|
|
203
|
+
}
|
|
204
|
+
interface PeerMailboxOptions {
|
|
205
|
+
readonly scope: Scope<unknown>;
|
|
206
|
+
/** Publish one attempt as a coordination event. Awaited, so a durable subscriber commits the
|
|
207
|
+
* record before the sender learns the outcome. */
|
|
208
|
+
readonly publish: (event: PeerMailEvent) => Promise<void>;
|
|
209
|
+
readonly limits?: Partial<PeerMailLimits>;
|
|
210
|
+
readonly now?: () => number;
|
|
211
|
+
}
|
|
212
|
+
/** True when `text` carries a phrase reserved for the run's authority. Case-insensitive, because
|
|
213
|
+
* the render is read by a model and case is not what distinguishes an instruction. */
|
|
214
|
+
declare function claimsAuthority(text: string): boolean;
|
|
215
|
+
/** True when `value` is an envelope this runtime produced. The worker inbox parses with this, so a
|
|
216
|
+
* malformed or partial wire object is discarded rather than rendered as a peer message. */
|
|
217
|
+
declare function isPeerMailEnvelope(value: unknown): value is PeerMailEnvelope;
|
|
218
|
+
/** Create the run's post office. One per manager scope; the manager's siblings are its addresses. */
|
|
219
|
+
declare function createPeerMailbox(opts: PeerMailboxOptions): PeerMailbox;
|
|
220
|
+
/**
|
|
221
|
+
* The two tools ONE capability serves. `capabilityId` is closed over and `from` is not a parameter,
|
|
222
|
+
* so the endpoint a worker holds can only ever speak as that worker. The descriptions carry the
|
|
223
|
+
* authority rule, because the receiving model reads them as part of the channel's contract.
|
|
224
|
+
*/
|
|
225
|
+
declare function peerMailTools(mailbox: PeerMailbox, capabilityId: string): McpToolDescriptor[];
|
|
226
|
+
//#endregion
|
|
227
|
+
//#region src/runtime/router-retry-policy.d.ts
|
|
228
|
+
/** Exact retry controls accepted at `AgentProfile.model.metadata.retry`. */
|
|
229
|
+
interface RouterRetryPolicy {
|
|
230
|
+
/** Total attempts, including the first request. */
|
|
231
|
+
readonly maxAttempts?: number;
|
|
232
|
+
/** Delay before the second attempt. Later delays grow exponentially. */
|
|
233
|
+
readonly initialBackoffMs?: number;
|
|
234
|
+
/** Maximum delay between attempts. */
|
|
235
|
+
readonly maxBackoffMs?: number;
|
|
236
|
+
/** Symmetric random variation around each delay, from 0 through 1. */
|
|
237
|
+
readonly jitter?: number;
|
|
238
|
+
/** HTTP statuses that may be retried. */
|
|
239
|
+
readonly retryStatuses?: ReadonlyArray<number>;
|
|
240
|
+
/** Deadline for receiving one attempt's response headers. Zero disables it. */
|
|
241
|
+
readonly requestTimeoutMs?: number;
|
|
242
|
+
}
|
|
243
|
+
//#endregion
|
|
244
|
+
//#region src/runtime/tool-loop.d.ts
|
|
245
|
+
/** Provider-neutral conversation record accepted by a tool-loop brain. */
|
|
246
|
+
type ToolLoopMessageRecord = Record<string, unknown>;
|
|
247
|
+
/** One provider-neutral tool request emitted by a tool-loop model. */
|
|
248
|
+
interface ToolLoopToolCall {
|
|
249
|
+
id: string;
|
|
250
|
+
name: string;
|
|
251
|
+
/** Raw JSON arguments emitted by the model. */
|
|
252
|
+
arguments: string;
|
|
253
|
+
}
|
|
254
|
+
/** Runtime-owned identity and cancellation for one logical inference call. The wrapper is frozen
|
|
255
|
+
* before dispatch; a transport may observe the signal but cannot replace the authority it names. */
|
|
256
|
+
interface ToolLoopCallContext {
|
|
257
|
+
readonly signal: AbortSignal;
|
|
258
|
+
readonly callId: string;
|
|
259
|
+
readonly correlationId: string;
|
|
260
|
+
}
|
|
261
|
+
/** One inference turn over the running conversation + the tool specs → the model's text, any
|
|
262
|
+
* tool calls, and token usage. The seam every brain satisfies. */
|
|
263
|
+
type ToolLoopChat = (messages: ReadonlyArray<ToolLoopMessageRecord>, tools: ReadonlyArray<ToolSpec>, context?: ToolLoopCallContext) => Promise<{
|
|
264
|
+
content?: string | null;
|
|
265
|
+
toolCalls: ToolLoopToolCall[];
|
|
266
|
+
usage?: {
|
|
267
|
+
input: number;
|
|
268
|
+
output: number;
|
|
269
|
+
reasoning?: number;
|
|
270
|
+
};
|
|
271
|
+
/** Caller-measured resource totals for this turn; omission makes enforced dimensions unknown. */
|
|
272
|
+
resources?: Spend['resources'];
|
|
273
|
+
/** Dollar value reported for the turn. It is not billed spend unless provenance says so. */
|
|
274
|
+
costUsd?: number;
|
|
275
|
+
costProvenance?: 'provider-receipt' | 'billing-receipt' | 'catalog-estimate';
|
|
276
|
+
/** The turn ran but its usage was not reported when the transport EXPECTED one (the streamed
|
|
277
|
+
* router transport asks for usage and this says it never arrived). A metering caller records an
|
|
278
|
+
* unknown turn on it; `runBrainLoop` itself ignores it. */
|
|
279
|
+
usageUnknown?: true;
|
|
280
|
+
/** Provider-observed model identity. Profile-bound callers validate it before accepting output. */
|
|
281
|
+
model?: string;
|
|
282
|
+
/** Provider-reported prompt-cache evidence; missing fields remain missing. */
|
|
283
|
+
promptCache?: Readonly<Record<string, number | string>>;
|
|
284
|
+
/** Physical HTTP/injected-transport attempts spent by this one logical call. */
|
|
285
|
+
transportAttempts?: number;
|
|
286
|
+
}>;
|
|
287
|
+
/** Self-compaction — bound the loop's OWN context window the way a fresh-respawn (dumb-Ralph) loop
|
|
288
|
+
* does, but in place. A stateless chat API re-sends the WHOLE running conversation every turn, so an
|
|
289
|
+
* agent that accumulates dozens of turns of tool results re-bills its entire transcript on every
|
|
290
|
+
* inference — the context-overflow-one-level-up that the conserved budget pool cannot fix. With
|
|
291
|
+
* compaction set, once the conversation exceeds `thresholdTokens` the accumulated middle (every prior
|
|
292
|
+
* assistant turn + tool result) is distilled into ONE compact progress note and the conversation is
|
|
293
|
+
* reset to `[...head, digest]`: the preserved head (system + the original task) survives, the stale
|
|
294
|
+
* turn-by-turn history does not. The model keeps deciding; it stops re-billing the whole transcript.
|
|
295
|
+
* Fires at a CLEAN turn boundary (after a turn's tool results are folded in, before the next
|
|
296
|
+
* inference) so it never orphans an assistant `tool_calls` from its `tool` replies. */
|
|
297
|
+
interface ToolLoopCompaction {
|
|
298
|
+
/** Compact once the estimated token count of the conversation exceeds this. */
|
|
299
|
+
readonly thresholdTokens: number;
|
|
300
|
+
/** Distill the conversation into a compact progress note that REPLACES the middle. Receives the
|
|
301
|
+
* full conversation (so it can summarize everything done so far); returns the digest string. */
|
|
302
|
+
readonly distill: (messages: ReadonlyArray<ToolLoopMessageRecord>) => Promise<string> | string;
|
|
303
|
+
/** Leading messages preserved verbatim (system + the original task). Default 2. */
|
|
304
|
+
readonly preserveHead?: number;
|
|
305
|
+
/** Token estimator over the conversation. Default ≈ chars/4 (incl. tool-call arguments). */
|
|
306
|
+
readonly estimateTokens?: (messages: ReadonlyArray<ToolLoopMessageRecord>) => number;
|
|
307
|
+
/** Notified each time a compaction fires — for observability/metering. */
|
|
308
|
+
readonly onCompact?: (info: {
|
|
309
|
+
turn: number;
|
|
310
|
+
beforeTokens: number;
|
|
311
|
+
afterTokens: number;
|
|
312
|
+
}) => void;
|
|
313
|
+
}
|
|
314
|
+
/** Public supervisor-facing compaction config: same knobs as the primitive, but `distill` is optional
|
|
315
|
+
* because the supervisor has a default digest that combines a brain note with live worker state. */
|
|
316
|
+
type ToolLoopCompactionOptions = Omit<ToolLoopCompaction, 'distill'> & {
|
|
317
|
+
readonly distill?: ToolLoopCompaction['distill'];
|
|
318
|
+
};
|
|
319
|
+
//#endregion
|
|
320
|
+
//#region src/runtime/router-client.d.ts
|
|
321
|
+
/**
|
|
322
|
+
* Connection details for Runtime's Router-backed executors.
|
|
323
|
+
*
|
|
324
|
+
* This is deliberately transport-only: model, prompt, tools, generation settings, and retry
|
|
325
|
+
* policy belong to the exact executable `AgentProfile` consumed by `streamAgentTurn`.
|
|
326
|
+
*/
|
|
327
|
+
interface RouterTransportConfig {
|
|
328
|
+
routerBaseUrl: string;
|
|
329
|
+
routerKey: string;
|
|
330
|
+
/** Injectable OpenAI-compatible transport. Optional usage.resources carries measured turn totals. */
|
|
331
|
+
complete?: (body: Record<string, unknown>, request?: {
|
|
332
|
+
readonly headers: Readonly<Record<string, string>>;
|
|
333
|
+
readonly signal?: AbortSignal;
|
|
334
|
+
}) => Promise<unknown>;
|
|
335
|
+
}
|
|
336
|
+
/**
|
|
337
|
+
* Private request configuration used by Runtime's Router adapter.
|
|
338
|
+
*
|
|
339
|
+
* Do not export this through a package entry point. Public callers execute a concrete
|
|
340
|
+
* `AgentProfile` through `createExecutor` + `streamAgentTurn`; only Runtime may lower that profile
|
|
341
|
+
* into these provider request fields.
|
|
342
|
+
*/
|
|
343
|
+
interface RouterConfig extends RouterTransportConfig {
|
|
344
|
+
model: string;
|
|
345
|
+
/** Exact retry controls lowered from `AgentProfile.model.metadata.retry`. */
|
|
346
|
+
retry?: RouterRetryPolicy;
|
|
347
|
+
/**
|
|
348
|
+
* Optional ceiling for one completion, forwarded as `max_tokens`.
|
|
349
|
+
*
|
|
350
|
+
* A REASONING model spends this budget on hidden thinking BEFORE it emits a visible token, so
|
|
351
|
+
* the default can truncate one mid-thought and return no content at all — observed live with a
|
|
352
|
+
* model that spent 8,188 of the 8,192 on reasoning and answered with nothing. Raise it for a
|
|
353
|
+
* thinking model; the ceiling belongs to the router and model a caller chose, which is why it
|
|
354
|
+
* lives here rather than on one call site.
|
|
355
|
+
*/
|
|
356
|
+
maxTokens?: number;
|
|
357
|
+
/**
|
|
358
|
+
* Optional ceiling on TOTAL completion tokens — visible answer plus hidden reasoning —
|
|
359
|
+
* forwarded as `max_completion_tokens`. Distinct from `maxTokens`: a reasoning model can spend
|
|
360
|
+
* an entire `max_tokens` budget on hidden thinking, so only this field bounds what the provider
|
|
361
|
+
* bills for one completion.
|
|
362
|
+
*/
|
|
363
|
+
maxCompletionTokens?: number;
|
|
364
|
+
/**
|
|
365
|
+
* Take the tool-calling completion over SSE instead of one buffered POST. Off by default —
|
|
366
|
+
* `routerChatWithTools` never streams, and every existing caller keeps the buffered transport
|
|
367
|
+
* byte for byte.
|
|
368
|
+
*
|
|
369
|
+
* Why it exists: a buffered POST holds one connection idle for the WHOLE completion, and a
|
|
370
|
+
* supervisor turn is the longest completion in the system. An intermediary gateway with an
|
|
371
|
+
* idle-read timeout kills that connection mid-completion (the 524/503 family). A streamed
|
|
372
|
+
* response puts bytes on the wire from the first generated token on, so the connection is only
|
|
373
|
+
* idle through prefill. It does NOT shorten prefill, so a gateway whose deadline is
|
|
374
|
+
* time-to-FIRST-byte is unaffected; only an idle-timeout gateway is.
|
|
375
|
+
*
|
|
376
|
+
* Mutually exclusive with `complete`: the injected transport returns one parsed JSON body and has
|
|
377
|
+
* no stream to read, so setting both throws rather than silently taking the buffered path.
|
|
378
|
+
*
|
|
379
|
+
* WHICH PATHS CAN OPT IN. This flag is read in exactly one place (the private `chatWithTools` transport switch), so
|
|
380
|
+
* every entry point that takes a caller-supplied `RouterConfig` honors it: `routerBrain`,
|
|
381
|
+
* `routerToolLoop`, and `supervisorAgent` (which spreads `deps.router` into the brain's config —
|
|
382
|
+
* the supervisor turn this exists for). Two production call sites build a `RouterConfig` literal
|
|
383
|
+
* from their own options and therefore CANNOT express it today: the bench strategy's
|
|
384
|
+
* `routerToolLoop` config in `strategy.ts` and the local sandbox client's `routerBrain` config in
|
|
385
|
+
* `local-sandbox-client.ts`. Neither drives a supervisor-length turn; setting `stream` on a
|
|
386
|
+
* config handed to either has no path to reach them, and they stay buffered.
|
|
387
|
+
*/
|
|
388
|
+
stream?: boolean;
|
|
389
|
+
}
|
|
390
|
+
interface ToolSpec {
|
|
391
|
+
type: 'function';
|
|
392
|
+
function: {
|
|
393
|
+
name: string;
|
|
394
|
+
description?: string;
|
|
395
|
+
parameters: unknown;
|
|
396
|
+
};
|
|
397
|
+
}
|
|
398
|
+
//#endregion
|
|
399
|
+
//#region src/mcp/local-harness.d.ts
|
|
400
|
+
/**
|
|
401
|
+
* Local coding harness available inside the sandbox — a narrowing of the shared `HarnessType`
|
|
402
|
+
* vocabulary, NOT a private spelling of it. The harness id is `claude-code`; `claude` is the
|
|
403
|
+
* EXECUTABLE name and lives only in the `command` field below. Keeping one vocabulary is what
|
|
404
|
+
* lets a `LocalHarness` be handed straight to the profile materializer and the capability table
|
|
405
|
+
* with no translation step.
|
|
406
|
+
*/
|
|
407
|
+
type LocalHarness = Extract<HarnessType, 'claude-code' | 'codex' | 'opencode' | 'pi'>;
|
|
408
|
+
/** Every local harness, in table order — the one list `AGENT_RUNTIME_LOCAL_HARNESSES` and any
|
|
409
|
+
* other harness enumeration reads, so adding a row above is the only edit a new harness needs. */
|
|
410
|
+
declare const LOCAL_HARNESSES: ReadonlyArray<LocalHarness>;
|
|
411
|
+
/** The harness a caller gets when it expresses no preference. A composition-root default, not a
|
|
412
|
+
* capability claim: one constant so the several entry points cannot drift apart. */
|
|
413
|
+
declare const DEFAULT_LOCAL_HARNESS: LocalHarness;
|
|
414
|
+
/** The CLI binary a harness id runs. The two are NOT the same string (`claude-code` runs `claude`),
|
|
415
|
+
* so anything spawning a harness — a version probe, a login check — reads it from here rather than
|
|
416
|
+
* passing the harness id as a command. */
|
|
417
|
+
declare function localHarnessExecutable(harness: LocalHarness): string;
|
|
418
|
+
/**
|
|
419
|
+
* Whether the harness's native control can express this reasoning effort. Admission checks read
|
|
420
|
+
* this so a profile the invocation would later refuse is rejected BEFORE any workspace state is
|
|
421
|
+
* created, against the same table that emits the argv.
|
|
422
|
+
*/
|
|
423
|
+
declare function harnessSupportsReasoningEffort(harness: LocalHarness, reasoningEffort: ReasoningEffort): boolean;
|
|
424
|
+
/** @experimental */
|
|
425
|
+
interface RunLocalHarnessOptions {
|
|
426
|
+
harness: LocalHarness;
|
|
427
|
+
/** Working directory for the subprocess (typically a worktree path). */
|
|
428
|
+
cwd: string;
|
|
429
|
+
/** Prompt forwarded as the harness CLI's task argument. */
|
|
430
|
+
taskPrompt: string;
|
|
431
|
+
/**
|
|
432
|
+
* Pre-built command + args (e.g. from `harnessInvocation` so the full authored
|
|
433
|
+
* `AgentProfile` — systemPrompt + model — reaches the harness). When set it OVERRIDES the
|
|
434
|
+
* default prompt-only `buildArgs(taskPrompt)` path; `command` defaults to the harness's
|
|
435
|
+
* default binary when only `args` is supplied. When absent the legacy prompt-only shape
|
|
436
|
+
* is used unchanged.
|
|
437
|
+
*/
|
|
438
|
+
invocation?: {
|
|
439
|
+
command?: string;
|
|
440
|
+
args: ReadonlyArray<string>;
|
|
441
|
+
};
|
|
442
|
+
/** Allow autonomous edits without an interactive approval gate, using whichever bypass argv the
|
|
443
|
+
* harness declares. Use only when `cwd` is an isolated candidate worktree. */
|
|
444
|
+
dangerouslySkipPermissions?: boolean;
|
|
445
|
+
/** Isolate Codex from ambient configuration/instructions and require JSONL token usage.
|
|
446
|
+
* The invocation should come from `harnessInvocation(..., { codexReproducible: true })`. */
|
|
447
|
+
codexReproducible?: boolean;
|
|
448
|
+
/** Absolute host paths that reproducible Codex must not read. The normalized set is compiled
|
|
449
|
+
* into the controlled permission profile and its digest is returned in execution evidence. */
|
|
450
|
+
codexReadDeniedPaths?: ReadonlyArray<string>;
|
|
451
|
+
/** Optional wall-clock kill deadline (ms). Omit it for no timer. A positive value sends
|
|
452
|
+
* SIGTERM on expiry. */
|
|
453
|
+
timeoutMs?: number;
|
|
454
|
+
/** Newest stdout/stderr bytes retained per stream. Default 64 MiB. */
|
|
455
|
+
maxOutputBytes?: number;
|
|
456
|
+
/** Caller cancellation. SIGTERM is sent on abort. */
|
|
457
|
+
signal?: AbortSignal;
|
|
458
|
+
/** Override env (defaults to inheriting from the parent). */
|
|
459
|
+
env?: NodeJS.ProcessEnv;
|
|
460
|
+
/**
|
|
461
|
+
* Test seam — inject a custom spawner so unit tests can mock the
|
|
462
|
+
* subprocess without touching the OS. Defaults to node's `child_process.spawn`.
|
|
463
|
+
*/
|
|
464
|
+
spawn?: (command: string, args: ReadonlyArray<string>, opts: {
|
|
465
|
+
cwd: string;
|
|
466
|
+
env: NodeJS.ProcessEnv;
|
|
467
|
+
stdio: 'pipe';
|
|
468
|
+
detached: boolean;
|
|
469
|
+
}) => ChildProcess;
|
|
470
|
+
/** Test seam for locating the native Codex executable before it is staged in the worktree. */
|
|
471
|
+
resolveCodexExecutable?: (command: string, env: NodeJS.ProcessEnv) => Promise<string>;
|
|
472
|
+
}
|
|
473
|
+
/**
|
|
474
|
+
* Exact aggregate usage emitted by Codex's terminal `turn.completed` JSONL event.
|
|
475
|
+
*
|
|
476
|
+
* `cachedInputTokens` is a part of `inputTokens` and `reasoningOutputTokens` is a part of
|
|
477
|
+
* `outputTokens`; neither adds to the total it describes. `cacheWriteInputTokens` is optional
|
|
478
|
+
* because the codex CLI reports it and a provider-normalized capture omits it, and an absent
|
|
479
|
+
* counter must stay absent rather than become a zero that claims no cache write was measured.
|
|
480
|
+
*/
|
|
481
|
+
interface CodexTokenUsage {
|
|
482
|
+
inputTokens: number;
|
|
483
|
+
cachedInputTokens: number;
|
|
484
|
+
outputTokens: number;
|
|
485
|
+
reasoningOutputTokens: number;
|
|
486
|
+
cacheWriteInputTokens?: number;
|
|
487
|
+
}
|
|
488
|
+
/** Isolation settings asserted before a reproducible Codex run is allowed to start. */
|
|
489
|
+
interface CodexExecutionPolicy {
|
|
490
|
+
sessionPersistence: 'ephemeral';
|
|
491
|
+
userConfig: false;
|
|
492
|
+
rules: false;
|
|
493
|
+
projectInstructions: false;
|
|
494
|
+
skillInstructions: false;
|
|
495
|
+
appInstructions: false;
|
|
496
|
+
toolSuggestions: false;
|
|
497
|
+
multiAgentInstructions: false;
|
|
498
|
+
sandbox: 'workspace-write';
|
|
499
|
+
permissionProfile: 'agent_runtime_reproducible';
|
|
500
|
+
approvalPolicy: 'never';
|
|
501
|
+
shellNetwork: false;
|
|
502
|
+
webSearch: false;
|
|
503
|
+
serviceTier: 'default';
|
|
504
|
+
shellEnvironment: 'core-filtered';
|
|
505
|
+
loginShell: false;
|
|
506
|
+
credentialsReadable: false;
|
|
507
|
+
hostHomeReadable: false;
|
|
508
|
+
procEnvironment: 'private-sanitized';
|
|
509
|
+
sensitiveEnvironmentNamesVisible: false;
|
|
510
|
+
parentRepoRead: false;
|
|
511
|
+
gitMetadata: false;
|
|
512
|
+
temporaryDirectory: 'workspace-private';
|
|
513
|
+
stagedExecutable: 'static-elf-read-only';
|
|
514
|
+
callerReadDeniedPaths: 'enforced';
|
|
515
|
+
containerSockets: false;
|
|
516
|
+
}
|
|
517
|
+
/** Zero-model-call evidence for the exact Codex process about to run. */
|
|
518
|
+
interface CodexExecutionEvidence {
|
|
519
|
+
cliVersion: string;
|
|
520
|
+
executableSha256: string;
|
|
521
|
+
/** SHA-256 of the exact composed prompt argument proved present in the rendered prompt. */
|
|
522
|
+
requestedPromptSha256: string;
|
|
523
|
+
effectivePromptSha256: string;
|
|
524
|
+
nonPromptArgsSha256: string;
|
|
525
|
+
controlledConfigSha256: string;
|
|
526
|
+
/** Sorted normalized paths compiled into the permission profile. */
|
|
527
|
+
readDeniedPaths: string[];
|
|
528
|
+
readDeniedPathsSha256: string;
|
|
529
|
+
readDeniedPathCount: number;
|
|
530
|
+
policy: CodexExecutionPolicy;
|
|
531
|
+
}
|
|
532
|
+
/** @experimental */
|
|
533
|
+
interface LocalHarnessResult {
|
|
534
|
+
/** OS exit code. `null` when killed before exit. */
|
|
535
|
+
exitCode: number | null;
|
|
536
|
+
/** Concatenated stdout. */
|
|
537
|
+
stdout: string;
|
|
538
|
+
/** Concatenated stderr. */
|
|
539
|
+
stderr: string;
|
|
540
|
+
/** Set when the process exited via signal (timeout / abort). */
|
|
541
|
+
killedBySignal: NodeJS.Signals | null;
|
|
542
|
+
/** Wall-clock duration ms (spawn → exit). */
|
|
543
|
+
durationMs: number;
|
|
544
|
+
/** Set when timeoutMs elapsed before exit. */
|
|
545
|
+
timedOut: boolean;
|
|
546
|
+
/**
|
|
547
|
+
* Set when the caller's AbortSignal fired before this result settled.
|
|
548
|
+
* Optional so injected runners and stored results from older releases remain valid.
|
|
549
|
+
*/
|
|
550
|
+
aborted?: boolean;
|
|
551
|
+
/** Present for a reproducible Codex run; parsed from the real terminal JSONL event. */
|
|
552
|
+
usage?: CodexTokenUsage;
|
|
553
|
+
/** Present for reproducible Codex runs; generated and checked before model execution. */
|
|
554
|
+
evidence?: CodexExecutionEvidence;
|
|
555
|
+
}
|
|
556
|
+
/**
|
|
557
|
+
* Spawn a local coding harness CLI as a subprocess + collect its output.
|
|
558
|
+
*
|
|
559
|
+
* NOT responsible for parsing the harness's output or extracting a diff —
|
|
560
|
+
* the in-process executor's `streamPrompt` orchestrates `git diff` against
|
|
561
|
+
* the worktree after this resolves. This function is intentionally narrow:
|
|
562
|
+
* spawn, wait, capture, return.
|
|
563
|
+
*
|
|
564
|
+
* Fails loud — throws when:
|
|
565
|
+
* - `cwd` doesn't exist (subprocess emits ENOENT; surfaced as Error)
|
|
566
|
+
* - the harness binary is not on PATH (ENOENT)
|
|
567
|
+
* - the caller signal was already aborted before process launch
|
|
568
|
+
*
|
|
569
|
+
* Does NOT throw when:
|
|
570
|
+
* - the subprocess exits non-zero (`result.exitCode` carries the code)
|
|
571
|
+
* - a non-reproducible subprocess is aborted / timed out (`result.aborted` /
|
|
572
|
+
* `result.timedOut` carries the reason even when a TERM-aware child exits zero)
|
|
573
|
+
*
|
|
574
|
+
* Reproducible Codex additionally requires a terminal usage event. If cancellation
|
|
575
|
+
* prevents that event, this rejects with `CodexExecutionDiagnosticError` instead of
|
|
576
|
+
* returning an incomplete reproducibility receipt.
|
|
577
|
+
*
|
|
578
|
+
* @experimental
|
|
579
|
+
*/
|
|
580
|
+
declare function runLocalHarness(options: RunLocalHarnessOptions): Promise<LocalHarnessResult>;
|
|
581
|
+
/**
|
|
582
|
+
* Parse and validate the one terminal usage event emitted by `codex exec --json`.
|
|
583
|
+
*
|
|
584
|
+
* The JSONL framing is this surface's own; the usage RECORD is read by `parseCodexUsageRecord`,
|
|
585
|
+
* the one codex usage reader the sandbox decoder also calls, so both surfaces hold the same field
|
|
586
|
+
* policy and the same two cross-field invariants.
|
|
587
|
+
*/
|
|
588
|
+
declare function parseCodexTokenUsage(stdout: string): CodexTokenUsage;
|
|
589
|
+
//#endregion
|
|
590
|
+
//#region src/mcp/worktree.d.ts
|
|
591
|
+
/**
|
|
592
|
+
*
|
|
593
|
+
* Git worktree helpers for the in-process delegation executor. Each
|
|
594
|
+
* delegation runs in its own worktree so multiple parallel harness
|
|
595
|
+
* subprocesses (claude / codex / opencode in a 3-way fanout) don't clobber
|
|
596
|
+
* each other's edits on the shared workspace.
|
|
597
|
+
*
|
|
598
|
+
* Worktrees live under `<repoRoot>/.agent-worktrees/<runId>/`. After the
|
|
599
|
+
* harness exits + the diff is captured, the worktree is removed.
|
|
600
|
+
*
|
|
601
|
+
* All operations spawn `git` via `child_process.spawn` synchronously
|
|
602
|
+
* (via a `runGit` helper). Stays narrow on purpose: no commits, no rebases.
|
|
603
|
+
* Diff capture stages all changes (`git add -A`) into the ephemeral worktree's
|
|
604
|
+
* index so created (untracked) files appear in the `--cached` diff.
|
|
605
|
+
*
|
|
606
|
+
* @experimental
|
|
607
|
+
*/
|
|
608
|
+
/** @experimental */
|
|
609
|
+
interface WorktreeHandle {
|
|
610
|
+
/** Absolute path to the worktree directory. */
|
|
611
|
+
path: string;
|
|
612
|
+
/** SHA the worktree was created at. */
|
|
613
|
+
baseSha: string;
|
|
614
|
+
/** Branch name created for this worktree (typically `delegate/<runId>`). */
|
|
615
|
+
branch: string;
|
|
616
|
+
}
|
|
617
|
+
/** @experimental */
|
|
618
|
+
interface CreateWorktreeOptions {
|
|
619
|
+
/** Absolute path to the main git checkout. */
|
|
620
|
+
repoRoot: string;
|
|
621
|
+
/** Unique id for the worktree path + branch. Use the delegation run id. */
|
|
622
|
+
runId: string;
|
|
623
|
+
/** Parent directory the worktree lives under. Defaults to `.agent-worktrees`. */
|
|
624
|
+
variantsDir?: string;
|
|
625
|
+
/** Override the base ref (default `HEAD`). */
|
|
626
|
+
baseRef?: string;
|
|
627
|
+
/** Test seam — inject a custom git runner. */
|
|
628
|
+
runGit?: GitRunner;
|
|
629
|
+
}
|
|
630
|
+
/** @experimental */
|
|
631
|
+
interface DiffOptions {
|
|
632
|
+
/** Worktree to diff. */
|
|
633
|
+
worktree: WorktreeHandle;
|
|
634
|
+
/** What to compare against. Default `worktree.baseSha`. */
|
|
635
|
+
baseRef?: string;
|
|
636
|
+
/**
|
|
637
|
+
* Repository-relative input paths to omit from the captured worker patch.
|
|
638
|
+
* Paths are passed to Git with literal exclusion magic, so profile-provided
|
|
639
|
+
* `*`, `?`, `[` and `:` characters can never expand into broader pathspecs.
|
|
640
|
+
*/
|
|
641
|
+
excludePaths?: ReadonlyArray<string>;
|
|
642
|
+
/** Test seam. */
|
|
643
|
+
runGit?: GitRunner;
|
|
644
|
+
}
|
|
645
|
+
/** @experimental */
|
|
646
|
+
interface DiffResult {
|
|
647
|
+
patch: string;
|
|
648
|
+
stats: {
|
|
649
|
+
filesChanged: number;
|
|
650
|
+
insertions: number;
|
|
651
|
+
deletions: number;
|
|
652
|
+
};
|
|
653
|
+
}
|
|
654
|
+
/** @experimental */
|
|
655
|
+
interface RemoveWorktreeOptions {
|
|
656
|
+
worktree: WorktreeHandle;
|
|
657
|
+
repoRoot: string;
|
|
658
|
+
/** Force removal even if dirty (default true; the loser of a fanout has uncommitted changes). */
|
|
659
|
+
force?: boolean;
|
|
660
|
+
/** Test seam. */
|
|
661
|
+
runGit?: GitRunner;
|
|
662
|
+
}
|
|
663
|
+
/** Pluggable git runner (sync) — replaceable in tests. */
|
|
664
|
+
type GitRunner = (args: ReadonlyArray<string>, opts: {
|
|
665
|
+
cwd: string;
|
|
666
|
+
}) => {
|
|
667
|
+
stdout: string;
|
|
668
|
+
stderr: string;
|
|
669
|
+
exitCode: number;
|
|
670
|
+
};
|
|
671
|
+
/** Checkout a fresh git worktree for a delegation run on a new branch under `variantsDir`. @experimental */
|
|
672
|
+
declare function createWorktree(options: CreateWorktreeOptions): Promise<WorktreeHandle>;
|
|
673
|
+
/** Stage worker changes and return the diff + shortstat, excluding declared input paths. @experimental */
|
|
674
|
+
declare function captureWorktreeDiff(options: DiffOptions): Promise<DiffResult>;
|
|
675
|
+
/**
|
|
676
|
+
* Remove a git worktree and delete its branch. Already-removed paths are harmless; every other
|
|
677
|
+
* Git failure rejects so callers cannot report a worktree as destroyed when cleanup failed.
|
|
678
|
+
* @experimental
|
|
679
|
+
*/
|
|
680
|
+
declare function removeWorktree(options: RemoveWorktreeOptions): Promise<void>;
|
|
681
|
+
//#endregion
|
|
682
|
+
//#region src/mcp/worktree-harness.d.ts
|
|
683
|
+
/** Outcome of one verification command run in the worktree (test or typecheck). */
|
|
684
|
+
interface WorktreeCommandResult {
|
|
685
|
+
/** The shell command line that was run. */
|
|
686
|
+
command: string;
|
|
687
|
+
/** Did the command exit 0? The PASS signal a deliverable gate / coder output reads. */
|
|
688
|
+
passed: boolean;
|
|
689
|
+
/** OS exit code, or `null` when killed before exit. */
|
|
690
|
+
exitCode: number | null;
|
|
691
|
+
/** Combined stdout+stderr (capped) — surfaced in traces for diagnosis. */
|
|
692
|
+
output: string;
|
|
693
|
+
}
|
|
694
|
+
/** Proof of the profile inputs delivered before the worker process started. */
|
|
695
|
+
interface WorktreeProfileMaterializationReceipt {
|
|
696
|
+
/** Digest of the exact materializer plan: files, modes, environment, flags, and unsupported rows. */
|
|
697
|
+
workspacePlanDigest: string;
|
|
698
|
+
/** Repository-relative profile input files written into the worker worktree. */
|
|
699
|
+
writtenPaths: string[];
|
|
700
|
+
/** Must be empty on a successful run because this path fails closed. */
|
|
701
|
+
unsupported: WorkspacePlanReceipt['unsupported'];
|
|
702
|
+
/** Environment variable names added to the worker process. Values remain out of telemetry. */
|
|
703
|
+
environmentNames: string[];
|
|
704
|
+
/** Exact additional CLI arguments emitted by the materializer. */
|
|
705
|
+
flags: string[];
|
|
706
|
+
/** `resources.instructions` bypasses native project files so reproducible Codex cannot drop it. */
|
|
707
|
+
resourceInstructions: {
|
|
708
|
+
delivery: 'none' | 'invocation-prompt';
|
|
709
|
+
sha256: string | null;
|
|
710
|
+
byteLength: number;
|
|
711
|
+
};
|
|
712
|
+
}
|
|
713
|
+
/** The canonical result of one worktree-harness run, projected by each port to its own shape. */
|
|
714
|
+
interface WorktreeHarnessResult {
|
|
715
|
+
/** The branch the worktree was cut on (`delegate/<runId>`). */
|
|
716
|
+
branch: string;
|
|
717
|
+
/** `git diff` of the worktree against its base — the unified patch the harness produced. */
|
|
718
|
+
patch: string;
|
|
719
|
+
/** Shortstat-derived change counts. */
|
|
720
|
+
stats: {
|
|
721
|
+
filesChanged: number;
|
|
722
|
+
insertions: number;
|
|
723
|
+
deletions: number;
|
|
724
|
+
};
|
|
725
|
+
/**
|
|
726
|
+
* Exact profile materialization applied before the harness launched.
|
|
727
|
+
* Absent on transports that cannot return a materializer receipt; never fabricated.
|
|
728
|
+
*/
|
|
729
|
+
profileMaterialization?: WorktreeProfileMaterializationReceipt;
|
|
730
|
+
/** The harness subprocess outcome. */
|
|
731
|
+
harness: {
|
|
732
|
+
name: LocalHarness | 'bridge';
|
|
733
|
+
exitCode: number | null;
|
|
734
|
+
timedOut: boolean;
|
|
735
|
+
killedBySignal: NodeJS.Signals | null;
|
|
736
|
+
durationMs: number;
|
|
737
|
+
stdout: string;
|
|
738
|
+
stderr: string;
|
|
739
|
+
/** Exact Codex JSONL usage when reproducible mode is enabled. */
|
|
740
|
+
usage?: CodexTokenUsage;
|
|
741
|
+
/** Installed CLI version captured immediately before execution. */
|
|
742
|
+
cliVersion?: string;
|
|
743
|
+
/** SHA-256 of the native Codex executable staged read-only in the candidate worktree. */
|
|
744
|
+
executableSha256?: string;
|
|
745
|
+
/** SHA-256 of the exact composed prompt argument proved present in Codex's rendered prompt. */
|
|
746
|
+
requestedPromptSha256?: string;
|
|
747
|
+
/** SHA-256 of `codex debug prompt-input` output for the exact isolated prompt. */
|
|
748
|
+
effectivePromptSha256?: string;
|
|
749
|
+
/** SHA-256 of the exact executable + argv with prompt content replaced by `<PROMPT>`. */
|
|
750
|
+
nonPromptArgsSha256?: string;
|
|
751
|
+
/** SHA-256 of the isolated config that fixes permissions and shell environment. */
|
|
752
|
+
controlledConfigSha256?: string;
|
|
753
|
+
/** SHA-256 of the normalized caller-supplied host read-denial paths. */
|
|
754
|
+
readDeniedPathsSha256?: string;
|
|
755
|
+
/** Sorted normalized caller-supplied host read-denial paths. */
|
|
756
|
+
readDeniedPaths?: string[];
|
|
757
|
+
/** Number of normalized caller-supplied host read-denial paths. */
|
|
758
|
+
readDeniedPathCount?: number;
|
|
759
|
+
/** Explicit isolation claims checked before model execution. */
|
|
760
|
+
executionPolicy?: CodexExecutionPolicy;
|
|
761
|
+
};
|
|
762
|
+
/** Verification signals derived in the live worktree (present only when commands were given). */
|
|
763
|
+
checks?: {
|
|
764
|
+
tests?: WorktreeCommandResult;
|
|
765
|
+
typecheck?: WorktreeCommandResult;
|
|
766
|
+
};
|
|
767
|
+
}
|
|
768
|
+
/** The single shell-command-in-worktree runner seam (replaces the per-executor copies). */
|
|
769
|
+
type WorktreeCheckRunner = (opts: {
|
|
770
|
+
command: string;
|
|
771
|
+
cwd: string;
|
|
772
|
+
timeoutMs: number;
|
|
773
|
+
signal?: AbortSignal;
|
|
774
|
+
}) => Promise<{
|
|
775
|
+
exitCode: number | null;
|
|
776
|
+
output: string;
|
|
777
|
+
}>;
|
|
778
|
+
/** The canonical result of one in-place harness run. The edits are the DIRECTORY, not a patch:
|
|
779
|
+
* the caller supplied the workspace and reads it directly. */
|
|
780
|
+
interface InPlaceHarnessResult {
|
|
781
|
+
/** The directory the harness ran in, exactly as supplied. */
|
|
782
|
+
workspacePath: string;
|
|
783
|
+
/** Exact profile materialization applied before the harness launched, and removed after it. */
|
|
784
|
+
profileMaterialization: WorktreeProfileMaterializationReceipt;
|
|
785
|
+
/** The harness subprocess outcome. */
|
|
786
|
+
harness: WorktreeHarnessResult['harness'];
|
|
787
|
+
}
|
|
788
|
+
//#endregion
|
|
789
|
+
//#region src/runtime/provider-placement.d.ts
|
|
790
|
+
/** Caller-declared execution placement. Matching never changes the authored profile.
|
|
791
|
+
* @experimental */
|
|
792
|
+
interface ProviderPlacement {
|
|
793
|
+
id: string;
|
|
794
|
+
match: {
|
|
795
|
+
harness: NonNullable<AgentProfile['harness']>;
|
|
796
|
+
provider?: string;
|
|
797
|
+
model?: string;
|
|
798
|
+
};
|
|
799
|
+
create: Omit<Partial<CreateAgentEnvironmentInput>, 'profile' | 'signal' | 'idempotencyKey' | 'requestedId' | 'runtimeAttachments' | 'backend'> & {
|
|
800
|
+
backend: string;
|
|
801
|
+
};
|
|
802
|
+
promptOptions?: ProviderPromptOptions;
|
|
803
|
+
}
|
|
804
|
+
//#endregion
|
|
805
|
+
//#region src/runtime/tangle-sandbox-exact-process-provider.d.ts
|
|
806
|
+
type SandboxControlClient = Pick<Sandbox, 'create' | 'get' | 'list'>;
|
|
807
|
+
interface CreateTangleSandboxExactProcessProviderOptions {
|
|
808
|
+
name?: string;
|
|
809
|
+
}
|
|
810
|
+
/**
|
|
811
|
+
* Adapt Tangle Sandbox's managed control runtime to Runtime's exact-process provider.
|
|
812
|
+
*
|
|
813
|
+
* The adapter deliberately exposes no ordinary agent environment: an exact experiment
|
|
814
|
+
* must start a fresh Sandbox with no managed agent and launch its declared argv directly.
|
|
815
|
+
*/
|
|
816
|
+
declare function createTangleSandboxExactProcessProvider(client: SandboxControlClient, options?: CreateTangleSandboxExactProcessProviderOptions): AgentEnvironmentProvider;
|
|
817
|
+
//#endregion
|
|
818
|
+
//#region src/runtime/environment-provider.d.ts
|
|
819
|
+
/** Provider object or registry name accepted by runtime provider adapters.
|
|
820
|
+
* @experimental */
|
|
821
|
+
type AgentEnvironmentProviderRef = AgentEnvironmentProvider | string;
|
|
822
|
+
/** In-memory registry for named `AgentEnvironmentProvider` instances.
|
|
823
|
+
* @experimental */
|
|
824
|
+
interface AgentEnvironmentProviderRegistry {
|
|
825
|
+
register(provider: AgentEnvironmentProvider, options?: {
|
|
826
|
+
replace?: boolean;
|
|
827
|
+
}): void;
|
|
828
|
+
has(name: string): boolean;
|
|
829
|
+
get(name: string): AgentEnvironmentProvider | undefined;
|
|
830
|
+
require(name: string): AgentEnvironmentProvider;
|
|
831
|
+
names(): string[];
|
|
832
|
+
providers(): AgentEnvironmentProvider[];
|
|
833
|
+
capabilities(name: string): Promise<AgentEnvironmentCapabilities>;
|
|
834
|
+
}
|
|
835
|
+
/** Create a registry that resolves provider names to concrete provider instances.
|
|
836
|
+
* @experimental */
|
|
837
|
+
declare function createAgentEnvironmentProviderRegistry(providers?: Iterable<AgentEnvironmentProvider>): AgentEnvironmentProviderRegistry;
|
|
838
|
+
/** Resolve a provider instance or registry name, failing loudly when a name is unknown.
|
|
839
|
+
* @experimental */
|
|
840
|
+
declare function resolveAgentEnvironmentProvider(provider: AgentEnvironmentProviderRef, registry?: AgentEnvironmentProviderRegistry): AgentEnvironmentProvider;
|
|
841
|
+
/** Options for exposing an `AgentEnvironmentProvider` through the legacy sandbox client port.
|
|
842
|
+
* @experimental */
|
|
843
|
+
interface ProviderAsSandboxClientOptions {
|
|
844
|
+
defaults?: Partial<CreateAgentEnvironmentInput>;
|
|
845
|
+
requireTerminalEvent?: boolean;
|
|
846
|
+
/** Require declared live continuation plus concrete session controls. */
|
|
847
|
+
requireSession?: boolean;
|
|
848
|
+
mapCreateOptions?: (options: CreateSandboxOptions | undefined) => Partial<CreateAgentEnvironmentInput>;
|
|
849
|
+
}
|
|
850
|
+
/** Adapt a neutral environment provider to the `SandboxClient` interface used by existing loop paths.
|
|
851
|
+
* @experimental */
|
|
852
|
+
declare function providerAsSandboxClient(provider: AgentEnvironmentProvider, options?: ProviderAsSandboxClientOptions): SandboxClient;
|
|
853
|
+
/** Options for wrapping the current Tangle sandbox client as an environment provider.
|
|
854
|
+
* @experimental */
|
|
855
|
+
interface SandboxClientProviderOptions {
|
|
856
|
+
name?: string;
|
|
857
|
+
defaultBackend?: BackendType;
|
|
858
|
+
capabilities?: AgentEnvironmentCapabilities | (() => AgentEnvironmentCapabilities | Promise<AgentEnvironmentCapabilities>);
|
|
859
|
+
validateProfile?: (profile: AgentProfileRef) => AgentProfileValidationResult | Promise<AgentProfileValidationResult>;
|
|
860
|
+
/** Resolve a named profile before calling Sandbox, which accepts inline profiles only. */
|
|
861
|
+
resolveProfile?: (profileId: string) => AgentProfile | Promise<AgentProfile>;
|
|
862
|
+
/** Map portable creation into a supported SDK or deployment contract. Runtime attachments
|
|
863
|
+
* require this explicit mapper until the maintained Sandbox SDK transports them. */
|
|
864
|
+
mapCreateInput?: (input: CreateAgentEnvironmentInput) => CreateSandboxOptions;
|
|
865
|
+
/**
|
|
866
|
+
* `idleTimeoutSeconds` sent on every Sandbox create this adapter makes (a mapped create, a
|
|
867
|
+
* `mapCreateInput` result, or a fork), unless those create options already name one. Defaults to
|
|
868
|
+
* {@link DEFAULT_SANDBOX_IDLE_TIMEOUT_SECONDS}; a positive whole number of seconds.
|
|
869
|
+
*/
|
|
870
|
+
idleTimeoutSeconds?: number;
|
|
871
|
+
}
|
|
872
|
+
/**
|
|
873
|
+
* The idle timeout this adapter sends when nothing else names one: 1,800 seconds.
|
|
874
|
+
*
|
|
875
|
+
* Sandbox substitutes no value of its own: an omitted field falls back to the platform's global
|
|
876
|
+
* idle timeout, documented as 30 minutes unless an operator changed it, and Runtime sent none. Idle
|
|
877
|
+
* means inactivity to Sandbox, which suspends the sandbox (the container stops; the workspace is
|
|
878
|
+
* kept) rather than deleting it. The SDK does not define inactivity further. A request in flight
|
|
879
|
+
* counts as activity: a Discovery Lab seat created with `idleTimeoutSeconds: 1800` ran one request
|
|
880
|
+
* to 2,252 seconds, ended by a per-request cap, not by idling (fleet-launch-2026-08-22).
|
|
881
|
+
*
|
|
882
|
+
* The value restates the documented default, so it can tighten a longer operator setting but never
|
|
883
|
+
* loosen the default. It is 3.8 times the longest gap between frames recorded across a healthy
|
|
884
|
+
* fleet run (469 seconds, same report), and a supervised provider child is observed through an open
|
|
885
|
+
* stream for its whole turn. It is a backstop for a process that dies holding an environment; the
|
|
886
|
+
* settlement barrier releases retained environments itself (`Executor.releaseRetained`). It does
|
|
887
|
+
* nothing on a driver with a create/delete-only lifecycle, which the SDK says skips suspension.
|
|
888
|
+
*/
|
|
889
|
+
declare const DEFAULT_SANDBOX_IDLE_TIMEOUT_SECONDS = 1800;
|
|
890
|
+
/**
|
|
891
|
+
* Adapt a `SandboxClient` into the shared `AgentEnvironmentProvider` contract.
|
|
892
|
+
* The provider declares the public SDK contract before it creates an environment.
|
|
893
|
+
* Each environment exposes interactive methods only when its deployment declares every required capability.
|
|
894
|
+
* @experimental */
|
|
895
|
+
declare function sandboxClientAsProvider(client: SandboxClient, options?: SandboxClientProviderOptions): AgentEnvironmentProvider;
|
|
896
|
+
/**
|
|
897
|
+
* What one provider-executed turn settles on: the visible answer plus the event archive the
|
|
898
|
+
* environment streamed. It is the value a `ProviderExecutorOptions.validator` scores.
|
|
899
|
+
*
|
|
900
|
+
* The archive is the streamed sequence, in order, without superseded part updates. A harness
|
|
901
|
+
* streams a text or reasoning part cumulatively: every `message.part.updated` frame restates the
|
|
902
|
+
* part's whole text so far. When a later frame of the same part extends a frame's text, the
|
|
903
|
+
* earlier frame is left out, so each such part is archived once, at its latest frame. Keeping
|
|
904
|
+
* every frame retains frames times text length, and the settled archive is hashed and stored.
|
|
905
|
+
* A frame that does not extend the part's text is kept, and every other event is kept verbatim.
|
|
906
|
+
* Read a part's text from `part.text`; a retained frame's `delta` is only that frame's increment.
|
|
907
|
+
*
|
|
908
|
+
* @experimental
|
|
909
|
+
*/
|
|
910
|
+
interface ProviderLeafOut {
|
|
911
|
+
content: string;
|
|
912
|
+
events: AgentEnvironmentEvent[];
|
|
913
|
+
/** How many streamed part updates the archive left out because a later frame superseded them. */
|
|
914
|
+
supersededPartUpdates?: number;
|
|
915
|
+
/**
|
|
916
|
+
* The child's own harness transcript, read before the environment was destroyed.
|
|
917
|
+
*
|
|
918
|
+
* `events` above is the provider's stream and `trace` on the settlement is the supervisor's
|
|
919
|
+
* tool spans; neither carries the harness's session files, which used to die with the
|
|
920
|
+
* environment. Always present on the settled path: an environment that cannot be read says
|
|
921
|
+
* so with a `reason` rather than being silently absent. #1214.
|
|
922
|
+
*/
|
|
923
|
+
nativeSession?: NativeSessionEvidence;
|
|
924
|
+
}
|
|
925
|
+
/**
|
|
926
|
+
* Per-run Sandbox prompt options for the provider path — the same field, the same name, and the
|
|
927
|
+
* same kernel-owned exclusions as `ExecCtx.promptOptions` on the sandbox path.
|
|
928
|
+
*
|
|
929
|
+
* The kernel owns `sessionId` and `signal`, so neither is declarable: a caller-chosen session id
|
|
930
|
+
* would make every worker share one server session, and the abort channel belongs to the run.
|
|
931
|
+
* `model` is excluded too, and for a different reason: this executor's materialization record
|
|
932
|
+
* names the model from `AgentProfile`, so a turn-level override would make the record state a
|
|
933
|
+
* model the provider did not run. Declare the instrument on `AgentProfile.model`.
|
|
934
|
+
*
|
|
935
|
+
* Everything else is the per-call configuration a portable profile cannot carry. `backend` is the
|
|
936
|
+
* load-bearing one: `backend.model.authMode` plus `authFiles` is how a caller-owned subscription
|
|
937
|
+
* seat reaches the harness inside the environment. Runtime lowers these onto the turn with the one
|
|
938
|
+
* mapper it already uses in the other direction, so a sandbox-shaped provider reads them from
|
|
939
|
+
* `AgentTurnInput.providerOptions.backend` exactly as it reads a sandbox box's prompt options.
|
|
940
|
+
*
|
|
941
|
+
* @experimental
|
|
942
|
+
*/
|
|
943
|
+
type ProviderPromptOptions = Omit<PromptOptions, 'model' | 'sessionId' | 'signal'>;
|
|
944
|
+
/** Options for running a provider as a supervise-mode executor.
|
|
945
|
+
* @experimental */
|
|
946
|
+
interface ProviderExecutorOptions {
|
|
947
|
+
/** Select exactly one caller-declared placement from each child's unchanged profile. */
|
|
948
|
+
placements?: readonly ProviderPlacement[];
|
|
949
|
+
defaults?: Partial<CreateAgentEnvironmentInput>;
|
|
950
|
+
runtime?: Runtime;
|
|
951
|
+
destroyOnSettle?: boolean;
|
|
952
|
+
requireTerminalEvent?: boolean;
|
|
953
|
+
/**
|
|
954
|
+
* Per-run prompt options merged UNDER every streamed turn: a mapped turn's own field wins, and
|
|
955
|
+
* the runtime's abort signal is applied last. `providerOptions` merges one level, so a
|
|
956
|
+
* `taskToTurn` that sets its own provider option cannot silently drop the session credential
|
|
957
|
+
* declared here.
|
|
958
|
+
*/
|
|
959
|
+
promptOptions?: ProviderPromptOptions;
|
|
960
|
+
/**
|
|
961
|
+
* OPT-IN executable score for this worker, with the SAME contract the sandbox seam's validator
|
|
962
|
+
* has: `validate` runs while the environment is still alive, so `ValidationCtx.box` can read
|
|
963
|
+
* files and run commands in the environment it is scoring. Every other supervised hook fires
|
|
964
|
+
* after teardown and can only read the artifact.
|
|
965
|
+
* `ValidationCtx.node` identifies the supervised node, including its recursion depth, so a
|
|
966
|
+
* shared validator can apply a root-only contract without applying it to nested managers.
|
|
967
|
+
*
|
|
968
|
+
* The verdict becomes the settled artifact's verdict. Absent, nothing changes and the leaf falls
|
|
969
|
+
* back to its own settle verdict.
|
|
970
|
+
*/
|
|
971
|
+
validator?: Validator<ProviderLeafOut>;
|
|
972
|
+
/** Transform only the profile sent to `provider.create`. The original profile
|
|
973
|
+
* remains the input to `taskToTurn`, so execution-only normalization cannot
|
|
974
|
+
* rewrite the caller's task mapping. */
|
|
975
|
+
profileForCreate?: (profile: AgentProfile) => AgentProfile;
|
|
976
|
+
/** Map the task while retaining the kernel's canonical prompt mapping by default. */
|
|
977
|
+
taskToTurn?: (task: unknown, specProfile: AgentProfile, defaultTurn: AgentTurnInput) => AgentTurnInput;
|
|
978
|
+
}
|
|
979
|
+
/** Adapt an environment provider into an `ExecutorFactory` for `createExecutor`.
|
|
980
|
+
*
|
|
981
|
+
* `createExecutor({ backend: 'provider', provider })` is the composition most callers want; it
|
|
982
|
+
* builds this factory and injects the seam. See `examples/provider-executor/`.
|
|
983
|
+
*
|
|
984
|
+
* Still `@experimental`: the entry point that consumes it, `createExecutor`, carries no stability
|
|
985
|
+
* tag and is therefore experimental by default, so a stable promise here would be reachable only
|
|
986
|
+
* through an experimental symbol.
|
|
987
|
+
*
|
|
988
|
+
* @experimental */
|
|
989
|
+
declare function providerAsExecutor(provider: AgentEnvironmentProvider, options?: ProviderExecutorOptions): ExecutorFactory<unknown>;
|
|
990
|
+
//#endregion
|
|
991
|
+
//#region src/runtime/sandbox-events.d.ts
|
|
992
|
+
/** The provider/model the platform reports it actually bound to a turn, when it reports one.
|
|
993
|
+
* `source` is the platform's own account of where that choice came from — `environment` means
|
|
994
|
+
* the platform chose, not the request. */
|
|
995
|
+
interface SandboxServedBackend {
|
|
996
|
+
readonly provider?: string;
|
|
997
|
+
readonly model?: string;
|
|
998
|
+
readonly source?: string;
|
|
999
|
+
}
|
|
1000
|
+
/**
|
|
1001
|
+
* Read the served execution identity off one Sandbox event.
|
|
1002
|
+
*
|
|
1003
|
+
* The platform reports `effectiveBackend` on `execution.started` and again on the terminal
|
|
1004
|
+
* event (`@tangle-network/sandbox`, `EffectiveBackend`). Absence returns `undefined`
|
|
1005
|
+
* and must stay unknown — a request is not a receipt, so nothing here may be inferred from
|
|
1006
|
+
* what was asked for.
|
|
1007
|
+
*/
|
|
1008
|
+
declare function sandboxEventServedBackend(event: SandboxEvent): SandboxServedBackend | undefined;
|
|
1009
|
+
/**
|
|
1010
|
+
* Fail the execution when the platform reports serving a model other than the exact one asked for.
|
|
1011
|
+
*
|
|
1012
|
+
* Measured motive (agent-runtime#892, live infrastructure 2026-08-17): 6 of 6 boxes whose profile
|
|
1013
|
+
* declared `zai-coding-plan/glm-5.2` reported
|
|
1014
|
+
* `{"provider":"openai-compat","model":"deepseek/deepseek-v4-flash","source":"environment"}`,
|
|
1015
|
+
* while the materialization receipt recorded the declared model as `status: "known"`. Sending
|
|
1016
|
+
* `backend.model` makes that substitution unlikely; only reading the report back makes it
|
|
1017
|
+
* detectable. A run that cannot say which model produced its evidence must not settle as one
|
|
1018
|
+
* that can.
|
|
1019
|
+
*
|
|
1020
|
+
* Silent when the platform reports no served model: unobserved stays unobserved.
|
|
1021
|
+
*/
|
|
1022
|
+
declare function assertSandboxServedModel(event: SandboxEvent, expected: {
|
|
1023
|
+
readonly provider?: string;
|
|
1024
|
+
readonly model?: string;
|
|
1025
|
+
} | undefined): void;
|
|
1026
|
+
/**
|
|
1027
|
+
* Extract a `RuntimeStreamEvent`-shaped `llm_call` from a sandbox event when
|
|
1028
|
+
* the event carries usage/cost data. Returns `undefined` for non-cost events
|
|
1029
|
+
* so the kernel can iterate the full stream without branching.
|
|
1030
|
+
*
|
|
1031
|
+
* Pure by contract: it never throws on a failed run. The terminal truth
|
|
1032
|
+
* boundary is the public Sandbox outcome tracker, applied after the complete
|
|
1033
|
+
* stream. Post-hoc readers — {@link sumSandboxUsage}, the
|
|
1034
|
+
* analyst trace store, the chat projection — must stay able to read a failed
|
|
1035
|
+
* turn's events, which is when reading them matters most.
|
|
1036
|
+
*
|
|
1037
|
+
* Canonical cost-carrying types observed in the wild:
|
|
1038
|
+
* - `llm_call` — `data: { model, tokensIn, tokensOut, costUsd, ... }`
|
|
1039
|
+
* - `message.completed` / `result` — `data: { usage: { inputTokens,
|
|
1040
|
+
* outputTokens, totalCostUsd? } }`
|
|
1041
|
+
* - `cost.usage` / `usage` — same shape under a dedicated type
|
|
1042
|
+
*
|
|
1043
|
+
* Numeric coercion is strict: `Number.isFinite` gates every accumulator write
|
|
1044
|
+
* so a sentinel `NaN` from a misbehaving backend cannot poison the ledger.
|
|
1045
|
+
*/
|
|
1046
|
+
declare function extractLlmCallEvent(event: SandboxEvent, agentRunName: string): (RuntimeStreamEvent & {
|
|
1047
|
+
type: 'llm_call';
|
|
1048
|
+
}) | undefined;
|
|
1049
|
+
/**
|
|
1050
|
+
* Per-turn usage accounting over BOTH the canonical events and the harness-native ones.
|
|
1051
|
+
*
|
|
1052
|
+
* Some harnesses report a turn's tokens only inside their own event (`harness-usage.ts`), and a
|
|
1053
|
+
* stream may carry that report AND a canonical usage event for the same turn. Crediting both
|
|
1054
|
+
* counts one turn twice, so this ledger holds the precedence rule: a canonical usage event WINS,
|
|
1055
|
+
* and a harness-native report is credited only for a turn in which no canonical usage arrived.
|
|
1056
|
+
*
|
|
1057
|
+
* The harness-native report is held until the turn ends, because it can arrive before the
|
|
1058
|
+
* canonical answer is known — codex emits `turn.completed` ahead of the terminal transport
|
|
1059
|
+
* events. Call {@link SandboxUsageLedger.observe} for every event of a turn, then
|
|
1060
|
+
* {@link SandboxUsageLedger.settleTurn} once at the turn boundary; settling also resets the
|
|
1061
|
+
* ledger for the next turn, so one ledger serves a whole multi-turn session.
|
|
1062
|
+
*
|
|
1063
|
+
* The ledger never throws on a receipt it cannot read: `observe` returns a receipt with
|
|
1064
|
+
* `tokensKnown: false` and `tokensUnknownReason`, so one policy serves every consumer.
|
|
1065
|
+
*/
|
|
1066
|
+
interface SandboxUsageLedger {
|
|
1067
|
+
/** Cumulative worker tokens, including cache classifications reported after the prompt total. */
|
|
1068
|
+
tokenUsage(): LoopTokenUsage;
|
|
1069
|
+
/** Account one event. Returns the canonical usage receipt to credit now, if the event is one. */
|
|
1070
|
+
observe(event: SandboxEvent, agentRunName: string): (RuntimeStreamEvent & {
|
|
1071
|
+
type: 'llm_call';
|
|
1072
|
+
}) | undefined;
|
|
1073
|
+
/** End the turn. Returns the held harness-native receipt when no canonical usage arrived. */
|
|
1074
|
+
settleTurn(agentRunName: string): (RuntimeStreamEvent & {
|
|
1075
|
+
type: 'llm_call';
|
|
1076
|
+
}) | undefined;
|
|
1077
|
+
}
|
|
1078
|
+
/** A {@link SandboxUsageLedger} for one worker. Pass the worker's harness to decode with that
|
|
1079
|
+
* harness's adapter; omit it to try every registered adapter. */
|
|
1080
|
+
declare function createSandboxUsageLedger(harness?: HarnessType): SandboxUsageLedger;
|
|
1081
|
+
/**
|
|
1082
|
+
* Sum the token usage + USD cost of a sandbox turn's events — the one honest way to meter an
|
|
1083
|
+
* `openSandboxRun` cell. Folds a {@link SandboxUsageLedger} over the stream, so it reads usage off
|
|
1084
|
+
* EVERY backend event shape — the canonical events plus a harness that reports usage only in its
|
|
1085
|
+
* own event — and a `runProfileMatrix` dispatch can report it to `ctx.cost`:
|
|
1086
|
+
*
|
|
1087
|
+
* receipt: (turn) => {
|
|
1088
|
+
* const u = sumSandboxUsage(turn.events)
|
|
1089
|
+
* return { model, inputTokens: u.input, outputTokens: u.output,
|
|
1090
|
+
* ...(u.tokensKnown === false ? { usageUnknown: true } : {}),
|
|
1091
|
+
* ...(u.usdKnown !== false && u.costUsd > 0 ? { actualCostUsd: u.costUsd } : {}),
|
|
1092
|
+
* ...(u.usdKnown === false ? { costUnknown: true } : {}),
|
|
1093
|
+
* ...(u.estimatedCostUsd !== undefined ? { estimatedCostUsd: u.estimatedCostUsd } : {}) }
|
|
1094
|
+
* }
|
|
1095
|
+
*
|
|
1096
|
+
* Without this a cell reads `{tokens:0, cost:0}` and the backend-integrity guard correctly aborts the
|
|
1097
|
+
* matrix as a stub. `agentRunName` is the fallback model label for cost-only events (default `'agent'`).
|
|
1098
|
+
*
|
|
1099
|
+
* Pure by contract, like the ledger it folds: it never throws. A harness receipt the ledger cannot
|
|
1100
|
+
* read leaves the result at `tokensKnown: false` with `tokensUnknownReason` carrying the decode
|
|
1101
|
+
* message — an unreadable receipt is a different fact from a turn that reported no usage, and a
|
|
1102
|
+
* post-hoc reader that threw would lose the whole failed turn it exists to report.
|
|
1103
|
+
*/
|
|
1104
|
+
declare function sumSandboxUsage(events: readonly SandboxEvent[], agentRunName?: string): {
|
|
1105
|
+
input: number;
|
|
1106
|
+
output: number;
|
|
1107
|
+
costUsd: number;
|
|
1108
|
+
tokensKnown?: false;
|
|
1109
|
+
usdKnown?: false;
|
|
1110
|
+
estimatedCostUsd?: number;
|
|
1111
|
+
tokensUnknownReason?: string;
|
|
1112
|
+
};
|
|
1113
|
+
/**
|
|
1114
|
+
* Cross-event state for {@link mapSandboxToolEvent}. Sandbox backends emit a
|
|
1115
|
+
* tool invocation as MANY `message.part.updated` frames on the same call id
|
|
1116
|
+
* (pending → running → completed), so faithful projection needs per-call
|
|
1117
|
+
* status memory: one `tool_call` on first sighting, at most one `tool_result`
|
|
1118
|
+
* on the terminal transition, nothing on intermediate re-frames. Create one
|
|
1119
|
+
* state per turn via {@link createSandboxToolPartState}.
|
|
1120
|
+
*
|
|
1121
|
+
* @experimental
|
|
1122
|
+
*/
|
|
1123
|
+
interface SandboxToolPartState {
|
|
1124
|
+
/** Last seen status per tool call id. A terminal status is sticky — later
|
|
1125
|
+
* frames on a settled call project to nothing. */
|
|
1126
|
+
statusByCall: Map<string, string>;
|
|
1127
|
+
/** Sequence for synthesized call ids when an event carries none. */
|
|
1128
|
+
seq: number;
|
|
1129
|
+
}
|
|
1130
|
+
/**
|
|
1131
|
+
* Fresh per-turn {@link SandboxToolPartState} for {@link mapSandboxToolEvent} — an
|
|
1132
|
+
* empty call-status map so each turn projects tool frames independently.
|
|
1133
|
+
*
|
|
1134
|
+
* @experimental
|
|
1135
|
+
*/
|
|
1136
|
+
declare function createSandboxToolPartState(): SandboxToolPartState;
|
|
1137
|
+
/**
|
|
1138
|
+
* Project one `SandboxEvent` onto the `tool_call` / `tool_result` variants of
|
|
1139
|
+
* `RuntimeStreamEvent` — the tool-part projection `mapSandboxEvent`
|
|
1140
|
+
* deliberately does NOT perform. Opt-in and additive: `mapSandboxEvent`'s
|
|
1141
|
+
* default vocabulary (text/reasoning deltas + `llm_call`) is unchanged;
|
|
1142
|
+
* consumers that need the tool surface (chat UIs rendering tool activity)
|
|
1143
|
+
* compose this projector alongside it — `streamAgentTurn` does exactly that
|
|
1144
|
+
* under its `preserveToolParts` option.
|
|
1145
|
+
*
|
|
1146
|
+
* Handled shapes (observed on the opencode / claude-code sandbox backends):
|
|
1147
|
+
* - `message.part.updated` with `part.type === 'tool'` — stateful: a
|
|
1148
|
+
* `tool_call` on the call id's first frame (args from `state.input` or
|
|
1149
|
+
* `state.metadata.input`), a `tool_result` when the status transitions to
|
|
1150
|
+
* `completed` (result from `state.output` / `metadata.output`) or to a
|
|
1151
|
+
* terminal failure (result is `{ error, status, output? }` — the error
|
|
1152
|
+
* surfaced in-band, never dropped).
|
|
1153
|
+
* - bare `tool*` event types (`tool.call`, `tool_result`, …) — stateless:
|
|
1154
|
+
* `*result*` types project to `tool_result`, the rest to `tool_call`.
|
|
1155
|
+
*
|
|
1156
|
+
* Returns `[]` for every non-tool event.
|
|
1157
|
+
*
|
|
1158
|
+
* @experimental
|
|
1159
|
+
*/
|
|
1160
|
+
declare function mapSandboxToolEvent(event: SandboxEvent, state: SandboxToolPartState): (RuntimeStreamEvent & {
|
|
1161
|
+
type: 'tool_call' | 'tool_result';
|
|
1162
|
+
})[];
|
|
1163
|
+
/**
|
|
1164
|
+
* Project one `SandboxEvent` onto the `RuntimeStreamEvent` chat-UX vocabulary,
|
|
1165
|
+
* for runtimes that bridge a sandbox `streamPrompt` into the
|
|
1166
|
+
* `AgentRuntime.act` streaming contract. Returns `undefined` for events that
|
|
1167
|
+
* have no faithful projection — the raw stream is preserved separately for the
|
|
1168
|
+
* `OutputAdapter`, so an unmapped event never loses data.
|
|
1169
|
+
*
|
|
1170
|
+
* Mapped (the task-optional incremental variants — no synthesized task
|
|
1171
|
+
* lifecycle, no guessed tool-part shapes):
|
|
1172
|
+
* - `message.part.updated` text part → `text_delta`
|
|
1173
|
+
* - `message.part.updated` reasoning/thinking part → `reasoning_delta`
|
|
1174
|
+
* - cost-bearing events → `llm_call` (shared with the ledger extractor)
|
|
1175
|
+
*
|
|
1176
|
+
* Tool parts are deliberately NOT mapped here (unchanged default) — compose
|
|
1177
|
+
* {@link mapSandboxToolEvent} alongside when a consumer needs them.
|
|
1178
|
+
*
|
|
1179
|
+
* The opencode backend emits incremental text as
|
|
1180
|
+
* `{ type: 'message.part.updated', data: { part: { type, text }, delta } }`;
|
|
1181
|
+
* `delta` is the increment, `part.text` the running accumulation.
|
|
1182
|
+
*/
|
|
1183
|
+
declare function mapSandboxEvent(event: SandboxEvent, opts?: {
|
|
1184
|
+
agentRunName?: string;
|
|
1185
|
+
}): RuntimeStreamEvent | undefined;
|
|
1186
|
+
/**
|
|
1187
|
+
* Project one `SandboxEvent` onto Runtime's executor progress vocabulary: incremental text and
|
|
1188
|
+
* reasoning, tool calls and results, and an interaction request. It composes the existing
|
|
1189
|
+
* projections ({@link mapSandboxEvent}, {@link mapSandboxToolEvent}, and the canonical Agent
|
|
1190
|
+
* Interface decode) so every sandbox-shaped executor publishes live output through one reader.
|
|
1191
|
+
* Usage-bearing events project to nothing here — accounting stays on the `tokens`/`cost`
|
|
1192
|
+
* channels.
|
|
1193
|
+
*
|
|
1194
|
+
* Pass one {@link SandboxToolPartState} per turn so a multi-frame tool call yields one call and
|
|
1195
|
+
* at most one result.
|
|
1196
|
+
*
|
|
1197
|
+
* @experimental
|
|
1198
|
+
*/
|
|
1199
|
+
declare function sandboxProgressEvents(event: SandboxEvent, state: SandboxToolPartState): ExecutorProgressEvent[];
|
|
1200
|
+
//#endregion
|
|
1201
|
+
//#region src/runtime/sandbox-executor-output.d.ts
|
|
1202
|
+
/**
|
|
1203
|
+
* What a settled turn produced, as an explicit marker.
|
|
1204
|
+
*
|
|
1205
|
+
* `text` carries the byte length of the answer, `empty` says a text-bearing terminal event was
|
|
1206
|
+
* observed and carried nothing, and `absent` says no text-bearing event was observed at all.
|
|
1207
|
+
* The three are distinct on purpose: an empty settle blob used to be indistinguishable from lost
|
|
1208
|
+
* output, so a reader could not tell a box that produced nothing from one whose answer never
|
|
1209
|
+
* arrived.
|
|
1210
|
+
*/
|
|
1211
|
+
type SandboxOutputMarker = {
|
|
1212
|
+
readonly kind: 'text';
|
|
1213
|
+
readonly bytes: number;
|
|
1214
|
+
} | {
|
|
1215
|
+
readonly kind: 'empty';
|
|
1216
|
+
} | {
|
|
1217
|
+
readonly kind: 'absent';
|
|
1218
|
+
};
|
|
1219
|
+
/** Parsed output of one Sandbox executor turn. */
|
|
1220
|
+
interface SandboxLeafOut {
|
|
1221
|
+
events: SandboxEvent[];
|
|
1222
|
+
/** The observed answer. `undefined` when no text-bearing event was observed — never `''`. */
|
|
1223
|
+
content: string | undefined;
|
|
1224
|
+
/** Explicit account of what the turn produced. */
|
|
1225
|
+
output: SandboxOutputMarker;
|
|
1226
|
+
/**
|
|
1227
|
+
* Provider and model the platform reported serving this turn, when it reported one. Absent means
|
|
1228
|
+
* the platform said nothing; it is never filled from the request, because a request is not a
|
|
1229
|
+
* receipt.
|
|
1230
|
+
*/
|
|
1231
|
+
servedBackend?: SandboxServedBackend;
|
|
1232
|
+
toolCalls?: ExecutorToolCall[];
|
|
1233
|
+
outcome?: AgentRunOutcome;
|
|
1234
|
+
}
|
|
1235
|
+
//#endregion
|
|
1236
|
+
//#region src/runtime/harness-usage.d.ts
|
|
1237
|
+
/**
|
|
1238
|
+
* One harness's own token-usage report for one turn, in the runtime's field names.
|
|
1239
|
+
*
|
|
1240
|
+
* `input` is the provider's TOTAL prompt count and `output` is its TOTAL completion count.
|
|
1241
|
+
* The other three counters CLASSIFY a part of one of those totals; none of them adds to it.
|
|
1242
|
+
* `cachedInput` and `cacheWriteInput` classify `input`, which is the convention
|
|
1243
|
+
* `promptCacheTokenClasses` (`util.ts`) folds: `freshInput = input - cacheRead - cacheWrite`.
|
|
1244
|
+
* `reasoningOutput` classifies `output`.
|
|
1245
|
+
*
|
|
1246
|
+
* A counter the harness does not report stays absent, because a zero would claim the harness
|
|
1247
|
+
* measured none.
|
|
1248
|
+
*/
|
|
1249
|
+
interface HarnessUsage {
|
|
1250
|
+
/** The harness family whose adapter produced this report. */
|
|
1251
|
+
readonly harness: HarnessType;
|
|
1252
|
+
/** Total prompt tokens the provider charged for the turn, the cached ones included. */
|
|
1253
|
+
readonly input: number;
|
|
1254
|
+
/** Total completion tokens the provider charged for the turn, the reasoning ones included. */
|
|
1255
|
+
readonly output: number;
|
|
1256
|
+
/** The part of `input` the provider served from its prompt cache. */
|
|
1257
|
+
readonly cachedInput?: number;
|
|
1258
|
+
/** The part of `input` the provider wrote into its prompt cache. */
|
|
1259
|
+
readonly cacheWriteInput?: number;
|
|
1260
|
+
/** The part of `output` the model spent on reasoning. Never added to `output`. */
|
|
1261
|
+
readonly reasoningOutput?: number;
|
|
1262
|
+
}
|
|
1263
|
+
/**
|
|
1264
|
+
* Decode a sandbox event with one harness's adapter, or `undefined` when the event carries no
|
|
1265
|
+
* harness-native usage.
|
|
1266
|
+
*
|
|
1267
|
+
* A NAMED harness reads with that harness's adapter only, and a named harness with no adapter
|
|
1268
|
+
* reports nothing. It never falls through to another harness's adapter: a different harness's
|
|
1269
|
+
* `turn.completed` decoded as codex would either drop the counters codex does not name or fail on
|
|
1270
|
+
* a field codex requires, and both answers would be about the wrong harness. The composite over
|
|
1271
|
+
* every registered adapter runs only when the caller cannot name the harness.
|
|
1272
|
+
*
|
|
1273
|
+
* Throws `ValidationError` when an adapter recognizes the event as its harness's usage carrier and
|
|
1274
|
+
* cannot read the numbers.
|
|
1275
|
+
*/
|
|
1276
|
+
declare function decodeHarnessUsage(event: SandboxEvent, harness?: HarnessType): HarnessUsage | undefined;
|
|
1277
|
+
//#endregion
|
|
1278
|
+
//#region src/runtime/codex-rollout-store.d.ts
|
|
1279
|
+
/** Who wrote one rollout, exactly as its own `session_meta` states it. Nothing here is inferred. */
|
|
1280
|
+
interface CodexRolloutIdentity {
|
|
1281
|
+
/** The rollout's own thread id (`session_meta.payload.id`). */
|
|
1282
|
+
readonly sessionId: string;
|
|
1283
|
+
/** The thread this one was spawned or forked from, when it was. */
|
|
1284
|
+
readonly parentThreadId?: string;
|
|
1285
|
+
/** The thread whose rows are prepended into this file, when this file is a fork. */
|
|
1286
|
+
readonly forkedFromId?: string;
|
|
1287
|
+
/** True when `thread_source` reads `subagent`: a harness-native child, invisible to the journal. */
|
|
1288
|
+
readonly nativeChild: boolean;
|
|
1289
|
+
/** The child's own path in the harness's agent tree (`/root/c1_b_grid`), when it has one. */
|
|
1290
|
+
readonly agentPath?: string;
|
|
1291
|
+
/** The harness's own nickname for the child ("Turing"), when it has one. */
|
|
1292
|
+
readonly agentNickname?: string;
|
|
1293
|
+
/** Spawn depth the harness recorded. `1` is a direct child of the seat. */
|
|
1294
|
+
readonly depth?: number;
|
|
1295
|
+
/** The working directory the session ran in, used to attribute a store to a workspace. */
|
|
1296
|
+
readonly cwd?: string;
|
|
1297
|
+
/** The codex build that wrote it. */
|
|
1298
|
+
readonly cliVersion?: string;
|
|
1299
|
+
/** When the session itself started, from its own `session_meta` timestamp. */
|
|
1300
|
+
readonly startedAtMs?: number;
|
|
1301
|
+
}
|
|
1302
|
+
/** How this reader isolated the session's own rows from the parent rows prepended to its file. */
|
|
1303
|
+
type CodexForkBoundary =
|
|
1304
|
+
/** Not a fork: every row in the file belongs to this session. */
|
|
1305
|
+
{
|
|
1306
|
+
readonly kind: 'whole-file';
|
|
1307
|
+
} |
|
|
1308
|
+
/** A fork whose own first turn was isolated, and by which rule. */
|
|
1309
|
+
{
|
|
1310
|
+
readonly kind: 'resolved';
|
|
1311
|
+
readonly rule: 'history-start-ordinal' | 'turn-is-session' | 'turn-uuid-v7' | 'turn-start-time';
|
|
1312
|
+
/** The `turn_id` of the session's own first turn. */
|
|
1313
|
+
readonly turnId?: string;
|
|
1314
|
+
/** Rows credited to the parent and excluded from `own`. */
|
|
1315
|
+
readonly inheritedTurns: number;
|
|
1316
|
+
} |
|
|
1317
|
+
/** A fork this reader could not isolate. `own` is absent; nothing may be charged. */
|
|
1318
|
+
{
|
|
1319
|
+
readonly kind: 'unresolved';
|
|
1320
|
+
readonly reason: string;
|
|
1321
|
+
};
|
|
1322
|
+
/** One turn of one session, with the counters it added to the session's cumulative total. */
|
|
1323
|
+
interface CodexRolloutTurn {
|
|
1324
|
+
readonly turnId?: string;
|
|
1325
|
+
readonly startedAtMs?: number;
|
|
1326
|
+
readonly usage: HarnessUsage;
|
|
1327
|
+
}
|
|
1328
|
+
/** One rollout file, read. */
|
|
1329
|
+
interface CodexRolloutSession {
|
|
1330
|
+
readonly identity: CodexRolloutIdentity;
|
|
1331
|
+
readonly boundary: CodexForkBoundary;
|
|
1332
|
+
/**
|
|
1333
|
+
* The session's OWN spend — the cumulative delta from its fork boundary to its last report.
|
|
1334
|
+
* ABSENT when the boundary is unresolved: an unattributable number must not be charged.
|
|
1335
|
+
*/
|
|
1336
|
+
readonly own?: HarnessUsage;
|
|
1337
|
+
/** The session's own turns, newest last. Empty when the file reported no usage. */
|
|
1338
|
+
readonly turns: readonly CodexRolloutTurn[];
|
|
1339
|
+
/**
|
|
1340
|
+
* The file's final cumulative `total_token_usage`, kept ONLY as the diagnostic that shows how
|
|
1341
|
+
* far a naive file total is from the truth. Never charge this.
|
|
1342
|
+
*/
|
|
1343
|
+
readonly fileCumulativeInput: number;
|
|
1344
|
+
readonly fileCumulativeOutput: number;
|
|
1345
|
+
}
|
|
1346
|
+
/** What one incremental read of a store observed. */
|
|
1347
|
+
interface CodexStoreDelta {
|
|
1348
|
+
/** Spend by sessions that are NOT native children — the seat's own turns. */
|
|
1349
|
+
readonly seat: HarnessUsage;
|
|
1350
|
+
/** Spend by `thread_source: subagent` sessions — the harness-native children. */
|
|
1351
|
+
readonly native: HarnessUsage;
|
|
1352
|
+
/** Sessions whose fork boundary could not be isolated, so their spend is absent, not zero. */
|
|
1353
|
+
readonly unresolved: ReadonlyArray<{
|
|
1354
|
+
readonly sessionId: string;
|
|
1355
|
+
readonly reason: string;
|
|
1356
|
+
}>;
|
|
1357
|
+
/**
|
|
1358
|
+
* Every session this read touched, for evidence. Each one states its WHOLE own spend and turn
|
|
1359
|
+
* list, which is not the same number as `seat` / `native`: those two carry only what this read
|
|
1360
|
+
* newly observed.
|
|
1361
|
+
*/
|
|
1362
|
+
readonly sessions: readonly CodexRolloutSession[];
|
|
1363
|
+
}
|
|
1364
|
+
/** A store reader that credits each turn once: it tails only the bytes appended since the last read. */
|
|
1365
|
+
interface CodexRolloutStoreReader {
|
|
1366
|
+
/**
|
|
1367
|
+
* Read everything appended since the previous call and attribute it.
|
|
1368
|
+
*
|
|
1369
|
+
* The FIRST call establishes the baseline. Call it before the first turn so pre-existing rows are
|
|
1370
|
+
* consumed and credited to nothing; every later call returns exactly that turn's spend.
|
|
1371
|
+
*/
|
|
1372
|
+
read(): Promise<CodexStoreDelta>;
|
|
1373
|
+
}
|
|
1374
|
+
/** Where a harness keeps its own session store, and which workspace may be credited from it. */
|
|
1375
|
+
interface CodexRolloutStoreRef {
|
|
1376
|
+
/**
|
|
1377
|
+
* Absolute path to the harness home the CLI writes into — `CODEX_HOME`, or `$HOME/.codex`.
|
|
1378
|
+
* This MUST be the run's own isolated store. Pointing it at an ambient host store credits one
|
|
1379
|
+
* run with another run's files, which is the exact defect this reader exists to end.
|
|
1380
|
+
*/
|
|
1381
|
+
readonly root: string;
|
|
1382
|
+
/**
|
|
1383
|
+
* Credit only sessions whose recorded `cwd` is this path or below it. Absent credits every
|
|
1384
|
+
* session under `root`, which is correct only for a store no other run writes to.
|
|
1385
|
+
*/
|
|
1386
|
+
readonly workspaceRoot?: string;
|
|
1387
|
+
}
|
|
1388
|
+
/**
|
|
1389
|
+
* Read one rollout's rows into a session record.
|
|
1390
|
+
*
|
|
1391
|
+
* `rows` is the file's JSON values in file order. Pass the whole file to read a completed session;
|
|
1392
|
+
* the store reader passes appended slices and carries the identity forward itself.
|
|
1393
|
+
*/
|
|
1394
|
+
declare function readCodexRolloutSession(rows: Iterable<unknown>): CodexRolloutSession | undefined;
|
|
1395
|
+
/**
|
|
1396
|
+
* Open an incremental reader over a codex store.
|
|
1397
|
+
*
|
|
1398
|
+
* Nothing is read until `read()` is called, and every read is bounded by the bytes appended since
|
|
1399
|
+
* the previous one, so a 695MB rollout is scanned once rather than once per turn.
|
|
1400
|
+
*/
|
|
1401
|
+
declare function createCodexRolloutStoreReader(ref: CodexRolloutStoreRef): CodexRolloutStoreReader;
|
|
1402
|
+
/** Sum two usage reports on every counter both of them state. */
|
|
1403
|
+
declare function addHarnessUsage(left: HarnessUsage, right: HarnessUsage): HarnessUsage;
|
|
1404
|
+
/** True when a report states any spend at all. */
|
|
1405
|
+
declare function harnessUsageIsEmpty(usage: HarnessUsage): boolean;
|
|
1406
|
+
//#endregion
|
|
1407
|
+
//#region src/runtime/key-provider.d.ts
|
|
1408
|
+
/** Resolve named secrets. The ONE seam every secret store adapts to. */
|
|
1409
|
+
interface KeyProvider {
|
|
1410
|
+
/** The value for `name`, or `undefined` when this provider does not hold it. */
|
|
1411
|
+
get(name: string): Promise<string | undefined>;
|
|
1412
|
+
}
|
|
1413
|
+
/** The env-backed provider: reads the (dotenvx-loaded) process env. Empty /
|
|
1414
|
+
* whitespace-only values count as absent — fail loud, not with a blank key. */
|
|
1415
|
+
declare function envKeyProvider(env?: Record<string, string | undefined>): KeyProvider;
|
|
1416
|
+
/** The `AgentProfileMcpServer.metadata` key the declarative secret-env map
|
|
1417
|
+
* rides under: `{ ENV_VAR_NAME: 'PROVIDER_KEY_NAME' }`. Names only — values
|
|
1418
|
+
* are resolved at materialize time and never stored. */
|
|
1419
|
+
declare const mcpSecretEnvMetadataKey = "secretEnv";
|
|
1420
|
+
/** Read (and validate) a server entry's declared secret-env map, if any.
|
|
1421
|
+
* Malformed metadata throws — a half-declared secret must never half-boot. */
|
|
1422
|
+
declare function secretEnvOfMcpServer(server: AgentProfileMcpServer): Record<string, string> | undefined;
|
|
1423
|
+
/**
|
|
1424
|
+
* Resolve a declared secret-env map into the real env entries for a server
|
|
1425
|
+
* spawn. Fail-closed: no provider or a missing key throws, naming the KEY
|
|
1426
|
+
* NAME only (the value never appears in any message). `label` names the
|
|
1427
|
+
* server for the error (e.g. `profile.mcp['exa']`).
|
|
1428
|
+
*/
|
|
1429
|
+
declare function resolveSecretEnv(secretEnv: Record<string, string>, keys: KeyProvider | undefined, label: string): Promise<Record<string, string>>;
|
|
1430
|
+
/** The spawn-ready strings for one stdio MCP server: profile config values
|
|
1431
|
+
* resolved, secrets separated so the client can redact them. */
|
|
1432
|
+
interface ResolvedMcpServerLaunch {
|
|
1433
|
+
args?: string[];
|
|
1434
|
+
/** Public env, safe to appear in diagnostics. */
|
|
1435
|
+
env?: Record<string, string>;
|
|
1436
|
+
/** Resolved secret env. Reaches only the child process; redacted everywhere else. */
|
|
1437
|
+
protectedEnv?: Record<string, string>;
|
|
1438
|
+
}
|
|
1439
|
+
/**
|
|
1440
|
+
* Resolve a profile MCP server's `args`/`env` config values (interface ≥0.40
|
|
1441
|
+
* `AgentProfileConfigValue`) plus the legacy `metadata.secretEnv` channel into
|
|
1442
|
+
* the plain strings a spawn needs.
|
|
1443
|
+
*
|
|
1444
|
+
* Rules, all fail-closed:
|
|
1445
|
+
* - `args` must be public values. A secret-ref in argv is refused: argv is
|
|
1446
|
+
* readable by every host process (/proc/PID/cmdline) and outside the
|
|
1447
|
+
* protected-value redaction channel, so a secret there cannot be contained.
|
|
1448
|
+
* - `env` secret-refs resolve through the KeyProvider (missing provider or key
|
|
1449
|
+
* throws, naming the KEY NAME only) and land in `protectedEnv`.
|
|
1450
|
+
* - An env var declared secret on BOTH channels (env secret-ref and
|
|
1451
|
+
* metadata.secretEnv) is ambiguous configuration and throws.
|
|
1452
|
+
* - A public `env` entry shadowed by a legacy metadata secret keeps the
|
|
1453
|
+
* pre-0.40 spawn precedence: the secret value wins in the child env.
|
|
1454
|
+
*/
|
|
1455
|
+
declare function resolveMcpServerLaunch(server: AgentProfileMcpServer, keys: KeyProvider | undefined, label: string): Promise<ResolvedMcpServerLaunch>;
|
|
1456
|
+
//#endregion
|
|
1457
|
+
//#region src/runtime/supervise/bridge-config.d.ts
|
|
1458
|
+
/**
|
|
1459
|
+
* cli-bridge seam. A local OpenAI-compatible bridge that fronts harness CLIs
|
|
1460
|
+
* (claude-code / opencode / kimi / pi) behind one HTTP surface. The spawned
|
|
1461
|
+
* `AgentProfile` is the sole harness/provider/model and behavioral authority and
|
|
1462
|
+
* is forwarded verbatim per request; this seam carries transport data only.
|
|
1463
|
+
*
|
|
1464
|
+
* The executor opens a resumable cli-bridge session. `sessionId` identifies the
|
|
1465
|
+
* harness conversation across turns; each turn also receives its own durable run id.
|
|
1466
|
+
* A dropped HTTP reader reattaches to that exact run and explicit cancel is the only
|
|
1467
|
+
* operation allowed to stop it. Omit `sessionId` and the executor mints one per spawn.
|
|
1468
|
+
*
|
|
1469
|
+
* ── HOW TO CONTROL WHAT THE HARNESS LOADS (there is no argv field, by design) ──
|
|
1470
|
+
*
|
|
1471
|
+
* A worker often needs the harness started in a KNOWN state — no ambient extensions, skills,
|
|
1472
|
+
* context files, or prompt templates — because ambient state is how a paired experiment silently
|
|
1473
|
+
* loses its pairing: an installed extension that persists memory across runs carries arm A's state
|
|
1474
|
+
* into arm B, and nothing reports it.
|
|
1475
|
+
*
|
|
1476
|
+
* That is what the spawned `AgentProfile` is FOR. `agent_profile`
|
|
1477
|
+
* rides every request verbatim, and cli-bridge maps it onto each harness's own native controls:
|
|
1478
|
+
*
|
|
1479
|
+
* - Materializing any profile at all already starts the harness isolated from ambient
|
|
1480
|
+
* workspace state — for pi that is `--no-context-files --no-skills --no-prompt-templates`,
|
|
1481
|
+
* applied to every request that carries an `agent_profile`.
|
|
1482
|
+
* - `AgentProfile.extensions.<harness>` is the named, per-harness control channel. An explicit
|
|
1483
|
+
* `extensions: { pi: { load: [] } }` disables ambient extension discovery outright
|
|
1484
|
+
* (pi's `--no-extensions`); listing package names loads exactly those and nothing else.
|
|
1485
|
+
* - `permissions` / `tools` / `mcp` map onto the harness's native tool and server controls.
|
|
1486
|
+
*
|
|
1487
|
+
* A caller therefore does NOT need to hand-roll an `Executor` to isolate a harness run, and the
|
|
1488
|
+
* profile expressing it stays portable: the same declaration means the same thing on a different
|
|
1489
|
+
* harness, whereas an argv string means nothing anywhere else.
|
|
1490
|
+
*
|
|
1491
|
+
* WHY NOT A GENERAL ARGV PASSTHROUGH. `bridgeUrl` addresses a process-spawning server. Forwarding
|
|
1492
|
+
* an arbitrary argv array to it would let any caller holding a bearer token choose the flags of a
|
|
1493
|
+
* process on the bridge host — which for real harness CLIs includes flags that load code from a
|
|
1494
|
+
* path, read a file into the prompt, redirect the working directory, or turn off the isolation the
|
|
1495
|
+
* bridge applies. cli-bridge deliberately confines workers (a filesystem jail and deny-by-default
|
|
1496
|
+
* network egress), and every one of those confinements is expressed as spawn configuration, so an
|
|
1497
|
+
* argv channel is a channel for unwinding them. It would also break this executor's own contract:
|
|
1498
|
+
* the durable-run replay protocol, session pinning, and streaming mode are all argv the bridge
|
|
1499
|
+
* owns, and a caller-supplied duplicate silently wins or corrupts the parse. The structured profile
|
|
1500
|
+
* channel is validated, per-harness, portable, and refuses controls it does not understand — keep
|
|
1501
|
+
* new harness capability there.
|
|
1502
|
+
*/
|
|
1503
|
+
interface BridgeSeam {
|
|
1504
|
+
bridgeUrl: string;
|
|
1505
|
+
bridgeBearer: string;
|
|
1506
|
+
/**
|
|
1507
|
+
* Optional request-scoped model credential.
|
|
1508
|
+
*
|
|
1509
|
+
* The key name is portable configuration. The provider is a live service and is intentionally
|
|
1510
|
+
* not serialised. Runtime resolves both values immediately before every bridge POST and sends
|
|
1511
|
+
* them only to a loopback bridge through private request headers.
|
|
1512
|
+
*/
|
|
1513
|
+
modelCredential?: BridgeModelCredential;
|
|
1514
|
+
/** Optional working directory forwarded to cli-bridge and persisted with the session. */
|
|
1515
|
+
cwd?: string;
|
|
1516
|
+
/**
|
|
1517
|
+
* The harness's OWN on-disk session store, read as a spend receipt.
|
|
1518
|
+
*
|
|
1519
|
+
* cli-bridge forwards no token usage for a codex worker, so a turn whose provider counters exist
|
|
1520
|
+
* only in codex's rollout meters `{0, 0}` with `tokensKnown: false`. Measured on one live seat
|
|
1521
|
+
* (discovery#80): 9 of 9 `metered` events read zero while 27,320,482 codex tokens sat in the same
|
|
1522
|
+
* run directory, 1,453,948 of them belonging to harness-native children the journal never saw.
|
|
1523
|
+
*
|
|
1524
|
+
* Naming the store here turns those rows into evidence. The executor tails it once per turn and
|
|
1525
|
+
* credits the DELTA, so each turn is charged once, and it reports the counters with
|
|
1526
|
+
* `provenance: 'harness-store'` so a reader can tell a disk receipt from a stream receipt.
|
|
1527
|
+
*
|
|
1528
|
+
* The path must be the run's OWN isolated store. An ambient host store credits this run with
|
|
1529
|
+
* another run's files, and `workspaceRoot` is the structural guard against it.
|
|
1530
|
+
*/
|
|
1531
|
+
harnessStore?: BridgeHarnessStore;
|
|
1532
|
+
/** Caller-owned deadline for each bridge turn. Runtime enforces it locally and sends the
|
|
1533
|
+
* same value in `execution.timeoutMs` so the bridge-owned process follows the same policy. */
|
|
1534
|
+
timeoutMs?: number;
|
|
1535
|
+
/** Stable, caller-owned cli-bridge session id for harness-side resume. Defaults
|
|
1536
|
+
* to a freshly minted per-spawn id so each worker is its own resumable session. */
|
|
1537
|
+
sessionId?: string;
|
|
1538
|
+
/** Transport reconnects allowed after the first POST. Default 3; set 0 to disable. */
|
|
1539
|
+
maxReconnects?: number;
|
|
1540
|
+
/** Newest-last activity window `progress()` reports. Default 12. */
|
|
1541
|
+
activityWindow?: number;
|
|
1542
|
+
}
|
|
1543
|
+
/**
|
|
1544
|
+
* A harness's own session store on the bridge host, named so the runtime may read it.
|
|
1545
|
+
*
|
|
1546
|
+
* Only `codex` has a reader today. Any other harness is REFUSED rather than read with codex's
|
|
1547
|
+
* decoder: a different harness's file decoded as a codex rollout would either drop counters it does
|
|
1548
|
+
* not name or credit a number that is about the wrong wire shape.
|
|
1549
|
+
*/
|
|
1550
|
+
interface BridgeHarnessStore extends CodexRolloutStoreRef {
|
|
1551
|
+
/** The harness family that wrote the store. */
|
|
1552
|
+
readonly harness: HarnessType;
|
|
1553
|
+
}
|
|
1554
|
+
/** A live, request-scoped model credential reference for a local cli-bridge. */
|
|
1555
|
+
interface BridgeModelCredential {
|
|
1556
|
+
/** Provider key name for the scoped model token. */
|
|
1557
|
+
key: string;
|
|
1558
|
+
/** Provider key name for the exact scoped HTTPS model gateway URL. */
|
|
1559
|
+
baseUrlKey: string;
|
|
1560
|
+
/** Live credential service. Runtime retains this reference through reusable captures. */
|
|
1561
|
+
provider: KeyProvider;
|
|
1562
|
+
}
|
|
1563
|
+
//#endregion
|
|
1564
|
+
//#region src/runtime/supervise/inbox.d.ts
|
|
1565
|
+
/** A message from the run's AUTHORITY — the parent driver. These two kinds carry instruction. */
|
|
1566
|
+
interface AuthorityInboxMessage {
|
|
1567
|
+
readonly kind: 'steer' | 'answer';
|
|
1568
|
+
readonly text: string;
|
|
1569
|
+
/** Forceful messages abort the in-flight turn; queued ones wait for the boundary flush. */
|
|
1570
|
+
readonly interrupt: boolean;
|
|
1571
|
+
/** Present for an `answer` — the question id it resolves. */
|
|
1572
|
+
readonly questionId?: string;
|
|
1573
|
+
}
|
|
1574
|
+
/** A message from a SIBLING worker. Information, never instruction — the parent stays the only
|
|
1575
|
+
* authority over this worker's task. */
|
|
1576
|
+
interface PeerInboxMessage {
|
|
1577
|
+
readonly kind: 'mail';
|
|
1578
|
+
readonly text: string;
|
|
1579
|
+
/** Always false. Peer mail is queued by construction; see this file's header. */
|
|
1580
|
+
readonly interrupt: false;
|
|
1581
|
+
readonly envelope: PeerMailEnvelope;
|
|
1582
|
+
}
|
|
1583
|
+
type InboxMessage = AuthorityInboxMessage | PeerInboxMessage;
|
|
1584
|
+
interface Inbox {
|
|
1585
|
+
/** The `Executor.deliver` implementation. Returns false when the raw message is malformed and
|
|
1586
|
+
* therefore was not queued; callers must not acknowledge a message this inbox discarded. */
|
|
1587
|
+
deliver(msg: unknown): boolean;
|
|
1588
|
+
/** Remove and return all pending messages (the flush). */
|
|
1589
|
+
drain(): InboxMessage[];
|
|
1590
|
+
pending(): number;
|
|
1591
|
+
/** Pending messages from the run's AUTHORITY only. This is what the pre-settle fence counts:
|
|
1592
|
+
* a worker may not finish while a steer or answer it never read is queued, but peer mail must
|
|
1593
|
+
* never be able to hold a finished worker open. */
|
|
1594
|
+
pendingAuthority(): number;
|
|
1595
|
+
/** Open a fresh per-turn interrupt signal; a later forceful `deliver` aborts it. The loop links
|
|
1596
|
+
* this into the signal it passes to its inference call, then re-plans when it fires. */
|
|
1597
|
+
freshInterrupt(): AbortSignal;
|
|
1598
|
+
/** Render drained messages as ONE operator turn to fold into the worker's conversation. */
|
|
1599
|
+
fold(messages: ReadonlyArray<InboxMessage>): string;
|
|
1600
|
+
}
|
|
1601
|
+
/** Create the worker-side inbox for the down-leg: the driver's `steer_agent` / `answer_question` messages and a sibling's peer mail queue here, and the worker's loop drains them at step boundaries and before settle. */
|
|
1602
|
+
declare function createInbox(): Inbox;
|
|
1603
|
+
//#endregion
|
|
1604
|
+
//#region src/runtime/supervise/sandbox-session.d.ts
|
|
1605
|
+
/** Ceiling on continuation turns. Turn 0 is the task; every later turn is a folded steer, so
|
|
1606
|
+
* this bounds how many times a supervisor may redirect ONE worker before it must respawn. */
|
|
1607
|
+
declare const DEFAULT_SANDBOX_STEERING_MAX_TURNS = 24;
|
|
1608
|
+
/** Opt-in configuration for the steerable sandbox worker (`SandboxSeam.steering`). Absent, the
|
|
1609
|
+
* sandbox executor keeps its historical single-shot `runAgentRounds` composition verbatim. */
|
|
1610
|
+
interface SandboxSteeringOptions {
|
|
1611
|
+
/** Max turns for one worker (turn 0 + folded steers). Default {@link DEFAULT_SANDBOX_STEERING_MAX_TURNS}. */
|
|
1612
|
+
readonly maxTurns?: number;
|
|
1613
|
+
/** How many recent tool/turn notes `progress()` reports. Default 12. */
|
|
1614
|
+
readonly activityWindow?: number;
|
|
1615
|
+
/** Per-turn wall-clock ceiling; the turn's stream is aborted when it elapses. */
|
|
1616
|
+
readonly turnTimeoutMs?: number;
|
|
1617
|
+
}
|
|
1618
|
+
/** What the steerable session exposes to its executor: the usage stream plus the live reads. */
|
|
1619
|
+
interface SteerableSandboxSession {
|
|
1620
|
+
/** Drive the worker to settlement. `signal` is the spawn-scoped abort handed to `execute`. */
|
|
1621
|
+
stream(task: unknown, signal: AbortSignal): AsyncIterable<UsageEvent>;
|
|
1622
|
+
progress(): ExecutorProgress;
|
|
1623
|
+
/** Ask the box to stop the running execution on this exact session and report what it answered. */
|
|
1624
|
+
cancel(request: ExecutorCancellationRequest): Promise<ExecutorCancellation>;
|
|
1625
|
+
traceSource(): TraceSource;
|
|
1626
|
+
artifact(): {
|
|
1627
|
+
outRef: string;
|
|
1628
|
+
out: unknown;
|
|
1629
|
+
verdict?: DefaultVerdict;
|
|
1630
|
+
spent: Spend;
|
|
1631
|
+
} | undefined;
|
|
1632
|
+
teardown(): Promise<void>;
|
|
1633
|
+
}
|
|
1634
|
+
interface SteerableSandboxArgs {
|
|
1635
|
+
readonly controller: AbortController;
|
|
1636
|
+
readonly profile: AgentProfile;
|
|
1637
|
+
readonly harness: BackendType;
|
|
1638
|
+
readonly sandboxClient: SandboxClient;
|
|
1639
|
+
readonly inbox: Inbox;
|
|
1640
|
+
readonly taskToPrompt: (task: unknown) => string;
|
|
1641
|
+
readonly options?: SandboxSteeringOptions;
|
|
1642
|
+
readonly loopCtx?: Partial<Omit<ExecCtx, 'sandboxClient' | 'signal'>>;
|
|
1643
|
+
/**
|
|
1644
|
+
* Inherited `TRACE_ID` / `PARENT_SPAN_ID` for the box, merged into `CreateSandboxOptions.env` so
|
|
1645
|
+
* the remote worker's own spans join the supervisor's trace under the spawning node's span.
|
|
1646
|
+
* Absent when the run records no spans — the create options are then untouched.
|
|
1647
|
+
*/
|
|
1648
|
+
readonly traceEnv?: Record<string, string>;
|
|
1649
|
+
readonly contentRef: (prefix: string, value: unknown) => string;
|
|
1650
|
+
readonly now?: () => number;
|
|
1651
|
+
}
|
|
1652
|
+
/** One steerable sandbox worker. The returned session is inert until `stream()` is drained. */
|
|
1653
|
+
declare function createSteerableSandboxSession(args: SteerableSandboxArgs): SteerableSandboxSession;
|
|
1654
|
+
//#endregion
|
|
1655
|
+
//#region src/runtime/supervise/runtime.d.ts
|
|
1656
|
+
/**
|
|
1657
|
+
* Router/inline transport seam. The profile owns model, prompt, and generation behavior.
|
|
1658
|
+
*/
|
|
1659
|
+
interface RouterSeam {
|
|
1660
|
+
routerBaseUrl: string;
|
|
1661
|
+
routerKey: string;
|
|
1662
|
+
/** Injectable transport for offline/local execution; still passes through Runtime metering. */
|
|
1663
|
+
complete?: RouterConfig['complete'];
|
|
1664
|
+
/** When present, return one turn's requested tool calls without executing them. */
|
|
1665
|
+
tools?: ReadonlyArray<ToolSpec>;
|
|
1666
|
+
}
|
|
1667
|
+
/**
|
|
1668
|
+
* Sandbox executor seam. The `sandboxClient` the composed `runAgentRounds` creates
|
|
1669
|
+
* boxes through, plus the optional trace/run/lineage wiring forwarded into the
|
|
1670
|
+
* loop. `lineage` is opaque here (PR #150's `RunAgentRoundsOptions.lineage`): forwarded
|
|
1671
|
+
* forward-compatibly, never inspected — this executor does NOT reinvent
|
|
1672
|
+
* checkpoint/fork.
|
|
1673
|
+
*/
|
|
1674
|
+
interface SandboxSeam {
|
|
1675
|
+
sandboxClient: SandboxClient;
|
|
1676
|
+
/** Forwarded into the composed `runAgentRounds`'s `ctx` (trace emitter, run handle, etc.). */
|
|
1677
|
+
loopCtx?: Partial<Omit<ExecCtx, 'sandboxClient' | 'signal'>>;
|
|
1678
|
+
/** PR #150 `RunAgentRoundsOptions.lineage` passthrough — opaque; forwarded, not parsed. */
|
|
1679
|
+
lineage?: unknown;
|
|
1680
|
+
/** Hard cap on the composed loop's iterations. The budget pool reserves against
|
|
1681
|
+
* the spawn `Budget.maxIterations`; this is the leaf's own ceiling. Default 1. */
|
|
1682
|
+
maxIterations?: number;
|
|
1683
|
+
/**
|
|
1684
|
+
* OPT-IN executable score for this worker. Forwarded to the composed
|
|
1685
|
+
* `runAgentRounds` as its `validator`, so the kernel calls `validate` while the
|
|
1686
|
+
* iteration's box is still alive: `ValidationCtx.box` is a LIVE `SandboxInstance`
|
|
1687
|
+
* and the check can run commands or read files in the container it is scoring.
|
|
1688
|
+
* Every other supervised hook fires after teardown and can only read the artifact.
|
|
1689
|
+
*
|
|
1690
|
+
* The resulting verdict becomes the winner's verdict, which this executor already
|
|
1691
|
+
* surfaces on its `ExecutorResult`. Absent, nothing changes: the loop runs
|
|
1692
|
+
* unscored and the leaf falls back to its own settle verdict.
|
|
1693
|
+
*
|
|
1694
|
+
* Not representable with `steering` — a steerable session is a multi-turn session
|
|
1695
|
+
* on one box, not a `runAgentRounds` composition, so the pair is rejected instead
|
|
1696
|
+
* of silently dropping the score.
|
|
1697
|
+
*/
|
|
1698
|
+
validator?: Validator<SandboxLeafOut>;
|
|
1699
|
+
/**
|
|
1700
|
+
* OPT-IN: run this worker as a multi-turn, STEERABLE session instead of the historical
|
|
1701
|
+
* single-shot `runAgentRounds` composition. Setting it gives the sandbox worker an `Executor.deliver`
|
|
1702
|
+
* inbox (so `Scope.send` / `steer_agent` actually reach it), a live tool-activity trace, and a
|
|
1703
|
+
* `progress()` read — turning the default cloud worker from something a supervisor can only
|
|
1704
|
+
* wait on into something it can watch and correct.
|
|
1705
|
+
*
|
|
1706
|
+
* Absent, nothing changes: the same `runAgentRounds` leaf, no inbox, `steer_agent` still reports
|
|
1707
|
+
* `delivered:false`. Opt-in because a steerable worker holds ONE box across several turns,
|
|
1708
|
+
* which is a different resource profile from a fire-and-forget shot.
|
|
1709
|
+
*/
|
|
1710
|
+
steering?: SandboxSteeringOptions;
|
|
1711
|
+
}
|
|
1712
|
+
/**
|
|
1713
|
+
* UNMETERED CLI subprocess seam. `bin` + `args` describe the process to spawn.
|
|
1714
|
+
*
|
|
1715
|
+
* READ THIS BEFORE CHOOSING `backend: 'cli'`. This backend pipes a prompt to a subprocess's stdin
|
|
1716
|
+
* and reads its stdout. It has no usage receipt of any kind, so it reports its spend with
|
|
1717
|
+
* `Spend.tokensKnown: false`: the work is recorded, its `{0,0}` tokens and `$0` are a FLOOR rather
|
|
1718
|
+
* than a measurement, and a ceiling priced from either is a ceiling that cannot fire. The executor
|
|
1719
|
+
* is also `budgetExempt: true`, which is why `driveHarnessFromBackend` refuses it outright rather
|
|
1720
|
+
* than pretending to budget it.
|
|
1721
|
+
*
|
|
1722
|
+
* If you need a metered harness worker, use `backend: 'bridge'` (a cli-bridge session, which
|
|
1723
|
+
* reports the harness's real per-turn tokens and cost) or `backend: 'cli-worktree'` with
|
|
1724
|
+
* `codexReproducible`. Reach for this seam only when the subprocess genuinely is not an inference
|
|
1725
|
+
* agent, or when you have accepted that its cost is invisible.
|
|
1726
|
+
*
|
|
1727
|
+
* `args` is argv for a LOCAL, in-process spawn under this process's own privileges. It is not a
|
|
1728
|
+
* remote channel and nothing forwards it over a wire.
|
|
1729
|
+
*/
|
|
1730
|
+
interface CliSeam {
|
|
1731
|
+
bin: string;
|
|
1732
|
+
args?: string[];
|
|
1733
|
+
/** Extra environment for the subprocess (merged over `process.env`). */
|
|
1734
|
+
env?: Record<string, string>;
|
|
1735
|
+
/** Working directory for the subprocess. */
|
|
1736
|
+
cwd?: string;
|
|
1737
|
+
}
|
|
1738
|
+
/**
|
|
1739
|
+
* cli-worktree seam. A supervisor-authored `AgentProfile` driving a local coding-harness CLI
|
|
1740
|
+
* (claude / codex / opencode) on its own git worktree — the leaf `createWorktreeCliExecutor`
|
|
1741
|
+
* named as data. `repoRoot` is transport data; `AgentProfile.harness` selects the CLI.
|
|
1742
|
+
* `taskPrompt` remains an optional direct-call fallback for callers that execute with `undefined`.
|
|
1743
|
+
* The authored
|
|
1744
|
+
* `profile.prompt.systemPrompt` + `profile.model.default` reach the harness via the §1.5
|
|
1745
|
+
* `harnessInvocation` mapper. Everything else mirrors `WorktreeCliExecutorOptions`.
|
|
1746
|
+
*/
|
|
1747
|
+
interface CliWorktreeSeam {
|
|
1748
|
+
repoRoot: string;
|
|
1749
|
+
taskPrompt?: string;
|
|
1750
|
+
runId?: string;
|
|
1751
|
+
baseRef?: string;
|
|
1752
|
+
harnessTimeoutMs?: number;
|
|
1753
|
+
/** Isolated, network-off Codex execution with terminal JSONL usage capture. */
|
|
1754
|
+
codexReproducible?: boolean;
|
|
1755
|
+
/** Absolute host paths denied to reproducible Codex. */
|
|
1756
|
+
codexReadDeniedPaths?: ReadonlyArray<string>;
|
|
1757
|
+
testCmd?: string;
|
|
1758
|
+
typecheckCmd?: string;
|
|
1759
|
+
checkTimeoutMs?: number;
|
|
1760
|
+
checkOutputCap?: number;
|
|
1761
|
+
budgetExempt?: boolean;
|
|
1762
|
+
/** Live cli-bridge transport inside the worktree. When set, the worktree leaf accepts
|
|
1763
|
+
* `deliver()` messages and resumes the same bridge session in this worktree cwd. */
|
|
1764
|
+
bridge?: CliWorktreeBridgeSeam;
|
|
1765
|
+
/** Test seam — forwarded to worktree helpers. */
|
|
1766
|
+
runGit?: GitRunner;
|
|
1767
|
+
/** Test seam — forwarded to verification checks. */
|
|
1768
|
+
runCommand?: WorktreeCheckRunner;
|
|
1769
|
+
}
|
|
1770
|
+
/**
|
|
1771
|
+
* cli-in-place seam. A supervisor-authored `AgentProfile` driving a local coding-harness CLI
|
|
1772
|
+
* (claude-code / codex / opencode / pi) on a workspace the CALLER supplies — the leaf
|
|
1773
|
+
* `createInPlaceCliExecutor` named as data. `workspacePath` is transport data;
|
|
1774
|
+
* `AgentProfile.harness` selects the CLI, and the authored `profile.prompt.systemPrompt` +
|
|
1775
|
+
* `profile.model.default` reach the harness via the §1.5 `harnessInvocation` mapper.
|
|
1776
|
+
*
|
|
1777
|
+
* READ THIS BEFORE CHOOSING BETWEEN THIS AND `cli-worktree`. They differ in ONE thing, and it is
|
|
1778
|
+
* the thing that decides which one a caller wants:
|
|
1779
|
+
*
|
|
1780
|
+
* - `cli-worktree` cuts a git worktree of its OWN off `repoRoot`, runs the harness there,
|
|
1781
|
+
* returns the captured patch, and removes the worktree at teardown. The directory it was
|
|
1782
|
+
* given is never edited. That is correct for a fanout of N candidate authors that must not
|
|
1783
|
+
* clobber each other, and for a caller whose deliverable IS the patch.
|
|
1784
|
+
* - `cli-in-place` runs the harness in `workspacePath` itself. The edits stay in that directory
|
|
1785
|
+
* after the call, so the NEXT call sees them. That is what a caller needs when the workspace
|
|
1786
|
+
* has to persist between calls — a multi-shot author resuming on top of its own edits
|
|
1787
|
+
* (`agenticGenerator`), or a candidate directory the caller commits itself.
|
|
1788
|
+
*
|
|
1789
|
+
* Because the workspace persists, so would the profile inputs this path materializes into it. They
|
|
1790
|
+
* are removed before the call returns, so the directory a caller inspects afterwards holds the
|
|
1791
|
+
* harness's own edits and nothing else, and a `git status` over it answers "did the author change
|
|
1792
|
+
* anything" rather than "did Runtime write a settings file".
|
|
1793
|
+
*
|
|
1794
|
+
* There is no reproducible-Codex mode here: that mode stages an executable and a write probe INTO
|
|
1795
|
+
* its working directory, which a caller-owned workspace is not the place for. Use `cli-worktree`
|
|
1796
|
+
* with `codexReproducible` when the isolated, metered Codex run is what you want.
|
|
1797
|
+
*/
|
|
1798
|
+
interface CliInPlaceSeam {
|
|
1799
|
+
/** Absolute path to the EXISTING directory the harness edits. Runtime never creates, cleans, or
|
|
1800
|
+
* removes it. */
|
|
1801
|
+
workspacePath: string;
|
|
1802
|
+
taskPrompt?: string;
|
|
1803
|
+
harnessTimeoutMs?: number;
|
|
1804
|
+
/** Test seam — inject the harness runner so unit tests script a `LocalHarnessResult`. */
|
|
1805
|
+
runHarness?: typeof runLocalHarness;
|
|
1806
|
+
}
|
|
1807
|
+
interface CliWorktreeBridgeSeam {
|
|
1808
|
+
bridgeUrl: string;
|
|
1809
|
+
bridgeBearer: string;
|
|
1810
|
+
/** Caller-owned deadline for each bridge turn. Runtime enforces it locally and sends the
|
|
1811
|
+
* same value in `execution.timeoutMs` so cli-bridge cannot substitute its own cutoff. */
|
|
1812
|
+
timeoutMs?: number;
|
|
1813
|
+
/** Stable cli-bridge session id. Defaults to `bridge-worktree-${runId}`. */
|
|
1814
|
+
sessionId?: string;
|
|
1815
|
+
/** Transport reconnects allowed after the first POST. Default 3; set 0 to disable. */
|
|
1816
|
+
maxReconnects?: number;
|
|
1817
|
+
}
|
|
1818
|
+
/**
|
|
1819
|
+
* Generic environment provider executor config. External packages implement
|
|
1820
|
+
* `AgentEnvironmentProvider`; this built-in wrapper lets `createExecutor` consume them as backend
|
|
1821
|
+
* data while preserving the existing usage channel. Runtime depends on no provider package, so a
|
|
1822
|
+
* Tangle provider and a hand-written one compose identically. Worked wiring:
|
|
1823
|
+
* `examples/provider-executor/`.
|
|
1824
|
+
*
|
|
1825
|
+
* Everything a create needs travels on `CreateAgentEnvironmentInput` through
|
|
1826
|
+
* {@link ProviderExecutorOptions.defaults}; everything one turn needs travels on
|
|
1827
|
+
* {@link ProviderExecutorOptions.promptOptions}. Wrapping the provider's own client to reach a
|
|
1828
|
+
* field is what this seam exists to replace: the wrapper is invisible to Runtime, so its options
|
|
1829
|
+
* are absent from every record the run produces.
|
|
1830
|
+
*
|
|
1831
|
+
* READINESS IS THE PROVIDER'S CONTRACT. `provider.create` resolves with an environment that can
|
|
1832
|
+
* take a turn, so this seam streams straight into it and adds no readiness wait of its own. The
|
|
1833
|
+
* sandbox seam's `acquireSandbox` exists because a raw `SandboxClient.create` returns before the
|
|
1834
|
+
* box is ready; a second poll here would hide a provider that does not honor the contract, and
|
|
1835
|
+
* that provider is an upstream defect to report rather than a race to paper over.
|
|
1836
|
+
*/
|
|
1837
|
+
interface ProviderSeam extends ProviderExecutorOptions {
|
|
1838
|
+
provider: AgentEnvironmentProvider$1 | string;
|
|
1839
|
+
registry?: AgentEnvironmentProviderRegistry;
|
|
1840
|
+
/**
|
|
1841
|
+
* Compose the provider through the existing steerable sandbox session.
|
|
1842
|
+
* The exact profile must name its harness, and the provider must expose live
|
|
1843
|
+
* continuation plus session controls. The provider still owns environment
|
|
1844
|
+
* creation and session semantics.
|
|
1845
|
+
*/
|
|
1846
|
+
steering?: SandboxSteeringOptions;
|
|
1847
|
+
}
|
|
1848
|
+
/**
|
|
1849
|
+
* Router seam WITH tool use — the tool-using router backend. Same direct
|
|
1850
|
+
* OpenAI-compatible endpoint as `RouterSeam`, but each turn passes `tools`; when
|
|
1851
|
+
* the model emits tool_calls they run via `executeToolCall` ON THIS HOST and the
|
|
1852
|
+
* results fold back as `tool` messages, repeating until the model answers without
|
|
1853
|
+
* a tool or `maxTurns` is hit. A real agentic loop, OFF-BOX — no sandbox, so it
|
|
1854
|
+
* is unaffected by a box's egress allowlist. One turn = one completion = the
|
|
1855
|
+
* equal-compute unit. `executeToolCall` receives the task so per-task tool
|
|
1856
|
+
* surfaces (e.g. a gym keyed by task) can dispatch correctly.
|
|
1857
|
+
*/
|
|
1858
|
+
interface RouterToolsSeam {
|
|
1859
|
+
routerBaseUrl: string;
|
|
1860
|
+
routerKey: string;
|
|
1861
|
+
complete?: RouterConfig['complete'];
|
|
1862
|
+
tools: ReadonlyArray<ToolSpec>;
|
|
1863
|
+
executeToolCall: (name: string, args: Record<string, unknown>, task: unknown) => Promise<string>;
|
|
1864
|
+
/** Exact conversation to continue. Runtime validates its system message against the profile. */
|
|
1865
|
+
initialMessages?: ReadonlyArray<Readonly<Record<string, unknown>>>;
|
|
1866
|
+
/** Observe the detached final conversation for session persistence. */
|
|
1867
|
+
onMessages?: (messages: ReadonlyArray<Readonly<Record<string, unknown>>>) => void | Promise<void>;
|
|
1868
|
+
/** Online observer of each tool step — the seam a `DetectorMonitor` taps to watch the live pipe
|
|
1869
|
+
* (raise a `finding` when the worker loops/errors). Called after every tool call resolves, with
|
|
1870
|
+
* real per-call wall-clock (`startedAt`/`endedAt`/`durationMs`) so a push `TraceSource` can carry
|
|
1871
|
+
* non-zero span durations onto the unified timeline. */
|
|
1872
|
+
onToolStep?: (step: {
|
|
1873
|
+
toolName: string;
|
|
1874
|
+
args: Record<string, unknown>;
|
|
1875
|
+
status: 'ok' | 'error';
|
|
1876
|
+
startedAt?: number;
|
|
1877
|
+
endedAt?: number;
|
|
1878
|
+
durationMs?: number;
|
|
1879
|
+
}) => void;
|
|
1880
|
+
}
|
|
1881
|
+
/**
|
|
1882
|
+
* The leaf `createWorktreeCliExecutor` as a backend-as-data factory: a supervisor-authored
|
|
1883
|
+
* `AgentProfile` driving claude / codex / opencode on its own worktree. `budgetExempt` like
|
|
1884
|
+
* the other CLI leaves; the authored systemPrompt + model reach the harness via §1.5.
|
|
1885
|
+
*/
|
|
1886
|
+
declare const cliWorktreeExecutor: ExecutorFactory<unknown>;
|
|
1887
|
+
/**
|
|
1888
|
+
* The leaf `createInPlaceCliExecutor` as a backend-as-data factory: a supervisor-authored
|
|
1889
|
+
* `AgentProfile` driving a local coding CLI in the workspace the caller supplied, so its edits are
|
|
1890
|
+
* still there for the next spawn. `budgetExempt` like the other CLI leaves; the authored
|
|
1891
|
+
* systemPrompt + model reach the harness via §1.5.
|
|
1892
|
+
*/
|
|
1893
|
+
declare const cliInPlaceExecutor: ExecutorFactory<unknown>;
|
|
1894
|
+
/**
|
|
1895
|
+
* Config for {@link createExecutor}: the backend is DATA — the cost dial a profile,
|
|
1896
|
+
* an experiment config, or a replay journal can name — not an import choice. Each
|
|
1897
|
+
* variant carries its backend's seam.
|
|
1898
|
+
*/
|
|
1899
|
+
type ExecutorConfig = ({
|
|
1900
|
+
backend: 'router';
|
|
1901
|
+
} & RouterSeam) | ({
|
|
1902
|
+
backend: 'router-tools';
|
|
1903
|
+
} & RouterToolsSeam) | ({
|
|
1904
|
+
backend: 'bridge';
|
|
1905
|
+
} & BridgeSeam) | ({
|
|
1906
|
+
backend: 'cli';
|
|
1907
|
+
} & CliSeam) | ({
|
|
1908
|
+
backend: 'cli-worktree';
|
|
1909
|
+
} & CliWorktreeSeam) | ({
|
|
1910
|
+
backend: 'cli-in-place';
|
|
1911
|
+
} & CliInPlaceSeam) | ({
|
|
1912
|
+
backend: 'provider';
|
|
1913
|
+
} & ProviderSeam) | ({
|
|
1914
|
+
backend: 'sandbox';
|
|
1915
|
+
} & SandboxSeam);
|
|
1916
|
+
/**
|
|
1917
|
+
* The single built-in executor factory. Picks a leaf backend by data (`config.backend`),
|
|
1918
|
+
* injects the matching seam, and delegates to that backend's built-in implementation.
|
|
1919
|
+
* The `Executor` port stays OPEN: bring-your-own agents implement `Executor` directly, while Scope
|
|
1920
|
+
* or `createExecutorRegistry` still parses and seals their exact profile before use. Use this instead of a
|
|
1921
|
+
* per-vendor adapter or a closed `inline|sandbox|cli` switch — those bypass the
|
|
1922
|
+
* `UsageEvent` reporting channel.
|
|
1923
|
+
*/
|
|
1924
|
+
declare function createExecutor(config: ExecutorConfig): ExecutorFactory<unknown>;
|
|
1925
|
+
/**
|
|
1926
|
+
* The open resolver/registry. Pre-registers the three built-ins under their
|
|
1927
|
+
* runtime tags (`'router'`, `'sandbox'`, `'cli'`) and accepts `register(name,
|
|
1928
|
+
* factory)` for any additional runtime. A BYO `AgentSpec.executor` has highest routing precedence
|
|
1929
|
+
* after the same exact-profile intake validation. Registration + BYO remain open extension points.
|
|
1930
|
+
*
|
|
1931
|
+
* `resolve` precedence (frozen in `ExecutorRegistry`): a BYO `spec.executorFactory` →
|
|
1932
|
+
* `spec.executor` → `harness === null` → the `'router'` factory; else a registered factory for the
|
|
1933
|
+
* harness-derived runtime (`'sandbox'` for any `BackendType`); else fail loud.
|
|
1934
|
+
*/
|
|
1935
|
+
declare function createExecutorRegistry(): ExecutorRegistry;
|
|
1936
|
+
//#endregion
|
|
1937
|
+
//#region src/runtime/stream-agent-turn.d.ts
|
|
1938
|
+
/**
|
|
1939
|
+
* The execution substrate one turn runs on — a closed discriminated union over
|
|
1940
|
+
* the three stream surfaces the runtime already owns.
|
|
1941
|
+
*
|
|
1942
|
+
* @stable
|
|
1943
|
+
*/
|
|
1944
|
+
type AgentTurnBackend = {
|
|
1945
|
+
/** A Runtime-owned executor factory materialized from this exact canonical profile. */
|
|
1946
|
+
kind: 'executor';
|
|
1947
|
+
factory: ExecutorFactory<unknown>;
|
|
1948
|
+
/** Exact canonical identity materialized by the executor. */
|
|
1949
|
+
profile: AgentProfile;
|
|
1950
|
+
/** Model label stamped on cost-only `llm_call` events. Default `'agent'`. */
|
|
1951
|
+
agentRunName?: string;
|
|
1952
|
+
};
|
|
1953
|
+
/** @stable */
|
|
1954
|
+
interface StreamAgentTurnOptions {
|
|
1955
|
+
/** Caller-initiated cancellation. Terminates the stream with `final.status: 'aborted'`. */
|
|
1956
|
+
signal?: AbortSignal;
|
|
1957
|
+
/**
|
|
1958
|
+
* Wall-clock deadline for the whole turn in ms. An expired deadline aborts
|
|
1959
|
+
* the backend and terminates the stream with `final.status: 'failed'`
|
|
1960
|
+
* (a blown deadline is a turn failure, not a caller cancellation).
|
|
1961
|
+
*/
|
|
1962
|
+
timeoutMs?: number;
|
|
1963
|
+
/** Stable logical paid-call id, forwarded as the provider idempotency key and retained in evidence. */
|
|
1964
|
+
callId?: string;
|
|
1965
|
+
/** Caller trace tag retained in evidence and forwarded when the transport supports it. */
|
|
1966
|
+
correlationId?: string;
|
|
1967
|
+
/**
|
|
1968
|
+
* Opt-in tool-part projection for box and executor backends: sandbox tool
|
|
1969
|
+
* parts additionally surface in-stream as
|
|
1970
|
+
* `tool_call` / `tool_result` events (`mapSandboxToolEvent`), so a consumer
|
|
1971
|
+
* rendering tool activity needs no bespoke sandbox-event parser. Default
|
|
1972
|
+
* off — the stream vocabulary existing consumers see is unchanged. No-op
|
|
1973
|
+
* for the `chat` kind (its backend emits `RuntimeStreamEvent`s directly,
|
|
1974
|
+
* tool events included when the backend produces them).
|
|
1975
|
+
*/
|
|
1976
|
+
preserveToolParts?: boolean;
|
|
1977
|
+
/**
|
|
1978
|
+
* Raw-event tap for box-kind backends: called (and awaited) with every
|
|
1979
|
+
* unmapped `SandboxEvent` BEFORE it is projected, so a consumer can read
|
|
1980
|
+
* parts the chat-UX projection drops (part ids, step markers, custom
|
|
1981
|
+
* backend events) without forking the mapper. Purely observational — it
|
|
1982
|
+
* cannot alter the mapped stream. Never called for the `chat` kind, which
|
|
1983
|
+
* has no sandbox events.
|
|
1984
|
+
*/
|
|
1985
|
+
onRawEvent?: (event: SandboxEvent) => void | Promise<void>;
|
|
1986
|
+
}
|
|
1987
|
+
/**
|
|
1988
|
+
* Metered usage of one turn, summed over every cost-bearing event the backend
|
|
1989
|
+
* emitted. `input`/`output` are token counts and are accompanied by
|
|
1990
|
+
* `tokensKnown: false` when the backend did not report them. `costUsd`/`model`
|
|
1991
|
+
* are present only when the backend actually reported them.
|
|
1992
|
+
*
|
|
1993
|
+
* @stable
|
|
1994
|
+
*/
|
|
1995
|
+
interface AgentTurnUsage {
|
|
1996
|
+
input: number;
|
|
1997
|
+
output: number;
|
|
1998
|
+
/** Present when a real turn ran but the provider did not report token usage. */
|
|
1999
|
+
tokensKnown?: false;
|
|
2000
|
+
costUsd?: number;
|
|
2001
|
+
/** Present when Runtime could not prove the full dollar amount. */
|
|
2002
|
+
usdKnown?: false;
|
|
2003
|
+
/** Separately-labelled local/catalog estimate; never billed spend. */
|
|
2004
|
+
estimatedCostUsd?: number;
|
|
2005
|
+
/** Provider-reported prompt-cache fields; absent fields remain unknown. */
|
|
2006
|
+
promptCache?: Readonly<Record<string, number | string>>;
|
|
2007
|
+
/** Provider-reported reasoning-token subset of output, when available. */
|
|
2008
|
+
reasoningTokens?: number;
|
|
2009
|
+
model?: string;
|
|
2010
|
+
}
|
|
2011
|
+
/**
|
|
2012
|
+
* A drained turn: the terminal summary plus every event the stream yielded.
|
|
2013
|
+
* `status`/`error` mirror the terminal `final` event so a failed or aborted
|
|
2014
|
+
* turn stays inspectable without re-scanning `events`.
|
|
2015
|
+
*
|
|
2016
|
+
* @stable
|
|
2017
|
+
*/
|
|
2018
|
+
interface CollectedAgentTurn {
|
|
2019
|
+
finalText: string;
|
|
2020
|
+
/** Exact terminal artifact output from a Runtime-owned executor. */
|
|
2021
|
+
output?: unknown;
|
|
2022
|
+
usage: AgentTurnUsage;
|
|
2023
|
+
/** Exact underlying transport calls when the Runtime-owned executor reports them. */
|
|
2024
|
+
transportAttempts?: number;
|
|
2025
|
+
toolCalls: Array<{
|
|
2026
|
+
id?: string;
|
|
2027
|
+
name: string;
|
|
2028
|
+
arguments: string;
|
|
2029
|
+
}>;
|
|
2030
|
+
events: RuntimeStreamEvent[];
|
|
2031
|
+
status: AgentTaskStatus;
|
|
2032
|
+
error?: BackendErrorDetail;
|
|
2033
|
+
/** Public Sandbox outcome, when the turn ran through a Sandbox stream or executor. */
|
|
2034
|
+
sandboxOutcome?: AgentRunOutcome;
|
|
2035
|
+
}
|
|
2036
|
+
/**
|
|
2037
|
+
* Run ONE agent turn on any backend kind and stream its events. Yields the
|
|
2038
|
+
* `RuntimeStreamEvent` vocabulary incrementally and always ends with a `final`
|
|
2039
|
+
* event carrying the turn's text and usage (`metadata.tokenUsage`,
|
|
2040
|
+
* `metadata.costUsd?`, `metadata.model?`) — on success, failure, abort, and
|
|
2041
|
+
* timeout alike. The generator never throws; failures surface in-band as
|
|
2042
|
+
* `backend_error` + `final` with a typed `error` detail.
|
|
2043
|
+
*
|
|
2044
|
+
* @stable
|
|
2045
|
+
*/
|
|
2046
|
+
declare function streamAgentTurn(backend: AgentTurnBackend, input: AgentTurnInput, opts?: StreamAgentTurnOptions): AsyncGenerator<RuntimeStreamEvent>;
|
|
2047
|
+
/**
|
|
2048
|
+
* Drain a `streamAgentTurn` stream (or any `RuntimeStreamEvent` stream that
|
|
2049
|
+
* honors its terminal contract) into the turn summary plus the full event
|
|
2050
|
+
* list. Fail-loud: throws when the stream ends without a terminal `final`
|
|
2051
|
+
* event — a stream that violates the contract must not read as an empty turn.
|
|
2052
|
+
*
|
|
2053
|
+
* @stable
|
|
2054
|
+
*/
|
|
2055
|
+
declare function collectAgentTurn(stream: AsyncIterable<RuntimeStreamEvent>): Promise<CollectedAgentTurn>;
|
|
2056
|
+
//#endregion
|
|
2057
|
+
export { decodeHarnessUsage as $, McpTransport as $n, WorktreeCheckRunner as $t, createInbox as A, ToolLoopMessageRecord as An, CreateAgentEnvironmentInput$1 as At, secretEnvOfMcpServer as B, PeerMailReadout as Bn, ResourceRequest as Bt, SteerableSandboxArgs as C, runLocalHarness as Cn, AgentProfileRef$1 as Ct, Inbox as D, ToolLoopChat as Dn, AgentTurnResult$2 as Dt, AuthorityInboxMessage as E, ToolLoopCallContext as En, AgentSessionStatus$1 as Et, ResolvedMcpServerLaunch as F, PeerMailEnvelope as Fn, PlacementInfo as Ft, CodexRolloutStoreRef as G, claimsAuthority as Gn, providerAsSandboxClient as Gt, CodexRolloutIdentity as H, PeerMailSendInput as Hn, WorkspaceRequest as Ht, envKeyProvider as I, PeerMailEvent as In, ProviderAsSandboxClientOptions as It, addHarnessUsage as J, peerMailTools as Jn, CreateTangleSandboxExactProcessProviderOptions as Jt, CodexRolloutTurn as K, createPeerMailbox as Kn, resolveAgentEnvironmentProvider as Kt, mcpSecretEnvMetadataKey as L, PeerMailKind as Ln, ProviderExecutorOptions as Lt, BridgeModelCredential as M, AUTHORITY_MARKERS as Mn, ExecRequest as Mt, BridgeSeam as N, DEFAULT_PEER_MAIL_LIMITS as Nn, ExecResult as Nt, InboxMessage as O, ToolLoopCompaction as On, CheckpointRef as Ot, KeyProvider as P, PEER_MAIL_WIRE_KEY as Pn, ForkRequest as Pt, HarnessUsage as Q, McpToolDescriptor as Qn, InPlaceHarnessResult as Qt, resolveMcpServerLaunch as R, PeerMailLimits as Rn, ProviderLeafOut as Rt, SandboxSteeringOptions as S, parseCodexTokenUsage as Sn, AgentEnvironmentSummary as St, createSteerableSandboxSession as T, ToolSpec as Tn, AgentSessionRef as Tt, CodexRolloutSession as U, PeerMailbox as Un, createAgentEnvironmentProviderRegistry as Ut, CodexForkBoundary as V, PeerMailRefusal as Vn, SandboxClientProviderOptions as Vt, CodexRolloutStoreReader as W, PeerMailboxOptions as Wn, providerAsExecutor as Wt, harnessUsageIsEmpty as X, JsonRpcMessage as Xn, createTangleSandboxExactProcessProvider as Xt, createCodexRolloutStoreReader as Y, peerMailVerbNames as Yn, SandboxControlClient as Yt, readCodexRolloutSession as Z, JsonRpcResponse as Zn, ProviderPlacement as Zt, cliInPlaceExecutor as _, LocalHarness as _n, AgentEnvironmentProvider$1 as _t, StreamAgentTurnOptions as a, DiffResult as an, assertSandboxServedModel as at, createExecutorRegistry as b, harnessSupportsReasoningEffort as bn, AgentEnvironmentQuery as bt, CliInPlaceSeam as c, WorktreeHandle as cn, extractLlmCallEvent as ct, CliWorktreeSeam as d, removeWorktree as dn, sandboxEventServedBackend as dt, WorktreeCommandResult as en, Redactor as er, SandboxLeafOut as et, ExecutorConfig as f, CodexExecutionEvidence as fn, sandboxProgressEvents as ft, SandboxSeam as g, LOCAL_HARNESSES as gn, AgentEnvironmentEvent$1 as gt, RouterToolsSeam as h, DEFAULT_LOCAL_HARNESS as hn, AgentEnvironmentCapabilities$1 as ht, CollectedAgentTurn as i, DiffOptions as in, SandboxUsageLedger as it, BridgeHarnessStore as j, ToolLoopToolCall as jn, DEFAULT_SANDBOX_IDLE_TIMEOUT_SECONDS as jt, PeerInboxMessage as k, ToolLoopCompactionOptions as kn, CheckpointRequest as kt, CliSeam as l, captureWorktreeDiff as ln, mapSandboxEvent as lt, RouterSeam as m, CodexTokenUsage as mn, AgentEnvironment$1 as mt, AgentTurnInput$1 as n, WorktreeProfileMaterializationReceipt as nn, resolveRedactor as nr, SandboxServedBackend as nt, collectAgentTurn as o, GitRunner as on, createSandboxToolPartState as ot, ProviderSeam as p, CodexExecutionPolicy as pn, sumSandboxUsage as pt, CodexStoreDelta as q, isPeerMailEnvelope as qn, sandboxClientAsProvider as qt, AgentTurnUsage as r, CreateWorktreeOptions as rn, SandboxToolPartState as rt, streamAgentTurn as s, RemoveWorktreeOptions as sn, createSandboxUsageLedger as st, AgentTurnBackend as t, WorktreeHarnessResult as tn, defaultRedactor as tr, SandboxOutputMarker as tt, CliWorktreeBridgeSeam as u, createWorktree as un, mapSandboxToolEvent as ut, cliWorktreeExecutor as v, LocalHarnessResult as vn, AgentEnvironmentProviderRef as vt, SteerableSandboxSession as w, RouterTransportConfig as wn, AgentSession as wt, DEFAULT_SANDBOX_STEERING_MAX_TURNS as x, localHarnessExecutable as xn, AgentEnvironmentStatus as xt, createExecutor as y, RunLocalHarnessOptions as yn, AgentEnvironmentProviderRegistry as yt, resolveSecretEnv as z, PeerMailOutcome as zn, ProviderPromptOptions as zt };
|
|
2058
|
+
//# sourceMappingURL=stream-agent-turn-hiU4vgGX.d.ts.map
|