@tanstack/ai-sandbox 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +182 -0
- package/dist/esm/agents-file.d.ts +36 -0
- package/dist/esm/agents-file.js +44 -0
- package/dist/esm/agents-file.js.map +1 -0
- package/dist/esm/approvals.d.ts +38 -0
- package/dist/esm/approvals.js +36 -0
- package/dist/esm/approvals.js.map +1 -0
- package/dist/esm/bootstrap.d.ts +17 -0
- package/dist/esm/bootstrap.js +124 -0
- package/dist/esm/bootstrap.js.map +1 -0
- package/dist/esm/bridge-events.d.ts +21 -0
- package/dist/esm/bridge-events.js +76 -0
- package/dist/esm/bridge-events.js.map +1 -0
- package/dist/esm/capabilities.d.ts +26 -0
- package/dist/esm/capabilities.js +29 -0
- package/dist/esm/capabilities.js.map +1 -0
- package/dist/esm/contracts.d.ts +211 -0
- package/dist/esm/errors.d.ts +16 -0
- package/dist/esm/errors.js +25 -0
- package/dist/esm/errors.js.map +1 -0
- package/dist/esm/git-exec.d.ts +2 -0
- package/dist/esm/git-exec.js +68 -0
- package/dist/esm/git-exec.js.map +1 -0
- package/dist/esm/harness-cwd.d.ts +2 -0
- package/dist/esm/harness-cwd.js +24 -0
- package/dist/esm/harness-cwd.js.map +1 -0
- package/dist/esm/index.d.ts +39 -0
- package/dist/esm/index.js +103 -0
- package/dist/esm/index.js.map +1 -0
- package/dist/esm/key.d.ts +20 -0
- package/dist/esm/key.js +41 -0
- package/dist/esm/key.js.map +1 -0
- package/dist/esm/middleware.d.ts +5 -0
- package/dist/esm/middleware.js +140 -0
- package/dist/esm/middleware.js.map +1 -0
- package/dist/esm/ngrok.d.ts +16 -0
- package/dist/esm/ngrok.js +54 -0
- package/dist/esm/ngrok.js.map +1 -0
- package/dist/esm/policy.d.ts +47 -0
- package/dist/esm/policy.js +44 -0
- package/dist/esm/policy.js.map +1 -0
- package/dist/esm/projection.d.ts +31 -0
- package/dist/esm/projection.js +9 -0
- package/dist/esm/projection.js.map +1 -0
- package/dist/esm/remote-tools.d.ts +48 -0
- package/dist/esm/remote-tools.js +76 -0
- package/dist/esm/remote-tools.js.map +1 -0
- package/dist/esm/run-log.d.ts +81 -0
- package/dist/esm/run-log.js +107 -0
- package/dist/esm/run-log.js.map +1 -0
- package/dist/esm/run.d.ts +58 -0
- package/dist/esm/run.js +89 -0
- package/dist/esm/run.js.map +1 -0
- package/dist/esm/runner.d.ts +21 -0
- package/dist/esm/runner.js +54 -0
- package/dist/esm/runner.js.map +1 -0
- package/dist/esm/sandbox.d.ts +79 -0
- package/dist/esm/sandbox.js +125 -0
- package/dist/esm/sandbox.js.map +1 -0
- package/dist/esm/secrets.d.ts +37 -0
- package/dist/esm/secrets.js +59 -0
- package/dist/esm/secrets.js.map +1 -0
- package/dist/esm/setup-plan.d.ts +13 -0
- package/dist/esm/setup-plan.js +16 -0
- package/dist/esm/setup-plan.js.map +1 -0
- package/dist/esm/shell.d.ts +45 -0
- package/dist/esm/shell.js +164 -0
- package/dist/esm/shell.js.map +1 -0
- package/dist/esm/store.d.ts +53 -0
- package/dist/esm/store.js +34 -0
- package/dist/esm/store.js.map +1 -0
- package/dist/esm/tool-bridge.d.ts +130 -0
- package/dist/esm/tool-bridge.js +197 -0
- package/dist/esm/tool-bridge.js.map +1 -0
- package/dist/esm/watch.d.ts +36 -0
- package/dist/esm/watch.js +144 -0
- package/dist/esm/watch.js.map +1 -0
- package/dist/esm/workspace.d.ts +128 -0
- package/dist/esm/workspace.js +42 -0
- package/dist/esm/workspace.js.map +1 -0
- package/package.json +72 -0
- package/skills/ai-sandbox/SKILL.md +366 -0
- package/src/agents-file.ts +101 -0
- package/src/approvals.ts +96 -0
- package/src/bootstrap.ts +196 -0
- package/src/bridge-events.ts +112 -0
- package/src/capabilities.ts +47 -0
- package/src/contracts.ts +236 -0
- package/src/errors.ts +31 -0
- package/src/git-exec.ts +114 -0
- package/src/harness-cwd.ts +38 -0
- package/src/index.ts +222 -0
- package/src/key.ts +70 -0
- package/src/middleware.ts +233 -0
- package/src/ngrok.ts +85 -0
- package/src/policy.ts +111 -0
- package/src/projection.ts +46 -0
- package/src/remote-tools.ts +180 -0
- package/src/run-log.ts +224 -0
- package/src/run.ts +167 -0
- package/src/runner.ts +99 -0
- package/src/sandbox.ts +259 -0
- package/src/secrets.ts +101 -0
- package/src/setup-plan.ts +25 -0
- package/src/shell.ts +288 -0
- package/src/store.ts +83 -0
- package/src/tool-bridge.ts +399 -0
- package/src/watch.ts +256 -0
- package/src/workspace.ts +151 -0
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Host-tool delegation for the CO-LOCATED ("combined") sandbox model.
|
|
3
|
+
*
|
|
4
|
+
* In the co-located model the harness loop AND its MCP tool-bridge run INSIDE
|
|
5
|
+
* the container (the in-container sandbox is just `local-process`, so the
|
|
6
|
+
* existing adapter + `nodeHttpBridgeProvisioner` serve the bridge on the
|
|
7
|
+
* container's own `localhost` with native stdin — nothing new there). The one
|
|
8
|
+
* thing that still must cross the container→orchestrator boundary is the
|
|
9
|
+
* **execution** of `chat()`-provided server tools: their `execute()` closures
|
|
10
|
+
* (DB / secrets / app state) live in the orchestrator, not the container.
|
|
11
|
+
*
|
|
12
|
+
* This module is that narrow seam:
|
|
13
|
+
* - {@link remoteToolStubs} (container side) rebuilds `chat()` tools from
|
|
14
|
+
* serialized {@link ToolDescriptor}s; each stub's `execute` delegates to a
|
|
15
|
+
* {@link RemoteToolExecutor} instead of running locally. The adapter bridges
|
|
16
|
+
* these stubs exactly like real tools.
|
|
17
|
+
* - {@link httpRemoteToolExecutor} (container side) is the default executor: it
|
|
18
|
+
* POSTs `{ name, args }` to the orchestrator's tool-exec endpoint.
|
|
19
|
+
* - {@link executeHostTool} (orchestrator side) runs the REAL tool and returns
|
|
20
|
+
* its raw result — the only host code the container can reach.
|
|
21
|
+
*
|
|
22
|
+
* So the public network surface shrinks from "the whole MCP protocol" (served
|
|
23
|
+
* from the orchestrator in the DO-drives-container model) to "one authenticated
|
|
24
|
+
* tool-exec call" — the MCP transport itself never leaves the container.
|
|
25
|
+
*/
|
|
26
|
+
import type { AnyTool } from '@tanstack/ai'
|
|
27
|
+
import type { ToolDescriptor } from './tool-bridge'
|
|
28
|
+
|
|
29
|
+
/** Per-call options forwarded to a {@link RemoteToolExecutor}. */
|
|
30
|
+
export interface RemoteToolExecuteOptions {
|
|
31
|
+
/** Cancels the in-flight remote call when the in-container run aborts. */
|
|
32
|
+
signal?: AbortSignal
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** Runs a named host tool with the given args, returning its raw result. */
|
|
36
|
+
export interface RemoteToolExecutor {
|
|
37
|
+
execute: (
|
|
38
|
+
name: string,
|
|
39
|
+
args: unknown,
|
|
40
|
+
options?: RemoteToolExecuteOptions,
|
|
41
|
+
) => Promise<unknown>
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** Wire shape of a tool-exec request the container POSTs to the orchestrator. */
|
|
45
|
+
export interface ToolExecRequest {
|
|
46
|
+
name: string
|
|
47
|
+
args: unknown
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** Narrow an unknown body into a {@link ToolExecRequest} (project rule: no `as`). */
|
|
51
|
+
export function isToolExecRequest(value: unknown): value is ToolExecRequest {
|
|
52
|
+
return (
|
|
53
|
+
value !== null &&
|
|
54
|
+
typeof value === 'object' &&
|
|
55
|
+
'name' in value &&
|
|
56
|
+
typeof value.name === 'string'
|
|
57
|
+
)
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Rebuild `chat()` tool objects (container side) from serialized descriptors.
|
|
62
|
+
* Each stub advertises the descriptor's JSON-schema and delegates `execute` to
|
|
63
|
+
* the executor; the harness adapter bridges them like any other tool. The
|
|
64
|
+
* harness's `abortSignal` is forwarded so a cancelled run cancels the in-flight
|
|
65
|
+
* remote call too.
|
|
66
|
+
*/
|
|
67
|
+
export function remoteToolStubs(
|
|
68
|
+
descriptors: Array<ToolDescriptor>,
|
|
69
|
+
executor: RemoteToolExecutor,
|
|
70
|
+
): Array<AnyTool> {
|
|
71
|
+
return descriptors.map((descriptor) => ({
|
|
72
|
+
name: descriptor.name,
|
|
73
|
+
description: descriptor.description ?? '',
|
|
74
|
+
inputSchema: descriptor.inputSchema,
|
|
75
|
+
execute: (args: unknown, options?: { abortSignal?: AbortSignal }) =>
|
|
76
|
+
executor.execute(
|
|
77
|
+
descriptor.name,
|
|
78
|
+
args,
|
|
79
|
+
options?.abortSignal !== undefined
|
|
80
|
+
? { signal: options.abortSignal }
|
|
81
|
+
: {},
|
|
82
|
+
),
|
|
83
|
+
}))
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Serialize `chat()` tools to wire descriptors to send into the container.
|
|
88
|
+
* `inputSchema` must already be a plain JSON-schema object (convert Standard
|
|
89
|
+
* Schemas before calling, the same way harness adapters advertise tools).
|
|
90
|
+
*/
|
|
91
|
+
export function toolDescriptors(tools: Array<AnyTool>): Array<ToolDescriptor> {
|
|
92
|
+
return tools.map((tool) => ({
|
|
93
|
+
name: tool.name,
|
|
94
|
+
description: tool.description,
|
|
95
|
+
inputSchema: isJsonSchemaObject(tool.inputSchema)
|
|
96
|
+
? tool.inputSchema
|
|
97
|
+
: { type: 'object', properties: {} },
|
|
98
|
+
}))
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
function isJsonSchemaObject(
|
|
102
|
+
value: unknown,
|
|
103
|
+
): value is { type: 'object'; [key: string]: unknown } {
|
|
104
|
+
return (
|
|
105
|
+
value !== null &&
|
|
106
|
+
typeof value === 'object' &&
|
|
107
|
+
'type' in value &&
|
|
108
|
+
(value as { type?: unknown }).type === 'object'
|
|
109
|
+
)
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/** Wire shape of a tool-exec response from the orchestrator. */
|
|
113
|
+
interface ToolExecResponse {
|
|
114
|
+
result: unknown
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
function isToolExecResponse(value: unknown): value is ToolExecResponse {
|
|
118
|
+
return value !== null && typeof value === 'object' && 'result' in value
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* The default {@link RemoteToolExecutor}: POST `{ name, args }` (bearer-gated)
|
|
123
|
+
* to the orchestrator's tool-exec endpoint and return its `result`. A non-2xx
|
|
124
|
+
* or malformed response throws (surfaced to the agent as a failed tool call by
|
|
125
|
+
* the bridge) — never silently swallowed.
|
|
126
|
+
*/
|
|
127
|
+
export function httpRemoteToolExecutor(
|
|
128
|
+
url: string,
|
|
129
|
+
token: string,
|
|
130
|
+
): RemoteToolExecutor {
|
|
131
|
+
return {
|
|
132
|
+
async execute(name, args, options) {
|
|
133
|
+
const res = await fetch(url, {
|
|
134
|
+
method: 'POST',
|
|
135
|
+
headers: {
|
|
136
|
+
'content-type': 'application/json',
|
|
137
|
+
authorization: `Bearer ${token}`,
|
|
138
|
+
},
|
|
139
|
+
body: JSON.stringify({ name, args }),
|
|
140
|
+
...(options?.signal !== undefined ? { signal: options.signal } : {}),
|
|
141
|
+
})
|
|
142
|
+
if (!res.ok) {
|
|
143
|
+
const text = await res.text()
|
|
144
|
+
throw new Error(
|
|
145
|
+
`remote tool "${name}" failed: ${res.status} ${text.slice(0, 200)}`,
|
|
146
|
+
)
|
|
147
|
+
}
|
|
148
|
+
const body: unknown = await res.json()
|
|
149
|
+
if (!isToolExecResponse(body)) {
|
|
150
|
+
throw new Error(
|
|
151
|
+
`remote tool "${name}": malformed orchestrator response`,
|
|
152
|
+
)
|
|
153
|
+
}
|
|
154
|
+
return body.result
|
|
155
|
+
},
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* Run a host tool by name with the given args, returning its raw result
|
|
161
|
+
* (orchestrator side of {@link httpRemoteToolExecutor}). Throws for an unknown
|
|
162
|
+
* tool or one with no `execute` — the orchestrator surfaces that as a 4xx/5xx.
|
|
163
|
+
*/
|
|
164
|
+
export function executeHostTool(
|
|
165
|
+
tools: Array<AnyTool>,
|
|
166
|
+
name: string,
|
|
167
|
+
args: unknown,
|
|
168
|
+
options: { context?: unknown; signal?: AbortSignal } = {},
|
|
169
|
+
): Promise<unknown> {
|
|
170
|
+
const tool = tools.find((candidate) => candidate.name === name)
|
|
171
|
+
if (!tool?.execute) {
|
|
172
|
+
return Promise.reject(new Error(`Unknown tool: ${name}`))
|
|
173
|
+
}
|
|
174
|
+
return Promise.resolve(
|
|
175
|
+
tool.execute(args ?? {}, {
|
|
176
|
+
context: options.context,
|
|
177
|
+
abortSignal: options.signal,
|
|
178
|
+
}),
|
|
179
|
+
)
|
|
180
|
+
}
|
package/src/run-log.ts
ADDED
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Resumable run event-log — the primitive that lets a trigger (e.g. a
|
|
3
|
+
* Cloudflare Worker) start an agent run and return immediately while a durable
|
|
4
|
+
* orchestrator (e.g. a Durable Object) drives the run and persists every
|
|
5
|
+
* emitted {@link StreamChunk} under a monotonic `seq`.
|
|
6
|
+
*
|
|
7
|
+
* Clients tail the log from a cursor (`fromSeq`), so a dropped connection, a new
|
|
8
|
+
* browser tab, or an orchestrator that hibernated between chunks all reconnect
|
|
9
|
+
* cleanly: replay everything after the client's last-seen `seq`, then live-tail
|
|
10
|
+
* until the run reaches a terminal status. The *run* never depends on any single
|
|
11
|
+
* connection staying open — that is what makes the serverless/edge model work.
|
|
12
|
+
*
|
|
13
|
+
* This module is transport- and storage-agnostic. {@link InMemoryRunEventLog} is
|
|
14
|
+
* the default (single-process / tests); a durable backend (DO storage, KV, SQL)
|
|
15
|
+
* implements the same {@link RunEventLog} interface — see the Cloudflare example.
|
|
16
|
+
*/
|
|
17
|
+
import type { StreamChunk } from '@tanstack/ai'
|
|
18
|
+
|
|
19
|
+
/** A terminal run status: no further events will be appended. */
|
|
20
|
+
export type TerminalRunStatus = 'done' | 'error' | 'aborted'
|
|
21
|
+
|
|
22
|
+
/** Lifecycle status of a run. `done`/`error`/`aborted` are terminal. */
|
|
23
|
+
export type RunStatus = 'running' | TerminalRunStatus
|
|
24
|
+
|
|
25
|
+
const TERMINAL: ReadonlySet<RunStatus> = new Set<RunStatus>([
|
|
26
|
+
'done',
|
|
27
|
+
'error',
|
|
28
|
+
'aborted',
|
|
29
|
+
])
|
|
30
|
+
|
|
31
|
+
/** Whether a run status is terminal (no further events will be appended). */
|
|
32
|
+
export function isTerminalRunStatus(status: RunStatus): boolean {
|
|
33
|
+
return TERMINAL.has(status)
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export interface RunError {
|
|
37
|
+
message: string
|
|
38
|
+
code?: string
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Durable bookkeeping for a single run. */
|
|
42
|
+
export interface RunRecord {
|
|
43
|
+
runId: string
|
|
44
|
+
threadId?: string
|
|
45
|
+
status: RunStatus
|
|
46
|
+
/** Seq of the last appended event, or `-1` when no events yet. */
|
|
47
|
+
lastSeq: number
|
|
48
|
+
error?: RunError
|
|
49
|
+
createdAt: number
|
|
50
|
+
updatedAt: number
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/** One persisted event: a chunk plus its monotonic, gap-free sequence number. */
|
|
54
|
+
export interface RunEvent {
|
|
55
|
+
seq: number
|
|
56
|
+
chunk: StreamChunk
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export interface RunEventLogReadOptions {
|
|
60
|
+
/**
|
|
61
|
+
* Exclusive cursor: only events with `seq > fromSeq` are yielded. Pass the
|
|
62
|
+
* client's last-seen `seq` to resume; omit (or `-1`) to replay from the start.
|
|
63
|
+
*/
|
|
64
|
+
fromSeq?: number
|
|
65
|
+
/** Stop tailing when this fires (e.g. the client disconnected). */
|
|
66
|
+
signal?: AbortSignal
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Append-only, `seq`-indexed log of a run's stream, with resumable reads.
|
|
71
|
+
*
|
|
72
|
+
* Contract:
|
|
73
|
+
* - `append` assigns the next `seq` (0, 1, 2, …) and returns it.
|
|
74
|
+
* - `read` yields the backlog after `fromSeq` in order, then live-tails new
|
|
75
|
+
* events, and RETURNS once the run is terminal and the cursor has caught up.
|
|
76
|
+
* - All methods reject for an unknown `runId` except `get`, which resolves null.
|
|
77
|
+
*/
|
|
78
|
+
export interface RunEventLog {
|
|
79
|
+
/** Idempotently create (or return) the run record. */
|
|
80
|
+
open: (input: { runId: string; threadId?: string }) => Promise<RunRecord>
|
|
81
|
+
/** Append one chunk; resolves with its assigned `seq`. */
|
|
82
|
+
append: (runId: string, chunk: StreamChunk) => Promise<number>
|
|
83
|
+
/** Move the run to a terminal status. Idempotent for the same status. */
|
|
84
|
+
finish: (
|
|
85
|
+
runId: string,
|
|
86
|
+
status: TerminalRunStatus,
|
|
87
|
+
error?: RunError,
|
|
88
|
+
) => Promise<void>
|
|
89
|
+
/** Current record, or null if the run is unknown. */
|
|
90
|
+
get: (runId: string) => Promise<RunRecord | null>
|
|
91
|
+
/** Replay-then-tail events with `seq > fromSeq` until the run is terminal. */
|
|
92
|
+
read: (
|
|
93
|
+
runId: string,
|
|
94
|
+
options?: RunEventLogReadOptions,
|
|
95
|
+
) => AsyncIterable<RunEvent>
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/** Per-run state for the in-memory log. */
|
|
99
|
+
interface RunState {
|
|
100
|
+
record: RunRecord
|
|
101
|
+
chunks: Array<StreamChunk>
|
|
102
|
+
/** Resolved (and cleared) whenever an event is appended or status changes. */
|
|
103
|
+
waiters: Set<() => void>
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Single-process {@link RunEventLog}. Backs `read`'s live-tail with an internal
|
|
108
|
+
* waiter set: `append`/`finish` wake every blocked reader. Suitable for a
|
|
109
|
+
* long-running Node host, tests, and as the reference implementation a durable
|
|
110
|
+
* backend mirrors.
|
|
111
|
+
*/
|
|
112
|
+
export class InMemoryRunEventLog implements RunEventLog {
|
|
113
|
+
private readonly runs = new Map<string, RunState>()
|
|
114
|
+
|
|
115
|
+
private now(): number {
|
|
116
|
+
return Date.now()
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
private require(runId: string): RunState {
|
|
120
|
+
const state = this.runs.get(runId)
|
|
121
|
+
if (!state) throw new Error(`run-log: unknown runId "${runId}"`)
|
|
122
|
+
return state
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
private wake(state: RunState): void {
|
|
126
|
+
const waiters = [...state.waiters]
|
|
127
|
+
state.waiters.clear()
|
|
128
|
+
for (const resolve of waiters) resolve()
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
// Mutators return a Promise without `async` so contract violations REJECT
|
|
132
|
+
// (rather than throwing synchronously from a Promise-typed method — a
|
|
133
|
+
// `.catch()` footgun) without an `await`-less async body.
|
|
134
|
+
open(input: { runId: string; threadId?: string }): Promise<RunRecord> {
|
|
135
|
+
const existing = this.runs.get(input.runId)
|
|
136
|
+
if (existing) return Promise.resolve({ ...existing.record })
|
|
137
|
+
const now = this.now()
|
|
138
|
+
const record: RunRecord = {
|
|
139
|
+
runId: input.runId,
|
|
140
|
+
...(input.threadId !== undefined ? { threadId: input.threadId } : {}),
|
|
141
|
+
status: 'running',
|
|
142
|
+
lastSeq: -1,
|
|
143
|
+
createdAt: now,
|
|
144
|
+
updatedAt: now,
|
|
145
|
+
}
|
|
146
|
+
this.runs.set(input.runId, { record, chunks: [], waiters: new Set() })
|
|
147
|
+
return Promise.resolve({ ...record })
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
append(runId: string, chunk: StreamChunk): Promise<number> {
|
|
151
|
+
const state = this.runs.get(runId)
|
|
152
|
+
if (!state) {
|
|
153
|
+
return Promise.reject(new Error(`run-log: unknown runId "${runId}"`))
|
|
154
|
+
}
|
|
155
|
+
if (isTerminalRunStatus(state.record.status)) {
|
|
156
|
+
return Promise.reject(
|
|
157
|
+
new Error(
|
|
158
|
+
`run-log: cannot append to terminal run "${runId}" (status=${state.record.status})`,
|
|
159
|
+
),
|
|
160
|
+
)
|
|
161
|
+
}
|
|
162
|
+
// Derive seq from the record's cursor (not `chunks.length`) so the gap-free
|
|
163
|
+
// invariant holds the same way the durable backend computes it, even if the
|
|
164
|
+
// backlog is ever trimmed/compacted.
|
|
165
|
+
const seq = state.record.lastSeq + 1
|
|
166
|
+
state.chunks.push(chunk)
|
|
167
|
+
state.record.lastSeq = seq
|
|
168
|
+
state.record.updatedAt = this.now()
|
|
169
|
+
this.wake(state)
|
|
170
|
+
return Promise.resolve(seq)
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
finish(
|
|
174
|
+
runId: string,
|
|
175
|
+
status: TerminalRunStatus,
|
|
176
|
+
error?: RunError,
|
|
177
|
+
): Promise<void> {
|
|
178
|
+
const state = this.runs.get(runId)
|
|
179
|
+
if (!state) {
|
|
180
|
+
return Promise.reject(new Error(`run-log: unknown runId "${runId}"`))
|
|
181
|
+
}
|
|
182
|
+
if (isTerminalRunStatus(state.record.status)) return Promise.resolve()
|
|
183
|
+
state.record.status = status
|
|
184
|
+
if (error !== undefined) state.record.error = error
|
|
185
|
+
state.record.updatedAt = this.now()
|
|
186
|
+
this.wake(state)
|
|
187
|
+
return Promise.resolve()
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
get(runId: string): Promise<RunRecord | null> {
|
|
191
|
+
const state = this.runs.get(runId)
|
|
192
|
+
return Promise.resolve(state ? { ...state.record } : null)
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
async *read(
|
|
196
|
+
runId: string,
|
|
197
|
+
options?: RunEventLogReadOptions,
|
|
198
|
+
): AsyncIterable<RunEvent> {
|
|
199
|
+
const state = this.require(runId)
|
|
200
|
+
const signal = options?.signal
|
|
201
|
+
let cursor = options?.fromSeq ?? -1
|
|
202
|
+
while (!signal?.aborted) {
|
|
203
|
+
while (cursor < state.record.lastSeq) {
|
|
204
|
+
cursor += 1
|
|
205
|
+
const chunk = state.chunks[cursor]
|
|
206
|
+
if (chunk !== undefined) yield { seq: cursor, chunk }
|
|
207
|
+
}
|
|
208
|
+
if (isTerminalRunStatus(state.record.status)) return
|
|
209
|
+
await this.waitForChange(state, signal)
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
private waitForChange(state: RunState, signal?: AbortSignal): Promise<void> {
|
|
214
|
+
return new Promise<void>((resolve) => {
|
|
215
|
+
const wake = (): void => {
|
|
216
|
+
state.waiters.delete(wake)
|
|
217
|
+
if (signal) signal.removeEventListener('abort', wake)
|
|
218
|
+
resolve()
|
|
219
|
+
}
|
|
220
|
+
state.waiters.add(wake)
|
|
221
|
+
if (signal) signal.addEventListener('abort', wake, { once: true })
|
|
222
|
+
})
|
|
223
|
+
}
|
|
224
|
+
}
|
package/src/run.ts
ADDED
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The "run driver" for the inverted/serverless sandbox model: pump a `chat()`
|
|
3
|
+
* stream into a {@link RunEventLog} so a trigger can return immediately while a
|
|
4
|
+
* durable orchestrator drives the run and clients tail from a cursor.
|
|
5
|
+
*
|
|
6
|
+
* The key inversion vs. a classic request/response handler: there is no caller
|
|
7
|
+
* holding the stream open, so nothing to throw an error *back to*. The log is
|
|
8
|
+
* the only channel — every chunk (including a terminal {@link EventType.RUN_ERROR})
|
|
9
|
+
* is persisted under a `seq`, and a thrown stream error is recorded as a
|
|
10
|
+
* synthesized `RUN_ERROR` event plus the record's `error` field. Tailing clients
|
|
11
|
+
* therefore always observe failures; {@link pipeToRunLog} never rejects.
|
|
12
|
+
*/
|
|
13
|
+
import { EventType } from '@tanstack/ai'
|
|
14
|
+
import type { StreamChunk } from '@tanstack/ai'
|
|
15
|
+
import type { RunError, RunEvent, RunEventLog, RunRecord } from './run-log'
|
|
16
|
+
|
|
17
|
+
/** Whether a chunk is the terminal error event the chat engine emits. */
|
|
18
|
+
function isRunErrorChunk(
|
|
19
|
+
chunk: StreamChunk,
|
|
20
|
+
): chunk is StreamChunk & { message: string; code?: string } {
|
|
21
|
+
return chunk.type === EventType.RUN_ERROR
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/** Pull `{ message, code }` off a RUN_ERROR chunk for the run record. */
|
|
25
|
+
function runErrorFromChunk(
|
|
26
|
+
chunk: StreamChunk & { message: string; code?: string },
|
|
27
|
+
): RunError {
|
|
28
|
+
return chunk.code !== undefined
|
|
29
|
+
? { message: chunk.message, code: chunk.code }
|
|
30
|
+
: { message: chunk.message }
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** Render an unknown thrown value as a stable error message. */
|
|
34
|
+
function messageOf(error: unknown): string {
|
|
35
|
+
return error instanceof Error ? error.message : String(error)
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/** Build the synthetic RUN_ERROR chunk appended when the stream throws. */
|
|
39
|
+
function syntheticRunError(message: string): StreamChunk {
|
|
40
|
+
const chunk: { type: EventType.RUN_ERROR; message: string } = {
|
|
41
|
+
type: EventType.RUN_ERROR,
|
|
42
|
+
message,
|
|
43
|
+
}
|
|
44
|
+
return chunk
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export interface PipeToRunLogOptions {
|
|
48
|
+
log: RunEventLog
|
|
49
|
+
runId: string
|
|
50
|
+
threadId?: string
|
|
51
|
+
/** Abort consumption mid-stream; the run finishes as `aborted`. */
|
|
52
|
+
signal?: AbortSignal
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Open the run, append every chunk from `stream`, and finish with the right
|
|
57
|
+
* terminal status. Resolves with the final {@link RunRecord} and never rejects:
|
|
58
|
+
* a thrown stream error is surfaced as a `RUN_ERROR` event + the record's
|
|
59
|
+
* `error`, which is what tailing clients see.
|
|
60
|
+
*
|
|
61
|
+
* - normal completion → `finish('done')`
|
|
62
|
+
* - a `RUN_ERROR` chunk → append it, then `finish('error', { message, code })`
|
|
63
|
+
* - the stream throws → append a synthesized `RUN_ERROR`, then `finish('error')`
|
|
64
|
+
* - `signal` aborts mid-stream → stop consuming, `finish('aborted')`
|
|
65
|
+
*/
|
|
66
|
+
export async function pipeToRunLog(
|
|
67
|
+
stream: AsyncIterable<StreamChunk>,
|
|
68
|
+
opts: PipeToRunLogOptions,
|
|
69
|
+
): Promise<RunRecord> {
|
|
70
|
+
const { log, runId, threadId, signal } = opts
|
|
71
|
+
await log.open(threadId !== undefined ? { runId, threadId } : { runId })
|
|
72
|
+
if (signal?.aborted) {
|
|
73
|
+
await log.finish(runId, 'aborted')
|
|
74
|
+
return reread(log, runId)
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
try {
|
|
78
|
+
for await (const chunk of stream) {
|
|
79
|
+
if (signal?.aborted) {
|
|
80
|
+
await log.finish(runId, 'aborted')
|
|
81
|
+
return reread(log, runId)
|
|
82
|
+
}
|
|
83
|
+
await log.append(runId, chunk)
|
|
84
|
+
if (isRunErrorChunk(chunk)) {
|
|
85
|
+
await log.finish(runId, 'error', runErrorFromChunk(chunk))
|
|
86
|
+
return reread(log, runId)
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
} catch (error) {
|
|
90
|
+
// Detached run: no caller to throw to. Record the failure in the log so
|
|
91
|
+
// tailing clients observe it, then return — do NOT rethrow.
|
|
92
|
+
const message = messageOf(error)
|
|
93
|
+
await log.append(runId, syntheticRunError(message))
|
|
94
|
+
await log.finish(runId, 'error', { message })
|
|
95
|
+
return reread(log, runId)
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
await log.finish(runId, 'done')
|
|
99
|
+
return reread(log, runId)
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/** Re-read the now-terminal record; the run was just driven, so it must exist. */
|
|
103
|
+
async function reread(log: RunEventLog, runId: string): Promise<RunRecord> {
|
|
104
|
+
const latest = await log.get(runId)
|
|
105
|
+
if (!latest) throw new Error(`run: record for "${runId}" vanished mid-run`)
|
|
106
|
+
return latest
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
export interface RunControllerStartInput {
|
|
110
|
+
runId: string
|
|
111
|
+
threadId?: string
|
|
112
|
+
stream: AsyncIterable<StreamChunk>
|
|
113
|
+
/** Abort consumption mid-stream; the run finishes as `aborted`. */
|
|
114
|
+
signal?: AbortSignal
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
export interface RunHandle {
|
|
118
|
+
runId: string
|
|
119
|
+
/** Resolves with the final record once the run reaches a terminal status. */
|
|
120
|
+
done: Promise<RunRecord>
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Thin orchestration helper over a {@link RunEventLog}: fire-and-track a run via
|
|
125
|
+
* {@link pipeToRunLog}, tail it from a cursor, and `drain()` all in-flight runs
|
|
126
|
+
* (e.g. inside a `ctx.waitUntil`). Holds no run state of its own beyond the set
|
|
127
|
+
* of currently in-flight `done` promises.
|
|
128
|
+
*/
|
|
129
|
+
export class RunController {
|
|
130
|
+
private readonly inFlight = new Set<Promise<RunRecord>>()
|
|
131
|
+
|
|
132
|
+
constructor(private readonly log: RunEventLog) {}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Kick off `pipeToRunLog` without awaiting it and return the `runId`
|
|
136
|
+
* immediately plus a `done` promise the orchestrator may await or detach.
|
|
137
|
+
*/
|
|
138
|
+
start(input: RunControllerStartInput): RunHandle {
|
|
139
|
+
const done = pipeToRunLog(input.stream, {
|
|
140
|
+
log: this.log,
|
|
141
|
+
runId: input.runId,
|
|
142
|
+
...(input.threadId !== undefined ? { threadId: input.threadId } : {}),
|
|
143
|
+
...(input.signal !== undefined ? { signal: input.signal } : {}),
|
|
144
|
+
})
|
|
145
|
+
this.inFlight.add(done)
|
|
146
|
+
void done.finally(() => this.inFlight.delete(done))
|
|
147
|
+
return { runId: input.runId, done }
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/** Resumable client tail — replay from `fromSeq`, then live-tail to terminal. */
|
|
151
|
+
attach(
|
|
152
|
+
runId: string,
|
|
153
|
+
opts?: { fromSeq?: number; signal?: AbortSignal },
|
|
154
|
+
): AsyncIterable<RunEvent> {
|
|
155
|
+
return this.log.read(runId, opts)
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
/** Current run record, or null if the run is unknown. */
|
|
159
|
+
status(runId: string): Promise<RunRecord | null> {
|
|
160
|
+
return this.log.get(runId)
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/** Await every currently in-flight run's `done` promise. */
|
|
164
|
+
async drain(): Promise<void> {
|
|
165
|
+
await Promise.all([...this.inFlight])
|
|
166
|
+
}
|
|
167
|
+
}
|
package/src/runner.ts
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The reusable "run an agent CLI inside a sandbox and stream its events out"
|
|
3
|
+
* primitive. Harness adapters (claude-code, codex, …) spawn their CLI via the
|
|
4
|
+
* uniform {@link SandboxHandle} and consume newline-delimited JSON from stdout,
|
|
5
|
+
* which they then translate into AG-UI StreamChunks.
|
|
6
|
+
*
|
|
7
|
+
* This is intentionally transport-minimal: a stdout NDJSON pipe. Multi-client
|
|
8
|
+
* reconnect / replay belongs to the persistence/EventLog layer, not here.
|
|
9
|
+
*/
|
|
10
|
+
import type { ProcessOptions, SandboxHandle } from './contracts'
|
|
11
|
+
|
|
12
|
+
export interface SpawnNdjsonOptions extends ProcessOptions {
|
|
13
|
+
/**
|
|
14
|
+
* Called for each raw stdout line that is non-empty but fails JSON parsing
|
|
15
|
+
* (e.g. a CLI banner). Defaults to ignoring it. Stderr is never parsed.
|
|
16
|
+
*/
|
|
17
|
+
onNonJsonLine?: (line: string) => void
|
|
18
|
+
/**
|
|
19
|
+
* Written to the process stdin (then stdin is closed) right after spawn —
|
|
20
|
+
* e.g. the agent prompt for `claude -p`. Avoids putting the prompt in argv.
|
|
21
|
+
*/
|
|
22
|
+
input?: string
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/** Split a stream of arbitrary string chunks into complete lines. */
|
|
26
|
+
export async function* toLines(
|
|
27
|
+
chunks: AsyncIterable<string>,
|
|
28
|
+
): AsyncIterable<string> {
|
|
29
|
+
let buffer = ''
|
|
30
|
+
for await (const chunk of chunks) {
|
|
31
|
+
buffer += chunk
|
|
32
|
+
let newlineIndex = buffer.indexOf('\n')
|
|
33
|
+
while (newlineIndex !== -1) {
|
|
34
|
+
const line = buffer.slice(0, newlineIndex)
|
|
35
|
+
buffer = buffer.slice(newlineIndex + 1)
|
|
36
|
+
yield line
|
|
37
|
+
newlineIndex = buffer.indexOf('\n')
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
if (buffer.length > 0) yield buffer
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Spawn `command` in the sandbox and yield each stdout line parsed as JSON.
|
|
45
|
+
* Resolves the spawn handle's exit via `wait()` after stdout closes; a non-zero
|
|
46
|
+
* exit with no events surfaced is the adapter's concern to detect.
|
|
47
|
+
*/
|
|
48
|
+
export async function* spawnNdjson(
|
|
49
|
+
handle: SandboxHandle,
|
|
50
|
+
command: string,
|
|
51
|
+
options: SpawnNdjsonOptions = {},
|
|
52
|
+
): AsyncIterable<unknown> {
|
|
53
|
+
const { onNonJsonLine, input, ...processOptions } = options
|
|
54
|
+
const proc = await handle.process.spawn(command, processOptions)
|
|
55
|
+
|
|
56
|
+
if (input !== undefined) {
|
|
57
|
+
await proc.stdin.write(input)
|
|
58
|
+
await proc.stdin.end()
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
// Drain stderr concurrently. A CLI that fails before producing stdout (a
|
|
62
|
+
// broken install, an auth/permission refusal, …) prints to stderr and exits
|
|
63
|
+
// non-zero; without this, stdout-only parsing yields nothing and the failure
|
|
64
|
+
// vanishes. Capturing it lets us surface the cause below.
|
|
65
|
+
const stderrChunks: Array<string> = []
|
|
66
|
+
const stderrDrained = (async () => {
|
|
67
|
+
try {
|
|
68
|
+
for await (const chunk of proc.stderr) stderrChunks.push(chunk)
|
|
69
|
+
} catch {
|
|
70
|
+
// stderr stream torn down — use whatever was captured
|
|
71
|
+
}
|
|
72
|
+
})()
|
|
73
|
+
|
|
74
|
+
for await (const line of toLines(proc.stdout)) {
|
|
75
|
+
const trimmed = line.trim()
|
|
76
|
+
if (trimmed === '') continue
|
|
77
|
+
let parsed: unknown
|
|
78
|
+
try {
|
|
79
|
+
parsed = JSON.parse(trimmed)
|
|
80
|
+
} catch {
|
|
81
|
+
onNonJsonLine?.(trimmed)
|
|
82
|
+
continue
|
|
83
|
+
}
|
|
84
|
+
yield parsed
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
const exitCode = await proc.wait()
|
|
88
|
+
await stderrDrained
|
|
89
|
+
// A non-zero exit means the agent CLI itself failed. Throw so the adapter's
|
|
90
|
+
// catch turns it into a RUN_ERROR the UI can show, instead of ending the
|
|
91
|
+
// stream silently with no events.
|
|
92
|
+
if (exitCode !== 0) {
|
|
93
|
+
const stderr = stderrChunks.join('').trim()
|
|
94
|
+
throw new Error(
|
|
95
|
+
`Agent process exited with code ${exitCode}` +
|
|
96
|
+
(stderr ? `: ${stderr.slice(0, 1000)}` : ''),
|
|
97
|
+
)
|
|
98
|
+
}
|
|
99
|
+
}
|