@tanstack/ai-sandbox 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/README.md +182 -0
  2. package/dist/esm/agents-file.d.ts +36 -0
  3. package/dist/esm/agents-file.js +44 -0
  4. package/dist/esm/agents-file.js.map +1 -0
  5. package/dist/esm/approvals.d.ts +38 -0
  6. package/dist/esm/approvals.js +36 -0
  7. package/dist/esm/approvals.js.map +1 -0
  8. package/dist/esm/bootstrap.d.ts +17 -0
  9. package/dist/esm/bootstrap.js +124 -0
  10. package/dist/esm/bootstrap.js.map +1 -0
  11. package/dist/esm/bridge-events.d.ts +21 -0
  12. package/dist/esm/bridge-events.js +76 -0
  13. package/dist/esm/bridge-events.js.map +1 -0
  14. package/dist/esm/capabilities.d.ts +26 -0
  15. package/dist/esm/capabilities.js +29 -0
  16. package/dist/esm/capabilities.js.map +1 -0
  17. package/dist/esm/contracts.d.ts +211 -0
  18. package/dist/esm/errors.d.ts +16 -0
  19. package/dist/esm/errors.js +25 -0
  20. package/dist/esm/errors.js.map +1 -0
  21. package/dist/esm/git-exec.d.ts +2 -0
  22. package/dist/esm/git-exec.js +68 -0
  23. package/dist/esm/git-exec.js.map +1 -0
  24. package/dist/esm/harness-cwd.d.ts +2 -0
  25. package/dist/esm/harness-cwd.js +24 -0
  26. package/dist/esm/harness-cwd.js.map +1 -0
  27. package/dist/esm/index.d.ts +39 -0
  28. package/dist/esm/index.js +103 -0
  29. package/dist/esm/index.js.map +1 -0
  30. package/dist/esm/key.d.ts +20 -0
  31. package/dist/esm/key.js +41 -0
  32. package/dist/esm/key.js.map +1 -0
  33. package/dist/esm/middleware.d.ts +5 -0
  34. package/dist/esm/middleware.js +140 -0
  35. package/dist/esm/middleware.js.map +1 -0
  36. package/dist/esm/ngrok.d.ts +16 -0
  37. package/dist/esm/ngrok.js +54 -0
  38. package/dist/esm/ngrok.js.map +1 -0
  39. package/dist/esm/policy.d.ts +47 -0
  40. package/dist/esm/policy.js +44 -0
  41. package/dist/esm/policy.js.map +1 -0
  42. package/dist/esm/projection.d.ts +31 -0
  43. package/dist/esm/projection.js +9 -0
  44. package/dist/esm/projection.js.map +1 -0
  45. package/dist/esm/remote-tools.d.ts +48 -0
  46. package/dist/esm/remote-tools.js +76 -0
  47. package/dist/esm/remote-tools.js.map +1 -0
  48. package/dist/esm/run-log.d.ts +81 -0
  49. package/dist/esm/run-log.js +107 -0
  50. package/dist/esm/run-log.js.map +1 -0
  51. package/dist/esm/run.d.ts +58 -0
  52. package/dist/esm/run.js +89 -0
  53. package/dist/esm/run.js.map +1 -0
  54. package/dist/esm/runner.d.ts +21 -0
  55. package/dist/esm/runner.js +54 -0
  56. package/dist/esm/runner.js.map +1 -0
  57. package/dist/esm/sandbox.d.ts +79 -0
  58. package/dist/esm/sandbox.js +125 -0
  59. package/dist/esm/sandbox.js.map +1 -0
  60. package/dist/esm/secrets.d.ts +37 -0
  61. package/dist/esm/secrets.js +59 -0
  62. package/dist/esm/secrets.js.map +1 -0
  63. package/dist/esm/setup-plan.d.ts +13 -0
  64. package/dist/esm/setup-plan.js +16 -0
  65. package/dist/esm/setup-plan.js.map +1 -0
  66. package/dist/esm/shell.d.ts +45 -0
  67. package/dist/esm/shell.js +164 -0
  68. package/dist/esm/shell.js.map +1 -0
  69. package/dist/esm/store.d.ts +53 -0
  70. package/dist/esm/store.js +34 -0
  71. package/dist/esm/store.js.map +1 -0
  72. package/dist/esm/tool-bridge.d.ts +130 -0
  73. package/dist/esm/tool-bridge.js +197 -0
  74. package/dist/esm/tool-bridge.js.map +1 -0
  75. package/dist/esm/watch.d.ts +36 -0
  76. package/dist/esm/watch.js +144 -0
  77. package/dist/esm/watch.js.map +1 -0
  78. package/dist/esm/workspace.d.ts +128 -0
  79. package/dist/esm/workspace.js +42 -0
  80. package/dist/esm/workspace.js.map +1 -0
  81. package/package.json +72 -0
  82. package/skills/ai-sandbox/SKILL.md +366 -0
  83. package/src/agents-file.ts +101 -0
  84. package/src/approvals.ts +96 -0
  85. package/src/bootstrap.ts +196 -0
  86. package/src/bridge-events.ts +112 -0
  87. package/src/capabilities.ts +47 -0
  88. package/src/contracts.ts +236 -0
  89. package/src/errors.ts +31 -0
  90. package/src/git-exec.ts +114 -0
  91. package/src/harness-cwd.ts +38 -0
  92. package/src/index.ts +222 -0
  93. package/src/key.ts +70 -0
  94. package/src/middleware.ts +233 -0
  95. package/src/ngrok.ts +85 -0
  96. package/src/policy.ts +111 -0
  97. package/src/projection.ts +46 -0
  98. package/src/remote-tools.ts +180 -0
  99. package/src/run-log.ts +224 -0
  100. package/src/run.ts +167 -0
  101. package/src/runner.ts +99 -0
  102. package/src/sandbox.ts +259 -0
  103. package/src/secrets.ts +101 -0
  104. package/src/setup-plan.ts +25 -0
  105. package/src/shell.ts +288 -0
  106. package/src/store.ts +83 -0
  107. package/src/tool-bridge.ts +399 -0
  108. package/src/watch.ts +256 -0
  109. package/src/workspace.ts +151 -0
@@ -0,0 +1,180 @@
1
+ /**
2
+ * Host-tool delegation for the CO-LOCATED ("combined") sandbox model.
3
+ *
4
+ * In the co-located model the harness loop AND its MCP tool-bridge run INSIDE
5
+ * the container (the in-container sandbox is just `local-process`, so the
6
+ * existing adapter + `nodeHttpBridgeProvisioner` serve the bridge on the
7
+ * container's own `localhost` with native stdin — nothing new there). The one
8
+ * thing that still must cross the container→orchestrator boundary is the
9
+ * **execution** of `chat()`-provided server tools: their `execute()` closures
10
+ * (DB / secrets / app state) live in the orchestrator, not the container.
11
+ *
12
+ * This module is that narrow seam:
13
+ * - {@link remoteToolStubs} (container side) rebuilds `chat()` tools from
14
+ * serialized {@link ToolDescriptor}s; each stub's `execute` delegates to a
15
+ * {@link RemoteToolExecutor} instead of running locally. The adapter bridges
16
+ * these stubs exactly like real tools.
17
+ * - {@link httpRemoteToolExecutor} (container side) is the default executor: it
18
+ * POSTs `{ name, args }` to the orchestrator's tool-exec endpoint.
19
+ * - {@link executeHostTool} (orchestrator side) runs the REAL tool and returns
20
+ * its raw result — the only host code the container can reach.
21
+ *
22
+ * So the public network surface shrinks from "the whole MCP protocol" (served
23
+ * from the orchestrator in the DO-drives-container model) to "one authenticated
24
+ * tool-exec call" — the MCP transport itself never leaves the container.
25
+ */
26
+ import type { AnyTool } from '@tanstack/ai'
27
+ import type { ToolDescriptor } from './tool-bridge'
28
+
29
+ /** Per-call options forwarded to a {@link RemoteToolExecutor}. */
30
+ export interface RemoteToolExecuteOptions {
31
+ /** Cancels the in-flight remote call when the in-container run aborts. */
32
+ signal?: AbortSignal
33
+ }
34
+
35
+ /** Runs a named host tool with the given args, returning its raw result. */
36
+ export interface RemoteToolExecutor {
37
+ execute: (
38
+ name: string,
39
+ args: unknown,
40
+ options?: RemoteToolExecuteOptions,
41
+ ) => Promise<unknown>
42
+ }
43
+
44
+ /** Wire shape of a tool-exec request the container POSTs to the orchestrator. */
45
+ export interface ToolExecRequest {
46
+ name: string
47
+ args: unknown
48
+ }
49
+
50
+ /** Narrow an unknown body into a {@link ToolExecRequest} (project rule: no `as`). */
51
+ export function isToolExecRequest(value: unknown): value is ToolExecRequest {
52
+ return (
53
+ value !== null &&
54
+ typeof value === 'object' &&
55
+ 'name' in value &&
56
+ typeof value.name === 'string'
57
+ )
58
+ }
59
+
60
+ /**
61
+ * Rebuild `chat()` tool objects (container side) from serialized descriptors.
62
+ * Each stub advertises the descriptor's JSON-schema and delegates `execute` to
63
+ * the executor; the harness adapter bridges them like any other tool. The
64
+ * harness's `abortSignal` is forwarded so a cancelled run cancels the in-flight
65
+ * remote call too.
66
+ */
67
+ export function remoteToolStubs(
68
+ descriptors: Array<ToolDescriptor>,
69
+ executor: RemoteToolExecutor,
70
+ ): Array<AnyTool> {
71
+ return descriptors.map((descriptor) => ({
72
+ name: descriptor.name,
73
+ description: descriptor.description ?? '',
74
+ inputSchema: descriptor.inputSchema,
75
+ execute: (args: unknown, options?: { abortSignal?: AbortSignal }) =>
76
+ executor.execute(
77
+ descriptor.name,
78
+ args,
79
+ options?.abortSignal !== undefined
80
+ ? { signal: options.abortSignal }
81
+ : {},
82
+ ),
83
+ }))
84
+ }
85
+
86
+ /**
87
+ * Serialize `chat()` tools to wire descriptors to send into the container.
88
+ * `inputSchema` must already be a plain JSON-schema object (convert Standard
89
+ * Schemas before calling, the same way harness adapters advertise tools).
90
+ */
91
+ export function toolDescriptors(tools: Array<AnyTool>): Array<ToolDescriptor> {
92
+ return tools.map((tool) => ({
93
+ name: tool.name,
94
+ description: tool.description,
95
+ inputSchema: isJsonSchemaObject(tool.inputSchema)
96
+ ? tool.inputSchema
97
+ : { type: 'object', properties: {} },
98
+ }))
99
+ }
100
+
101
+ function isJsonSchemaObject(
102
+ value: unknown,
103
+ ): value is { type: 'object'; [key: string]: unknown } {
104
+ return (
105
+ value !== null &&
106
+ typeof value === 'object' &&
107
+ 'type' in value &&
108
+ (value as { type?: unknown }).type === 'object'
109
+ )
110
+ }
111
+
112
+ /** Wire shape of a tool-exec response from the orchestrator. */
113
+ interface ToolExecResponse {
114
+ result: unknown
115
+ }
116
+
117
+ function isToolExecResponse(value: unknown): value is ToolExecResponse {
118
+ return value !== null && typeof value === 'object' && 'result' in value
119
+ }
120
+
121
+ /**
122
+ * The default {@link RemoteToolExecutor}: POST `{ name, args }` (bearer-gated)
123
+ * to the orchestrator's tool-exec endpoint and return its `result`. A non-2xx
124
+ * or malformed response throws (surfaced to the agent as a failed tool call by
125
+ * the bridge) — never silently swallowed.
126
+ */
127
+ export function httpRemoteToolExecutor(
128
+ url: string,
129
+ token: string,
130
+ ): RemoteToolExecutor {
131
+ return {
132
+ async execute(name, args, options) {
133
+ const res = await fetch(url, {
134
+ method: 'POST',
135
+ headers: {
136
+ 'content-type': 'application/json',
137
+ authorization: `Bearer ${token}`,
138
+ },
139
+ body: JSON.stringify({ name, args }),
140
+ ...(options?.signal !== undefined ? { signal: options.signal } : {}),
141
+ })
142
+ if (!res.ok) {
143
+ const text = await res.text()
144
+ throw new Error(
145
+ `remote tool "${name}" failed: ${res.status} ${text.slice(0, 200)}`,
146
+ )
147
+ }
148
+ const body: unknown = await res.json()
149
+ if (!isToolExecResponse(body)) {
150
+ throw new Error(
151
+ `remote tool "${name}": malformed orchestrator response`,
152
+ )
153
+ }
154
+ return body.result
155
+ },
156
+ }
157
+ }
158
+
159
+ /**
160
+ * Run a host tool by name with the given args, returning its raw result
161
+ * (orchestrator side of {@link httpRemoteToolExecutor}). Throws for an unknown
162
+ * tool or one with no `execute` — the orchestrator surfaces that as a 4xx/5xx.
163
+ */
164
+ export function executeHostTool(
165
+ tools: Array<AnyTool>,
166
+ name: string,
167
+ args: unknown,
168
+ options: { context?: unknown; signal?: AbortSignal } = {},
169
+ ): Promise<unknown> {
170
+ const tool = tools.find((candidate) => candidate.name === name)
171
+ if (!tool?.execute) {
172
+ return Promise.reject(new Error(`Unknown tool: ${name}`))
173
+ }
174
+ return Promise.resolve(
175
+ tool.execute(args ?? {}, {
176
+ context: options.context,
177
+ abortSignal: options.signal,
178
+ }),
179
+ )
180
+ }
package/src/run-log.ts ADDED
@@ -0,0 +1,224 @@
1
+ /**
2
+ * Resumable run event-log — the primitive that lets a trigger (e.g. a
3
+ * Cloudflare Worker) start an agent run and return immediately while a durable
4
+ * orchestrator (e.g. a Durable Object) drives the run and persists every
5
+ * emitted {@link StreamChunk} under a monotonic `seq`.
6
+ *
7
+ * Clients tail the log from a cursor (`fromSeq`), so a dropped connection, a new
8
+ * browser tab, or an orchestrator that hibernated between chunks all reconnect
9
+ * cleanly: replay everything after the client's last-seen `seq`, then live-tail
10
+ * until the run reaches a terminal status. The *run* never depends on any single
11
+ * connection staying open — that is what makes the serverless/edge model work.
12
+ *
13
+ * This module is transport- and storage-agnostic. {@link InMemoryRunEventLog} is
14
+ * the default (single-process / tests); a durable backend (DO storage, KV, SQL)
15
+ * implements the same {@link RunEventLog} interface — see the Cloudflare example.
16
+ */
17
+ import type { StreamChunk } from '@tanstack/ai'
18
+
19
+ /** A terminal run status: no further events will be appended. */
20
+ export type TerminalRunStatus = 'done' | 'error' | 'aborted'
21
+
22
+ /** Lifecycle status of a run. `done`/`error`/`aborted` are terminal. */
23
+ export type RunStatus = 'running' | TerminalRunStatus
24
+
25
+ const TERMINAL: ReadonlySet<RunStatus> = new Set<RunStatus>([
26
+ 'done',
27
+ 'error',
28
+ 'aborted',
29
+ ])
30
+
31
+ /** Whether a run status is terminal (no further events will be appended). */
32
+ export function isTerminalRunStatus(status: RunStatus): boolean {
33
+ return TERMINAL.has(status)
34
+ }
35
+
36
+ export interface RunError {
37
+ message: string
38
+ code?: string
39
+ }
40
+
41
+ /** Durable bookkeeping for a single run. */
42
+ export interface RunRecord {
43
+ runId: string
44
+ threadId?: string
45
+ status: RunStatus
46
+ /** Seq of the last appended event, or `-1` when no events yet. */
47
+ lastSeq: number
48
+ error?: RunError
49
+ createdAt: number
50
+ updatedAt: number
51
+ }
52
+
53
+ /** One persisted event: a chunk plus its monotonic, gap-free sequence number. */
54
+ export interface RunEvent {
55
+ seq: number
56
+ chunk: StreamChunk
57
+ }
58
+
59
+ export interface RunEventLogReadOptions {
60
+ /**
61
+ * Exclusive cursor: only events with `seq > fromSeq` are yielded. Pass the
62
+ * client's last-seen `seq` to resume; omit (or `-1`) to replay from the start.
63
+ */
64
+ fromSeq?: number
65
+ /** Stop tailing when this fires (e.g. the client disconnected). */
66
+ signal?: AbortSignal
67
+ }
68
+
69
+ /**
70
+ * Append-only, `seq`-indexed log of a run's stream, with resumable reads.
71
+ *
72
+ * Contract:
73
+ * - `append` assigns the next `seq` (0, 1, 2, …) and returns it.
74
+ * - `read` yields the backlog after `fromSeq` in order, then live-tails new
75
+ * events, and RETURNS once the run is terminal and the cursor has caught up.
76
+ * - All methods reject for an unknown `runId` except `get`, which resolves null.
77
+ */
78
+ export interface RunEventLog {
79
+ /** Idempotently create (or return) the run record. */
80
+ open: (input: { runId: string; threadId?: string }) => Promise<RunRecord>
81
+ /** Append one chunk; resolves with its assigned `seq`. */
82
+ append: (runId: string, chunk: StreamChunk) => Promise<number>
83
+ /** Move the run to a terminal status. Idempotent for the same status. */
84
+ finish: (
85
+ runId: string,
86
+ status: TerminalRunStatus,
87
+ error?: RunError,
88
+ ) => Promise<void>
89
+ /** Current record, or null if the run is unknown. */
90
+ get: (runId: string) => Promise<RunRecord | null>
91
+ /** Replay-then-tail events with `seq > fromSeq` until the run is terminal. */
92
+ read: (
93
+ runId: string,
94
+ options?: RunEventLogReadOptions,
95
+ ) => AsyncIterable<RunEvent>
96
+ }
97
+
98
+ /** Per-run state for the in-memory log. */
99
+ interface RunState {
100
+ record: RunRecord
101
+ chunks: Array<StreamChunk>
102
+ /** Resolved (and cleared) whenever an event is appended or status changes. */
103
+ waiters: Set<() => void>
104
+ }
105
+
106
+ /**
107
+ * Single-process {@link RunEventLog}. Backs `read`'s live-tail with an internal
108
+ * waiter set: `append`/`finish` wake every blocked reader. Suitable for a
109
+ * long-running Node host, tests, and as the reference implementation a durable
110
+ * backend mirrors.
111
+ */
112
+ export class InMemoryRunEventLog implements RunEventLog {
113
+ private readonly runs = new Map<string, RunState>()
114
+
115
+ private now(): number {
116
+ return Date.now()
117
+ }
118
+
119
+ private require(runId: string): RunState {
120
+ const state = this.runs.get(runId)
121
+ if (!state) throw new Error(`run-log: unknown runId "${runId}"`)
122
+ return state
123
+ }
124
+
125
+ private wake(state: RunState): void {
126
+ const waiters = [...state.waiters]
127
+ state.waiters.clear()
128
+ for (const resolve of waiters) resolve()
129
+ }
130
+
131
+ // Mutators return a Promise without `async` so contract violations REJECT
132
+ // (rather than throwing synchronously from a Promise-typed method — a
133
+ // `.catch()` footgun) without an `await`-less async body.
134
+ open(input: { runId: string; threadId?: string }): Promise<RunRecord> {
135
+ const existing = this.runs.get(input.runId)
136
+ if (existing) return Promise.resolve({ ...existing.record })
137
+ const now = this.now()
138
+ const record: RunRecord = {
139
+ runId: input.runId,
140
+ ...(input.threadId !== undefined ? { threadId: input.threadId } : {}),
141
+ status: 'running',
142
+ lastSeq: -1,
143
+ createdAt: now,
144
+ updatedAt: now,
145
+ }
146
+ this.runs.set(input.runId, { record, chunks: [], waiters: new Set() })
147
+ return Promise.resolve({ ...record })
148
+ }
149
+
150
+ append(runId: string, chunk: StreamChunk): Promise<number> {
151
+ const state = this.runs.get(runId)
152
+ if (!state) {
153
+ return Promise.reject(new Error(`run-log: unknown runId "${runId}"`))
154
+ }
155
+ if (isTerminalRunStatus(state.record.status)) {
156
+ return Promise.reject(
157
+ new Error(
158
+ `run-log: cannot append to terminal run "${runId}" (status=${state.record.status})`,
159
+ ),
160
+ )
161
+ }
162
+ // Derive seq from the record's cursor (not `chunks.length`) so the gap-free
163
+ // invariant holds the same way the durable backend computes it, even if the
164
+ // backlog is ever trimmed/compacted.
165
+ const seq = state.record.lastSeq + 1
166
+ state.chunks.push(chunk)
167
+ state.record.lastSeq = seq
168
+ state.record.updatedAt = this.now()
169
+ this.wake(state)
170
+ return Promise.resolve(seq)
171
+ }
172
+
173
+ finish(
174
+ runId: string,
175
+ status: TerminalRunStatus,
176
+ error?: RunError,
177
+ ): Promise<void> {
178
+ const state = this.runs.get(runId)
179
+ if (!state) {
180
+ return Promise.reject(new Error(`run-log: unknown runId "${runId}"`))
181
+ }
182
+ if (isTerminalRunStatus(state.record.status)) return Promise.resolve()
183
+ state.record.status = status
184
+ if (error !== undefined) state.record.error = error
185
+ state.record.updatedAt = this.now()
186
+ this.wake(state)
187
+ return Promise.resolve()
188
+ }
189
+
190
+ get(runId: string): Promise<RunRecord | null> {
191
+ const state = this.runs.get(runId)
192
+ return Promise.resolve(state ? { ...state.record } : null)
193
+ }
194
+
195
+ async *read(
196
+ runId: string,
197
+ options?: RunEventLogReadOptions,
198
+ ): AsyncIterable<RunEvent> {
199
+ const state = this.require(runId)
200
+ const signal = options?.signal
201
+ let cursor = options?.fromSeq ?? -1
202
+ while (!signal?.aborted) {
203
+ while (cursor < state.record.lastSeq) {
204
+ cursor += 1
205
+ const chunk = state.chunks[cursor]
206
+ if (chunk !== undefined) yield { seq: cursor, chunk }
207
+ }
208
+ if (isTerminalRunStatus(state.record.status)) return
209
+ await this.waitForChange(state, signal)
210
+ }
211
+ }
212
+
213
+ private waitForChange(state: RunState, signal?: AbortSignal): Promise<void> {
214
+ return new Promise<void>((resolve) => {
215
+ const wake = (): void => {
216
+ state.waiters.delete(wake)
217
+ if (signal) signal.removeEventListener('abort', wake)
218
+ resolve()
219
+ }
220
+ state.waiters.add(wake)
221
+ if (signal) signal.addEventListener('abort', wake, { once: true })
222
+ })
223
+ }
224
+ }
package/src/run.ts ADDED
@@ -0,0 +1,167 @@
1
+ /**
2
+ * The "run driver" for the inverted/serverless sandbox model: pump a `chat()`
3
+ * stream into a {@link RunEventLog} so a trigger can return immediately while a
4
+ * durable orchestrator drives the run and clients tail from a cursor.
5
+ *
6
+ * The key inversion vs. a classic request/response handler: there is no caller
7
+ * holding the stream open, so nothing to throw an error *back to*. The log is
8
+ * the only channel — every chunk (including a terminal {@link EventType.RUN_ERROR})
9
+ * is persisted under a `seq`, and a thrown stream error is recorded as a
10
+ * synthesized `RUN_ERROR` event plus the record's `error` field. Tailing clients
11
+ * therefore always observe failures; {@link pipeToRunLog} never rejects.
12
+ */
13
+ import { EventType } from '@tanstack/ai'
14
+ import type { StreamChunk } from '@tanstack/ai'
15
+ import type { RunError, RunEvent, RunEventLog, RunRecord } from './run-log'
16
+
17
+ /** Whether a chunk is the terminal error event the chat engine emits. */
18
+ function isRunErrorChunk(
19
+ chunk: StreamChunk,
20
+ ): chunk is StreamChunk & { message: string; code?: string } {
21
+ return chunk.type === EventType.RUN_ERROR
22
+ }
23
+
24
+ /** Pull `{ message, code }` off a RUN_ERROR chunk for the run record. */
25
+ function runErrorFromChunk(
26
+ chunk: StreamChunk & { message: string; code?: string },
27
+ ): RunError {
28
+ return chunk.code !== undefined
29
+ ? { message: chunk.message, code: chunk.code }
30
+ : { message: chunk.message }
31
+ }
32
+
33
+ /** Render an unknown thrown value as a stable error message. */
34
+ function messageOf(error: unknown): string {
35
+ return error instanceof Error ? error.message : String(error)
36
+ }
37
+
38
+ /** Build the synthetic RUN_ERROR chunk appended when the stream throws. */
39
+ function syntheticRunError(message: string): StreamChunk {
40
+ const chunk: { type: EventType.RUN_ERROR; message: string } = {
41
+ type: EventType.RUN_ERROR,
42
+ message,
43
+ }
44
+ return chunk
45
+ }
46
+
47
+ export interface PipeToRunLogOptions {
48
+ log: RunEventLog
49
+ runId: string
50
+ threadId?: string
51
+ /** Abort consumption mid-stream; the run finishes as `aborted`. */
52
+ signal?: AbortSignal
53
+ }
54
+
55
+ /**
56
+ * Open the run, append every chunk from `stream`, and finish with the right
57
+ * terminal status. Resolves with the final {@link RunRecord} and never rejects:
58
+ * a thrown stream error is surfaced as a `RUN_ERROR` event + the record's
59
+ * `error`, which is what tailing clients see.
60
+ *
61
+ * - normal completion → `finish('done')`
62
+ * - a `RUN_ERROR` chunk → append it, then `finish('error', { message, code })`
63
+ * - the stream throws → append a synthesized `RUN_ERROR`, then `finish('error')`
64
+ * - `signal` aborts mid-stream → stop consuming, `finish('aborted')`
65
+ */
66
+ export async function pipeToRunLog(
67
+ stream: AsyncIterable<StreamChunk>,
68
+ opts: PipeToRunLogOptions,
69
+ ): Promise<RunRecord> {
70
+ const { log, runId, threadId, signal } = opts
71
+ await log.open(threadId !== undefined ? { runId, threadId } : { runId })
72
+ if (signal?.aborted) {
73
+ await log.finish(runId, 'aborted')
74
+ return reread(log, runId)
75
+ }
76
+
77
+ try {
78
+ for await (const chunk of stream) {
79
+ if (signal?.aborted) {
80
+ await log.finish(runId, 'aborted')
81
+ return reread(log, runId)
82
+ }
83
+ await log.append(runId, chunk)
84
+ if (isRunErrorChunk(chunk)) {
85
+ await log.finish(runId, 'error', runErrorFromChunk(chunk))
86
+ return reread(log, runId)
87
+ }
88
+ }
89
+ } catch (error) {
90
+ // Detached run: no caller to throw to. Record the failure in the log so
91
+ // tailing clients observe it, then return — do NOT rethrow.
92
+ const message = messageOf(error)
93
+ await log.append(runId, syntheticRunError(message))
94
+ await log.finish(runId, 'error', { message })
95
+ return reread(log, runId)
96
+ }
97
+
98
+ await log.finish(runId, 'done')
99
+ return reread(log, runId)
100
+ }
101
+
102
+ /** Re-read the now-terminal record; the run was just driven, so it must exist. */
103
+ async function reread(log: RunEventLog, runId: string): Promise<RunRecord> {
104
+ const latest = await log.get(runId)
105
+ if (!latest) throw new Error(`run: record for "${runId}" vanished mid-run`)
106
+ return latest
107
+ }
108
+
109
+ export interface RunControllerStartInput {
110
+ runId: string
111
+ threadId?: string
112
+ stream: AsyncIterable<StreamChunk>
113
+ /** Abort consumption mid-stream; the run finishes as `aborted`. */
114
+ signal?: AbortSignal
115
+ }
116
+
117
+ export interface RunHandle {
118
+ runId: string
119
+ /** Resolves with the final record once the run reaches a terminal status. */
120
+ done: Promise<RunRecord>
121
+ }
122
+
123
+ /**
124
+ * Thin orchestration helper over a {@link RunEventLog}: fire-and-track a run via
125
+ * {@link pipeToRunLog}, tail it from a cursor, and `drain()` all in-flight runs
126
+ * (e.g. inside a `ctx.waitUntil`). Holds no run state of its own beyond the set
127
+ * of currently in-flight `done` promises.
128
+ */
129
+ export class RunController {
130
+ private readonly inFlight = new Set<Promise<RunRecord>>()
131
+
132
+ constructor(private readonly log: RunEventLog) {}
133
+
134
+ /**
135
+ * Kick off `pipeToRunLog` without awaiting it and return the `runId`
136
+ * immediately plus a `done` promise the orchestrator may await or detach.
137
+ */
138
+ start(input: RunControllerStartInput): RunHandle {
139
+ const done = pipeToRunLog(input.stream, {
140
+ log: this.log,
141
+ runId: input.runId,
142
+ ...(input.threadId !== undefined ? { threadId: input.threadId } : {}),
143
+ ...(input.signal !== undefined ? { signal: input.signal } : {}),
144
+ })
145
+ this.inFlight.add(done)
146
+ void done.finally(() => this.inFlight.delete(done))
147
+ return { runId: input.runId, done }
148
+ }
149
+
150
+ /** Resumable client tail — replay from `fromSeq`, then live-tail to terminal. */
151
+ attach(
152
+ runId: string,
153
+ opts?: { fromSeq?: number; signal?: AbortSignal },
154
+ ): AsyncIterable<RunEvent> {
155
+ return this.log.read(runId, opts)
156
+ }
157
+
158
+ /** Current run record, or null if the run is unknown. */
159
+ status(runId: string): Promise<RunRecord | null> {
160
+ return this.log.get(runId)
161
+ }
162
+
163
+ /** Await every currently in-flight run's `done` promise. */
164
+ async drain(): Promise<void> {
165
+ await Promise.all([...this.inFlight])
166
+ }
167
+ }
package/src/runner.ts ADDED
@@ -0,0 +1,99 @@
1
+ /**
2
+ * The reusable "run an agent CLI inside a sandbox and stream its events out"
3
+ * primitive. Harness adapters (claude-code, codex, …) spawn their CLI via the
4
+ * uniform {@link SandboxHandle} and consume newline-delimited JSON from stdout,
5
+ * which they then translate into AG-UI StreamChunks.
6
+ *
7
+ * This is intentionally transport-minimal: a stdout NDJSON pipe. Multi-client
8
+ * reconnect / replay belongs to the persistence/EventLog layer, not here.
9
+ */
10
+ import type { ProcessOptions, SandboxHandle } from './contracts'
11
+
12
+ export interface SpawnNdjsonOptions extends ProcessOptions {
13
+ /**
14
+ * Called for each raw stdout line that is non-empty but fails JSON parsing
15
+ * (e.g. a CLI banner). Defaults to ignoring it. Stderr is never parsed.
16
+ */
17
+ onNonJsonLine?: (line: string) => void
18
+ /**
19
+ * Written to the process stdin (then stdin is closed) right after spawn —
20
+ * e.g. the agent prompt for `claude -p`. Avoids putting the prompt in argv.
21
+ */
22
+ input?: string
23
+ }
24
+
25
+ /** Split a stream of arbitrary string chunks into complete lines. */
26
+ export async function* toLines(
27
+ chunks: AsyncIterable<string>,
28
+ ): AsyncIterable<string> {
29
+ let buffer = ''
30
+ for await (const chunk of chunks) {
31
+ buffer += chunk
32
+ let newlineIndex = buffer.indexOf('\n')
33
+ while (newlineIndex !== -1) {
34
+ const line = buffer.slice(0, newlineIndex)
35
+ buffer = buffer.slice(newlineIndex + 1)
36
+ yield line
37
+ newlineIndex = buffer.indexOf('\n')
38
+ }
39
+ }
40
+ if (buffer.length > 0) yield buffer
41
+ }
42
+
43
+ /**
44
+ * Spawn `command` in the sandbox and yield each stdout line parsed as JSON.
45
+ * Resolves the spawn handle's exit via `wait()` after stdout closes; a non-zero
46
+ * exit with no events surfaced is the adapter's concern to detect.
47
+ */
48
+ export async function* spawnNdjson(
49
+ handle: SandboxHandle,
50
+ command: string,
51
+ options: SpawnNdjsonOptions = {},
52
+ ): AsyncIterable<unknown> {
53
+ const { onNonJsonLine, input, ...processOptions } = options
54
+ const proc = await handle.process.spawn(command, processOptions)
55
+
56
+ if (input !== undefined) {
57
+ await proc.stdin.write(input)
58
+ await proc.stdin.end()
59
+ }
60
+
61
+ // Drain stderr concurrently. A CLI that fails before producing stdout (a
62
+ // broken install, an auth/permission refusal, …) prints to stderr and exits
63
+ // non-zero; without this, stdout-only parsing yields nothing and the failure
64
+ // vanishes. Capturing it lets us surface the cause below.
65
+ const stderrChunks: Array<string> = []
66
+ const stderrDrained = (async () => {
67
+ try {
68
+ for await (const chunk of proc.stderr) stderrChunks.push(chunk)
69
+ } catch {
70
+ // stderr stream torn down — use whatever was captured
71
+ }
72
+ })()
73
+
74
+ for await (const line of toLines(proc.stdout)) {
75
+ const trimmed = line.trim()
76
+ if (trimmed === '') continue
77
+ let parsed: unknown
78
+ try {
79
+ parsed = JSON.parse(trimmed)
80
+ } catch {
81
+ onNonJsonLine?.(trimmed)
82
+ continue
83
+ }
84
+ yield parsed
85
+ }
86
+
87
+ const exitCode = await proc.wait()
88
+ await stderrDrained
89
+ // A non-zero exit means the agent CLI itself failed. Throw so the adapter's
90
+ // catch turns it into a RUN_ERROR the UI can show, instead of ending the
91
+ // stream silently with no events.
92
+ if (exitCode !== 0) {
93
+ const stderr = stderrChunks.join('').trim()
94
+ throw new Error(
95
+ `Agent process exited with code ${exitCode}` +
96
+ (stderr ? `: ${stderr.slice(0, 1000)}` : ''),
97
+ )
98
+ }
99
+ }