@tanstack/ai-sandbox 0.2.4 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/agents-file.js +53 -34
- package/dist/esm/agents-file.js.map +1 -1
- package/dist/esm/align.d.ts +121 -0
- package/dist/esm/align.js +197 -0
- package/dist/esm/align.js.map +1 -0
- package/dist/esm/approvals.js +63 -29
- package/dist/esm/approvals.js.map +1 -1
- package/dist/esm/attach-preflight.d.ts +85 -0
- package/dist/esm/attach-preflight.js +189 -0
- package/dist/esm/attach-preflight.js.map +1 -0
- package/dist/esm/bootstrap.js +103 -117
- package/dist/esm/bootstrap.js.map +1 -1
- package/dist/esm/bridge-events.js +96 -71
- package/dist/esm/bridge-events.js.map +1 -1
- package/dist/esm/capabilities.d.ts +0 -5
- package/dist/esm/capabilities.js +32 -28
- package/dist/esm/capabilities.js.map +1 -1
- package/dist/esm/chunk-identity.d.ts +52 -0
- package/dist/esm/chunk-identity.js +102 -0
- package/dist/esm/chunk-identity.js.map +1 -0
- package/dist/esm/claim.d.ts +187 -0
- package/dist/esm/claim.js +349 -0
- package/dist/esm/claim.js.map +1 -0
- package/dist/esm/contracts.d.ts +13 -0
- package/dist/esm/driver.d.ts +83 -0
- package/dist/esm/driver.js +138 -0
- package/dist/esm/driver.js.map +1 -0
- package/dist/esm/durability.d.ts +263 -0
- package/dist/esm/durability.js +230 -0
- package/dist/esm/durability.js.map +1 -0
- package/dist/esm/errors.js +28 -24
- package/dist/esm/errors.js.map +1 -1
- package/dist/esm/file-diff.js +151 -135
- package/dist/esm/file-diff.js.map +1 -1
- package/dist/esm/git-exec.js +51 -62
- package/dist/esm/git-exec.js.map +1 -1
- package/dist/esm/harness-cwd.js +24 -19
- package/dist/esm/harness-cwd.js.map +1 -1
- package/dist/esm/index.d.ts +30 -8
- package/dist/esm/index.js +23 -91
- package/dist/esm/instance-store.d.ts +88 -0
- package/dist/esm/instance-store.js +67 -0
- package/dist/esm/instance-store.js.map +1 -0
- package/dist/esm/journal-bytes.d.ts +67 -0
- package/dist/esm/journal-bytes.js +110 -0
- package/dist/esm/journal-bytes.js.map +1 -0
- package/dist/esm/journal-reader.d.ts +66 -0
- package/dist/esm/journal-reader.js +228 -0
- package/dist/esm/journal-reader.js.map +1 -0
- package/dist/esm/journal-sweep.d.ts +113 -0
- package/dist/esm/journal-sweep.js +309 -0
- package/dist/esm/journal-sweep.js.map +1 -0
- package/dist/esm/journal.d.ts +542 -0
- package/dist/esm/journal.js +679 -0
- package/dist/esm/journal.js.map +1 -0
- package/dist/esm/key.js +36 -33
- package/dist/esm/key.js.map +1 -1
- package/dist/esm/middleware.d.ts +50 -2
- package/dist/esm/middleware.js +335 -208
- package/dist/esm/middleware.js.map +1 -1
- package/dist/esm/ngrok.js +75 -49
- package/dist/esm/ngrok.js.map +1 -1
- package/dist/esm/policy.js +43 -34
- package/dist/esm/policy.js.map +1 -1
- package/dist/esm/projection.js +16 -8
- package/dist/esm/projection.js.map +1 -1
- package/dist/esm/reap.d.ts +238 -0
- package/dist/esm/reap.js +355 -0
- package/dist/esm/reap.js.map +1 -0
- package/dist/esm/reclaim.d.ts +84 -0
- package/dist/esm/reclaim.js +106 -0
- package/dist/esm/reclaim.js.map +1 -0
- package/dist/esm/remote-tools.js +73 -62
- package/dist/esm/remote-tools.js.map +1 -1
- package/dist/esm/run.d.ts +93 -25
- package/dist/esm/run.js +274 -79
- package/dist/esm/run.js.map +1 -1
- package/dist/esm/runner.d.ts +119 -2
- package/dist/esm/runner.js +270 -51
- package/dist/esm/runner.js.map +1 -1
- package/dist/esm/sandbox.d.ts +3 -2
- package/dist/esm/sandbox.js +139 -123
- package/dist/esm/sandbox.js.map +1 -1
- package/dist/esm/secrets.js +39 -47
- package/dist/esm/secrets.js.map +1 -1
- package/dist/esm/setup-plan.js +22 -14
- package/dist/esm/setup-plan.js.map +1 -1
- package/dist/esm/shell.d.ts +8 -0
- package/dist/esm/shell.js +197 -158
- package/dist/esm/shell.js.map +1 -1
- package/dist/esm/testkit/conformance.d.ts +16 -0
- package/dist/esm/testkit/conformance.js +97 -0
- package/dist/esm/testkit/conformance.js.map +1 -0
- package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
- package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
- package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
- package/dist/esm/testkit/journal-conformance.d.ts +51 -0
- package/dist/esm/testkit/journal-conformance.js +378 -0
- package/dist/esm/testkit/journal-conformance.js.map +1 -0
- package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
- package/dist/esm/testkit/reaper-conformance.js +847 -0
- package/dist/esm/testkit/reaper-conformance.js.map +1 -0
- package/dist/esm/testkit/shell-spawn.d.ts +2 -0
- package/dist/esm/testkit/shell-spawn.js +60 -0
- package/dist/esm/testkit/shell-spawn.js.map +1 -0
- package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
- package/dist/esm/testkit/takeover-conformance.js +685 -0
- package/dist/esm/testkit/takeover-conformance.js.map +1 -0
- package/dist/esm/tool-bridge.js +227 -180
- package/dist/esm/tool-bridge.js.map +1 -1
- package/dist/esm/tool-history.d.ts +62 -0
- package/dist/esm/tool-history.js +171 -0
- package/dist/esm/tool-history.js.map +1 -0
- package/dist/esm/watch.js +310 -236
- package/dist/esm/watch.js.map +1 -1
- package/dist/esm/workspace.d.ts +1 -1
- package/dist/esm/workspace.js +49 -28
- package/dist/esm/workspace.js.map +1 -1
- package/package.json +16 -6
- package/skills/ai-sandbox/SKILL.md +658 -20
- package/src/align.ts +297 -0
- package/src/attach-preflight.ts +292 -0
- package/src/capabilities.ts +4 -13
- package/src/chunk-identity.ts +154 -0
- package/src/claim.ts +479 -0
- package/src/contracts.ts +13 -0
- package/src/driver.ts +205 -0
- package/src/durability.ts +380 -0
- package/src/index.ts +212 -27
- package/src/instance-store.ts +122 -0
- package/src/journal-bytes.ts +136 -0
- package/src/journal-reader.ts +359 -0
- package/src/journal-sweep.ts +406 -0
- package/src/journal.ts +875 -0
- package/src/middleware.ts +470 -30
- package/src/reap.ts +723 -0
- package/src/reclaim.ts +191 -0
- package/src/run.ts +365 -75
- package/src/runner.ts +347 -3
- package/src/sandbox.ts +38 -8
- package/src/shell.ts +106 -38
- package/src/testkit/conformance.ts +117 -0
- package/src/testkit/durable-run-fields-conformance.ts +147 -0
- package/src/testkit/journal-conformance.ts +676 -0
- package/src/testkit/reaper-conformance.ts +1201 -0
- package/src/testkit/shell-spawn.ts +67 -0
- package/src/testkit/takeover-conformance.ts +1040 -0
- package/src/tool-history.ts +245 -0
- package/src/workspace.ts +1 -1
- package/dist/esm/index.js.map +0 -1
- package/dist/esm/run-log.d.ts +0 -81
- package/dist/esm/run-log.js +0 -107
- package/dist/esm/run-log.js.map +0 -1
- package/dist/esm/store.d.ts +0 -53
- package/dist/esm/store.js +0 -34
- package/dist/esm/store.js.map +0 -1
- package/src/run-log.ts +0 -224
- package/src/store.ts +0 -83
package/src/middleware.ts
CHANGED
|
@@ -1,12 +1,14 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `withSandbox(definition)` — the middleware that PROVIDES the
|
|
2
|
+
* `withSandbox(definition, options?)` — the middleware that PROVIDES the
|
|
3
3
|
* {@link SandboxCapability} a harness adapter requires.
|
|
4
4
|
*
|
|
5
5
|
* - `setup`: resume-or-create the sandbox (via the definition's ensure
|
|
6
|
-
* algorithm), provide the handle, using the
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
6
|
+
* algorithm), provide the handle, using the durability seams from
|
|
7
|
+
* {@link SandboxMiddlewareOptions} (or, failing that, a bus-provided
|
|
8
|
+
* SandboxInstanceStoreCapability / LocksCapability, then an in-memory
|
|
9
|
+
* fallback). If `fileEvents` is not false, starts a
|
|
10
|
+
* watcher that dispatches to sandbox-scoped hooks and forwards to the runtime
|
|
11
|
+
* sink.
|
|
10
12
|
* - `onFinish`/`onAbort`/`onError`: stop the watcher, snapshot (`after-run`)
|
|
11
13
|
* and/or destroy per lifecycle.
|
|
12
14
|
*
|
|
@@ -14,29 +16,54 @@
|
|
|
14
16
|
* are emitted by the harness adapter's chatStream (which can yield CUSTOM
|
|
15
17
|
* chunks), not from here — middleware setup runs before streaming begins.
|
|
16
18
|
*/
|
|
17
|
-
import { defineChatMiddleware } from '@tanstack/ai'
|
|
18
|
-
import { getSandboxRuntime } from '@tanstack/ai/adapter-internals'
|
|
19
19
|
import {
|
|
20
|
-
|
|
20
|
+
defineChatMiddleware,
|
|
21
|
+
provideDetachableRun,
|
|
22
|
+
provideRunDetached,
|
|
23
|
+
wasCancelRequested,
|
|
24
|
+
} from '@tanstack/ai'
|
|
25
|
+
import { InMemoryLockStore, LocksCapability } from '@tanstack/ai/locks'
|
|
26
|
+
import {
|
|
27
|
+
getPendingTurn,
|
|
28
|
+
getRunDisconnect,
|
|
29
|
+
getSandboxRuntime,
|
|
30
|
+
} from '@tanstack/ai/adapter-internals'
|
|
31
|
+
import {
|
|
21
32
|
SandboxCapability,
|
|
22
|
-
SandboxStoreCapability,
|
|
23
33
|
provideSandbox,
|
|
24
34
|
provideSandboxPolicy,
|
|
25
35
|
} from './capabilities'
|
|
36
|
+
import {
|
|
37
|
+
provideSandboxDurability,
|
|
38
|
+
resolveSandboxDurability,
|
|
39
|
+
} from './durability'
|
|
40
|
+
import { SandboxInstanceStoreCapability } from './instance-store'
|
|
26
41
|
import { computeWorkspaceHash } from './key'
|
|
27
42
|
import { buildFileHookEvent, resolveFileEvents } from './file-diff'
|
|
28
43
|
import { ProjectionCapability, provideWorkspaceProjection } from './projection'
|
|
29
44
|
import { resolveSecret } from './secrets'
|
|
45
|
+
import {
|
|
46
|
+
createToolHistoryRecorder,
|
|
47
|
+
stripObservedToolCalls,
|
|
48
|
+
} from './tool-history'
|
|
30
49
|
import { watchWorkspace } from './watch'
|
|
31
50
|
import { DEFAULT_WORKSPACE_ROOT } from './bootstrap'
|
|
32
51
|
import type { InternalLogger } from '@tanstack/ai/adapter-internals'
|
|
52
|
+
import type { LockStore } from '@tanstack/ai/locks'
|
|
33
53
|
import type {
|
|
34
54
|
AbortInfo,
|
|
35
55
|
ChatMiddlewareContext,
|
|
36
56
|
DefinedChatMiddleware,
|
|
57
|
+
RunStore,
|
|
37
58
|
SandboxFileEvent,
|
|
38
59
|
SandboxFileHookEvent,
|
|
39
60
|
} from '@tanstack/ai'
|
|
61
|
+
import type {
|
|
62
|
+
SandboxDurabilityOptions,
|
|
63
|
+
SandboxRunDurability,
|
|
64
|
+
} from './durability'
|
|
65
|
+
import type { SandboxInstanceStore } from './instance-store'
|
|
66
|
+
import type { ToolHistoryRecorder } from './tool-history'
|
|
40
67
|
import type { SandboxHandle } from './contracts'
|
|
41
68
|
import type {
|
|
42
69
|
SandboxDefinition,
|
|
@@ -47,7 +74,21 @@ import type { SandboxWatchHandle } from './watch'
|
|
|
47
74
|
|
|
48
75
|
/** Per-request state we need to carry from `setup` to the terminal hooks. */
|
|
49
76
|
interface SandboxRunState {
|
|
50
|
-
|
|
77
|
+
/**
|
|
78
|
+
* OPTIONAL because the state is registered BEFORE `definition.ensure()` is
|
|
79
|
+
* awaited, and `ensure` is the slowest thing in the whole run — cloning a repo
|
|
80
|
+
* into a fresh sandbox is minutes wide. That window is where the most common
|
|
81
|
+
* disconnect of all lands (a user starts a run and switches away while the UI
|
|
82
|
+
* still says "starting the sandbox"), so it is the one window the teardown and
|
|
83
|
+
* disconnect hooks most need to be able to act in. Registering only after the
|
|
84
|
+
* handle exists left exactly that window uncovered.
|
|
85
|
+
*
|
|
86
|
+
* Nothing the disconnect path does needs the handle: `detachedSince` and
|
|
87
|
+
* `sandboxKey` come from `ensureCtx`, which is built before `ensure` is called.
|
|
88
|
+
* Only `onFinish`'s snapshot needs it, and that cannot run before `setup` has
|
|
89
|
+
* completed.
|
|
90
|
+
*/
|
|
91
|
+
handle?: SandboxHandle
|
|
51
92
|
ensureCtx: SandboxEnsureContext
|
|
52
93
|
watcher?: SandboxWatchHandle
|
|
53
94
|
/** In-flight `enriched.diff()` promises queued by the `fileEvents.diff`
|
|
@@ -56,6 +97,18 @@ interface SandboxRunState {
|
|
|
56
97
|
pendingDiffs: Array<Promise<void>>
|
|
57
98
|
/** Logger captured at setup, so terminal hooks can log watcher teardown. */
|
|
58
99
|
logger?: InternalLogger
|
|
100
|
+
/**
|
|
101
|
+
* Durability resolved once at setup (absent when the run is not durable), so
|
|
102
|
+
* `onAbort` cannot reach a different verdict than the one `setup` published
|
|
103
|
+
* on the capability bus.
|
|
104
|
+
*/
|
|
105
|
+
durability?: SandboxRunDurability
|
|
106
|
+
/**
|
|
107
|
+
* Records the harness's own tool calls into the transcript, so a finished run
|
|
108
|
+
* restores its tool cards from the message store instead of only from the (live,
|
|
109
|
+
* rejoin-only) delivery log. See `./tool-history`.
|
|
110
|
+
*/
|
|
111
|
+
toolHistory: ToolHistoryRecorder
|
|
59
112
|
}
|
|
60
113
|
|
|
61
114
|
const runState = new WeakMap<object, SandboxRunState>()
|
|
@@ -82,6 +135,84 @@ async function drainWatcher(
|
|
|
82
135
|
if (state.watcher) state.logger?.sandbox('sandbox watcher stopped', { phase })
|
|
83
136
|
}
|
|
84
137
|
|
|
138
|
+
/**
|
|
139
|
+
* Record the two facts a later attach and the reaper both need, then publish the
|
|
140
|
+
* detach verdict core reads.
|
|
141
|
+
*
|
|
142
|
+
* Shared by the DISCONNECT subscriber registered in `setup` (the run is still
|
|
143
|
+
* going — the normal case) and `onAbort`'s detach branch (the run is being torn
|
|
144
|
+
* down while detachable), so the two can never write a different shape of detach.
|
|
145
|
+
*
|
|
146
|
+
* GUARDED, and reports failure rather than throwing. `update` is a documented
|
|
147
|
+
* no-op for an unknown runId, so a vanished record does not turn teardown into a
|
|
148
|
+
* throw; a genuinely rejecting store is the caller's to react to — `onAbort` falls
|
|
149
|
+
* through to destroying the sandbox, because a DESTROYED sandbox beats an
|
|
150
|
+
* unreachable one, while the disconnect subscriber has nothing to fall back to
|
|
151
|
+
* (the run is alive and still using the sandbox) and simply leaves the verdict
|
|
152
|
+
* unpublished.
|
|
153
|
+
*
|
|
154
|
+
* The verdict is published ONLY on success. Publishing it after a failed record
|
|
155
|
+
* write would leave core holding the log open for a takeover that can never be
|
|
156
|
+
* found, since nothing in the store points at the run.
|
|
157
|
+
*/
|
|
158
|
+
async function recordDetach(
|
|
159
|
+
definition: SandboxDefinition,
|
|
160
|
+
state: SandboxRunState,
|
|
161
|
+
durability: SandboxRunDurability,
|
|
162
|
+
ctx: ChatMiddlewareContext,
|
|
163
|
+
phase: 'disconnect' | 'abort',
|
|
164
|
+
): Promise<boolean> {
|
|
165
|
+
try {
|
|
166
|
+
// The record already exists: `setup` pre-creates it for every durable run
|
|
167
|
+
// BEFORE `ensure`, precisely so this stamp cannot land on a runId the store has
|
|
168
|
+
// never heard of — `RunStore.update` is a documented no-op for an unknown
|
|
169
|
+
// runId, which is how the detach used to be lost silently (measured against the
|
|
170
|
+
// browser repro: `detached_since` and `sandbox_key` both stayed NULL for a run
|
|
171
|
+
// that had genuinely detached). If it has since vanished, that no-op is the
|
|
172
|
+
// correct outcome and this must not throw.
|
|
173
|
+
await durability.runs.update(ctx.runId, {
|
|
174
|
+
detachedSince: Date.now(),
|
|
175
|
+
sandboxKey: definition.key(state.ensureCtx),
|
|
176
|
+
})
|
|
177
|
+
} catch (error) {
|
|
178
|
+
state.logger?.warn('sandbox detach record write failed', {
|
|
179
|
+
runId: ctx.runId,
|
|
180
|
+
phase,
|
|
181
|
+
error,
|
|
182
|
+
})
|
|
183
|
+
return false
|
|
184
|
+
}
|
|
185
|
+
// Core's durable delivery sink reads this (see `RunDetachedCapability`) and
|
|
186
|
+
// leaves the run's log OPEN instead of appending a synthetic terminal
|
|
187
|
+
// `RUN_ERROR` and closing it — a terminalized log ends a later attach's replay
|
|
188
|
+
// at the prefix and diverges the takeover's journal replay, which recorded a
|
|
189
|
+
// healthy detached run as `'failed'`.
|
|
190
|
+
provideRunDetached(ctx, true)
|
|
191
|
+
return true
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/**
|
|
195
|
+
* Whether an out-of-band cancel has been recorded for this run, in EITHER band.
|
|
196
|
+
* A user pressing Stop and a user closing the tab produce the IDENTICAL
|
|
197
|
+
* connection close, so intent is never inferred from the disconnect itself: it
|
|
198
|
+
* arrives in-process (the abort reason carried the cancel sentinel) or durably
|
|
199
|
+
* (another host recorded it on the run record).
|
|
200
|
+
*/
|
|
201
|
+
async function cancelIntent(
|
|
202
|
+
durability: SandboxRunDurability | undefined,
|
|
203
|
+
runId: string,
|
|
204
|
+
inProcess: boolean,
|
|
205
|
+
): Promise<boolean> {
|
|
206
|
+
if (inProcess) return true
|
|
207
|
+
if (durability === undefined) return false
|
|
208
|
+
// No guard needed here, and one would be dead code: `wasCancelRequested` already
|
|
209
|
+
// answers `false` for a store read that rejects. That matters on this path,
|
|
210
|
+
// because a rejection escaping into `onAbort` would skip BOTH of its branches at
|
|
211
|
+
// once, leaving a sandbox that is neither reclaimable nor destroyed. The test
|
|
212
|
+
// 'DETACHES when the cancel probe REJECTS' pins the composition.
|
|
213
|
+
return wasCancelRequested(durability.runs, runId)
|
|
214
|
+
}
|
|
215
|
+
|
|
85
216
|
/** Defensively pull tenant scoping out of the runtime context, if present. */
|
|
86
217
|
function tenantFrom(
|
|
87
218
|
context: unknown,
|
|
@@ -94,12 +225,72 @@ function tenantFrom(
|
|
|
94
225
|
return { userId, orgId }
|
|
95
226
|
}
|
|
96
227
|
|
|
97
|
-
|
|
228
|
+
/**
|
|
229
|
+
* Durability seams for a sandboxed run. Both are optional; each independently
|
|
230
|
+
* falls back to a process-lifetime in-memory default, which is correct for a
|
|
231
|
+
* single process but NOT across replicas.
|
|
232
|
+
*/
|
|
233
|
+
export interface SandboxMiddlewareOptions<TOffset extends string = string> {
|
|
234
|
+
/**
|
|
235
|
+
* Durable instance map (which provider sandbox to resume for a key). Pass
|
|
236
|
+
* your own store to make resume survive across processes/replicas.
|
|
237
|
+
*
|
|
238
|
+
* Takes precedence over a store provided on the capability bus (see
|
|
239
|
+
* `provideSandboxInstanceStore`), so the call site wins over ambient wiring.
|
|
240
|
+
*/
|
|
241
|
+
instances?: SandboxInstanceStore
|
|
242
|
+
/**
|
|
243
|
+
* Distributed lock serializing resume-or-create for one key. Needed for
|
|
244
|
+
* multi-replica correctness so two concurrent runs don't both create.
|
|
245
|
+
*
|
|
246
|
+
* Prefer `withLocks` from `@tanstack/ai/locks` when other middleware also
|
|
247
|
+
* needs the lock; use this option to scope one to this sandbox. Takes
|
|
248
|
+
* precedence over a bus-provided lock.
|
|
249
|
+
*/
|
|
250
|
+
locks?: LockStore
|
|
251
|
+
/**
|
|
252
|
+
* Run lifecycle records. Pair with `durability.adapter` to make a run
|
|
253
|
+
* DETACHABLE: a client disconnect then leaves the agent running and records
|
|
254
|
+
* `detachedSince` instead of destroying the sandbox.
|
|
255
|
+
*
|
|
256
|
+
* Pass the SAME store chat persistence uses (`persistence.stores.runs`) so
|
|
257
|
+
* one record describes the run instead of two that can disagree.
|
|
258
|
+
*
|
|
259
|
+
* Defaults to `undefined`: an app that passes neither this nor `durability`
|
|
260
|
+
* keeps today's destroy-on-disconnect behavior exactly.
|
|
261
|
+
*/
|
|
262
|
+
runs?: RunStore
|
|
263
|
+
/**
|
|
264
|
+
* Delivery durability for the run's event log, plus the journal and detach
|
|
265
|
+
* knobs. Requires `runs`; either alone is not durable.
|
|
266
|
+
*
|
|
267
|
+
* `TOffset` is inferred from the adapter passed here, so a branded-cursor
|
|
268
|
+
* backend (`durableStream`) wires without a cast and without the call site
|
|
269
|
+
* ever naming the parameter.
|
|
270
|
+
*/
|
|
271
|
+
durability?: SandboxDurabilityOptions<TOffset>
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
/**
|
|
275
|
+
* Resolve the ensure seams. Precedence is explicit option → capability bus →
|
|
276
|
+
* (in `ensure`) the in-memory fallback. The option wins because it is visible
|
|
277
|
+
* at the call site; the bus remains for platform/framework injection.
|
|
278
|
+
*/
|
|
279
|
+
function buildEnsureCtx(
|
|
280
|
+
ctx: ChatMiddlewareContext,
|
|
281
|
+
// Narrowed to the two seams it reads rather than taking the whole options
|
|
282
|
+
// object: `SandboxMiddlewareOptions` is now generic in the durability offset,
|
|
283
|
+
// and `SandboxMiddlewareOptions<TOffset>` is not assignable to
|
|
284
|
+
// `SandboxMiddlewareOptions<string>`. Both members here are offset-free, so
|
|
285
|
+
// the narrowing keeps this helper independent of that parameter entirely.
|
|
286
|
+
options: Pick<SandboxMiddlewareOptions, 'instances' | 'locks'> | undefined,
|
|
287
|
+
): SandboxEnsureContext {
|
|
98
288
|
return {
|
|
99
289
|
threadId: ctx.threadId,
|
|
100
290
|
runId: ctx.runId,
|
|
101
|
-
store:
|
|
102
|
-
|
|
291
|
+
store:
|
|
292
|
+
options?.instances ?? ctx.getOptional(SandboxInstanceStoreCapability),
|
|
293
|
+
locks: options?.locks ?? ctx.getOptional(LocksCapability),
|
|
103
294
|
tenant: tenantFrom(ctx.context),
|
|
104
295
|
signal: ctx.signal,
|
|
105
296
|
}
|
|
@@ -141,8 +332,9 @@ async function dispatchDefinitionHooks(
|
|
|
141
332
|
}
|
|
142
333
|
}
|
|
143
334
|
|
|
144
|
-
export function withSandbox(
|
|
335
|
+
export function withSandbox<TOffset extends string = string>(
|
|
145
336
|
definition: SandboxDefinition,
|
|
337
|
+
options?: SandboxMiddlewareOptions<TOffset>,
|
|
146
338
|
): DefinedChatMiddleware<
|
|
147
339
|
unknown,
|
|
148
340
|
readonly [],
|
|
@@ -153,14 +345,29 @@ export function withSandbox(
|
|
|
153
345
|
provides: [SandboxCapability, ProjectionCapability],
|
|
154
346
|
// SandboxPolicyCapability is provided conditionally (only when the
|
|
155
347
|
// definition has a policy), so it is intentionally NOT declared here —
|
|
156
|
-
// consumers read it via `getOptional`.
|
|
157
|
-
|
|
348
|
+
// consumers read it via `getOptional`. SandboxDurabilityCapability and
|
|
349
|
+
// DetachableRunCapability are conditional for the same reason (only when
|
|
350
|
+
// `runs` + `durability` are both wired), so they are intentionally NOT
|
|
351
|
+
// declared here either.
|
|
352
|
+
optionalRequires: [SandboxInstanceStoreCapability, LocksCapability],
|
|
158
353
|
|
|
159
354
|
async setup(ctx) {
|
|
160
|
-
const ensureCtx = buildEnsureCtx(ctx)
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
355
|
+
const ensureCtx = buildEnsureCtx(ctx, options)
|
|
356
|
+
|
|
357
|
+
// Resolving here (not lazily on the abort path) is what keeps `setup` and
|
|
358
|
+
// `onAbort` on one verdict: the payload the bus carries is the same object
|
|
359
|
+
// the teardown path consults.
|
|
360
|
+
// `TOffset` is passed explicitly: `options` is possibly `undefined` here,
|
|
361
|
+
// so inference has nothing to work from on that branch and would fall
|
|
362
|
+
// back to the `= string` default, re-erecting the very wall this
|
|
363
|
+
// parameter exists to remove.
|
|
364
|
+
const durability = resolveSandboxDurability<TOffset>(options)
|
|
365
|
+
if (durability !== undefined) {
|
|
366
|
+
provideSandboxDurability(ctx, durability)
|
|
367
|
+
// A neutral boolean core owns, so `@tanstack/ai-persistence` can ask
|
|
368
|
+
// "is this run detachable?" without depending on this package.
|
|
369
|
+
provideDetachableRun(ctx, true)
|
|
370
|
+
}
|
|
164
371
|
|
|
165
372
|
// Pull the runtime (and its logger) up front so `baseSha` capture and
|
|
166
373
|
// hook dispatch below can log through the same `sandbox`/`errors`
|
|
@@ -168,6 +375,166 @@ export function withSandbox(
|
|
|
168
375
|
const runtime = getSandboxRuntime(ctx, { optional: true })
|
|
169
376
|
const logger = runtime?.logger
|
|
170
377
|
|
|
378
|
+
// REGISTER THE RUN STATE NOW — before `definition.ensure()`, not merely
|
|
379
|
+
// before the end of `setup`.
|
|
380
|
+
//
|
|
381
|
+
// `onAbort` and the disconnect subscriber both need this state, so until
|
|
382
|
+
// this map is populated they are silent no-ops. `ensure` is the LONGEST
|
|
383
|
+
// await in the entire run (create a sandbox, clone a repo — minutes), and it
|
|
384
|
+
// is where the most common disconnect of all lands: a user starts a run and
|
|
385
|
+
// switches away while the UI still says "starting the sandbox". Registering
|
|
386
|
+
// after `ensure` returned still left that whole window uncovered.
|
|
387
|
+
//
|
|
388
|
+
// Leaving it uncovered loses every teardown behavior at once: no
|
|
389
|
+
// `detachedSince`/`sandboxKey`, so `listReclaimable` can never surface the
|
|
390
|
+
// run and the reaper can never reclaim it; no `definition.destroy`, so the
|
|
391
|
+
// sandbox leaks; and no detach verdict for core to read.
|
|
392
|
+
//
|
|
393
|
+
// Everything those hooks read is already resolved above: the ensure context
|
|
394
|
+
// (which is all `definition.key` needs), the durability verdict, and the
|
|
395
|
+
// logger. The fields discovered later (`handle`, `watcher`) are ASSIGNED onto
|
|
396
|
+
// this same object as they become available, so the teardown path always
|
|
397
|
+
// sees the most complete state that exists at the moment it runs.
|
|
398
|
+
const state: SandboxRunState = {
|
|
399
|
+
ensureCtx,
|
|
400
|
+
pendingDiffs: [],
|
|
401
|
+
toolHistory: createToolHistoryRecorder(),
|
|
402
|
+
...(logger ? { logger } : {}),
|
|
403
|
+
...(durability ? { durability } : {}),
|
|
404
|
+
}
|
|
405
|
+
runState.set(ctx, state)
|
|
406
|
+
|
|
407
|
+
// MAKE THE RUN FINDABLE BEFORE `ensure`, not after the run finally streams.
|
|
408
|
+
//
|
|
409
|
+
// Chat persistence creates the run record from `onConfig`, which runs after
|
|
410
|
+
// EVERY middleware `setup` — so for the whole of `definition.ensure` (create a
|
|
411
|
+
// sandbox, clone a repo: minutes) the run has no record at all, and
|
|
412
|
+
// `findActiveRun` answers "no active run" for a run that is demonstrably
|
|
413
|
+
// starting. Measured: a status sidebar read straight off `findActiveRun`
|
|
414
|
+
// reported `idle` for 6.5 minutes while the sandbox was being built, and a
|
|
415
|
+
// client returning to the thread in that window had nothing to tell it a run
|
|
416
|
+
// was in flight — so it rendered an empty pane instead of "starting sandbox".
|
|
417
|
+
//
|
|
418
|
+
// A crash in the same window is worse: no record means `listReclaimable` can
|
|
419
|
+
// never surface the run, so the sandbox leaks with no recovery path.
|
|
420
|
+
//
|
|
421
|
+
// `createOrResume` is idempotent and never resurrects a finished run, so
|
|
422
|
+
// persistence's own later call stays correct and simply finds this record.
|
|
423
|
+
if (durability !== undefined) {
|
|
424
|
+
try {
|
|
425
|
+
await durability.runs.createOrResume({
|
|
426
|
+
runId: ctx.runId,
|
|
427
|
+
threadId: ctx.threadId,
|
|
428
|
+
startedAt: Date.now(),
|
|
429
|
+
})
|
|
430
|
+
} catch (error) {
|
|
431
|
+
// Best-effort: a store blip must not stop a run that is otherwise fine.
|
|
432
|
+
// The run is simply invisible until persistence's own `onConfig` call.
|
|
433
|
+
logger?.warn('sandbox run record pre-create failed', {
|
|
434
|
+
runId: ctx.runId,
|
|
435
|
+
error,
|
|
436
|
+
})
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
// NO ATTACH MARKER HERE. A joiner does need a chunk in the log before the
|
|
440
|
+
// harness has emitted anything — an empty log fails every joiner's
|
|
441
|
+
// fast-fail (`memoryStream`'s first-chunk deadline, the client's rejoin
|
|
442
|
+
// connect deadline) and flushes no HTTP headers, so a reload during
|
|
443
|
+
// `ensure` reads a live run as gone. Core does it: a fresh durable producer
|
|
444
|
+
// appends `RUN_ACCEPTED_EVENT` before the producer stream is first pulled,
|
|
445
|
+
// for EVERY durable run rather than only sandboxed ones, and never on an
|
|
446
|
+
// attach. A second marker from here would only land mid-stream in a run
|
|
447
|
+
// that is already producing.
|
|
448
|
+
|
|
449
|
+
// STORE THE USER'S TURN NOW, before `ensure` takes minutes.
|
|
450
|
+
//
|
|
451
|
+
// Chat persistence stores it from `onStart`, which runs after every
|
|
452
|
+
// middleware `setup` — so without this the thread holds NOTHING for the
|
|
453
|
+
// whole sandbox build. Measured: a reload during the build asked the server
|
|
454
|
+
// for the conversation and got `{"messages":[],…}`, so the user saw no sign
|
|
455
|
+
// of the message they had just sent, and a second device saw an empty
|
|
456
|
+
// thread.
|
|
457
|
+
//
|
|
458
|
+
// The persistence layer owns WHAT gets stored (see `PendingTurnCapability`):
|
|
459
|
+
// `saveThread` replaces the thread, so deciding the list here would risk
|
|
460
|
+
// deleting the history. Absent when the app wires no persistence, which is
|
|
461
|
+
// simply a run with no transcript to store.
|
|
462
|
+
try {
|
|
463
|
+
await getPendingTurn(ctx, { optional: true })?.snapshot()
|
|
464
|
+
} catch (error) {
|
|
465
|
+
// Best-effort: the run is still worth doing, and `onStart` stores the
|
|
466
|
+
// turn again once setup completes.
|
|
467
|
+
logger?.warn('sandbox pending-turn snapshot failed', {
|
|
468
|
+
runId: ctx.runId,
|
|
469
|
+
error,
|
|
470
|
+
})
|
|
471
|
+
}
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
// SUBSCRIBE BEFORE `ensure`, for the same reason the state is registered
|
|
475
|
+
// before it: `ensure` is the minutes-wide await a disconnect actually lands
|
|
476
|
+
// in. Core calls back immediately if the socket has already closed, so
|
|
477
|
+
// subscribing here cannot miss a disconnect that beat us to it.
|
|
478
|
+
//
|
|
479
|
+
// This is what makes a durable run SURVIVE losing its viewer. The only route
|
|
480
|
+
// a disconnect previously had into this middleware was the application
|
|
481
|
+
// mirroring `request.signal` into `chat()`'s `abortController` — which aborts
|
|
482
|
+
// the run, so `chat()` returned right after this `setup` and the harness
|
|
483
|
+
// adapter's `chatStream` was never called: the agent in the sandbox we just
|
|
484
|
+
// spent minutes creating was NEVER LAUNCHED, and no takeover could recover it
|
|
485
|
+
// because an agent that never ran wrote no journal to replay.
|
|
486
|
+
if (durability !== undefined && durability.detachOnDisconnect) {
|
|
487
|
+
getRunDisconnect(ctx, { optional: true })?.subscribe(async () => {
|
|
488
|
+
// BOOKKEEPING ONLY — the run is still executing. Deliberately absent:
|
|
489
|
+
// `drainWatcher` (would blind a live agent's file events for the whole
|
|
490
|
+
// remainder) and `definition.destroy` (the run is still using the
|
|
491
|
+
// sandbox). Both belong to the terminal hooks, which still run exactly
|
|
492
|
+
// once afterwards.
|
|
493
|
+
//
|
|
494
|
+
// A run with a cancel already recorded is left alone: that is `onAbort`'s
|
|
495
|
+
// path, and stamping `detachedSince` on a deliberately-stopped run would
|
|
496
|
+
// hand it to the reaper as reclaimable work.
|
|
497
|
+
if (await cancelIntent(durability, ctx.runId, false)) return
|
|
498
|
+
if (
|
|
499
|
+
await recordDetach(definition, state, durability, ctx, 'disconnect')
|
|
500
|
+
) {
|
|
501
|
+
state.logger?.sandbox(
|
|
502
|
+
'sandbox run detached on disconnect; the run continues',
|
|
503
|
+
{ runId: ctx.runId },
|
|
504
|
+
)
|
|
505
|
+
}
|
|
506
|
+
})
|
|
507
|
+
}
|
|
508
|
+
|
|
509
|
+
const handle = await definition.ensure(ensureCtx)
|
|
510
|
+
// MUTATE, don't re-`set`: a disconnect that landed during `ensure` already
|
|
511
|
+
// captured this object.
|
|
512
|
+
state.handle = handle
|
|
513
|
+
provideSandbox(ctx, handle)
|
|
514
|
+
if (definition.policy) provideSandboxPolicy(ctx, definition.policy)
|
|
515
|
+
|
|
516
|
+
// Deliberately placed AFTER `logger` is in scope rather than next to the
|
|
517
|
+
// `provideSandboxDurability` call above — there is no logger to warn
|
|
518
|
+
// through until the runtime has been read.
|
|
519
|
+
//
|
|
520
|
+
// `ensureCtx.locks === undefined` counts as in-memory: `defineSandbox`'s
|
|
521
|
+
// `ensure` falls back to a process-lifetime `InMemoryLockStore` when no
|
|
522
|
+
// lock is wired, so an unwired lock has exactly the deficiency being
|
|
523
|
+
// warned about — it is the MOST in-memory case, not an exempt one.
|
|
524
|
+
if (
|
|
525
|
+
durability !== undefined &&
|
|
526
|
+
(ensureCtx.locks === undefined ||
|
|
527
|
+
ensureCtx.locks instanceof InMemoryLockStore)
|
|
528
|
+
) {
|
|
529
|
+
logger?.warn(
|
|
530
|
+
'sandbox durability is wired over an InMemoryLockStore: run claims are ' +
|
|
531
|
+
'serialized within this process only and the lease never signals loss, ' +
|
|
532
|
+
'so two hosts can drive one run and duplicate its event log. Use a ' +
|
|
533
|
+
'distributed LockStore via withLocks for any multi-replica deploy.',
|
|
534
|
+
{ runId: ctx.runId },
|
|
535
|
+
)
|
|
536
|
+
}
|
|
537
|
+
|
|
171
538
|
const watchRoot = definition.workspace?.root ?? DEFAULT_WORKSPACE_ROOT
|
|
172
539
|
let baseSha = ''
|
|
173
540
|
try {
|
|
@@ -229,7 +596,11 @@ export function withSandbox(
|
|
|
229
596
|
await hooks?.onReady?.(handle)
|
|
230
597
|
|
|
231
598
|
const fe = resolveFileEvents(definition.fileEvents)
|
|
232
|
-
|
|
599
|
+
// THE SAME array the run state already holds, not a fresh one. The watcher
|
|
600
|
+
// callback below closes over this reference, and `drainWatcher` awaits
|
|
601
|
+
// `state.pendingDiffs` — a second array would silently drop every in-flight
|
|
602
|
+
// diff from the teardown drain.
|
|
603
|
+
const pendingDiffs = state.pendingDiffs
|
|
233
604
|
let watcher: SandboxWatchHandle | undefined
|
|
234
605
|
if (fe.enabled) {
|
|
235
606
|
watcher = await watchWorkspace(handle, {
|
|
@@ -274,13 +645,38 @@ export function withSandbox(
|
|
|
274
645
|
})
|
|
275
646
|
}
|
|
276
647
|
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
648
|
+
// MUTATE the object registered above rather than `set`-ing a second one: an
|
|
649
|
+
// abort that landed mid-setup already captured a reference to it (and may
|
|
650
|
+
// already be draining `pendingDiffs`), so replacing the entry would hand the
|
|
651
|
+
// teardown path a different object than the watcher writes into.
|
|
652
|
+
// `pendingDiffs` needs no copying — it IS `state.pendingDiffs`.
|
|
653
|
+
if (watcher) state.watcher = watcher
|
|
654
|
+
},
|
|
655
|
+
|
|
656
|
+
// Keep the recorded tool history OUT of the request to the model. It is stored
|
|
657
|
+
// history for the next turn, it names tools the provider was never given, and one
|
|
658
|
+
// triage-sized run is hundreds of kilobytes — so replaying it is wasteful at best
|
|
659
|
+
// and rejected at worst. `ctx.messages` keeps it (that is what gets stored and
|
|
660
|
+
// rendered); only `config.messages` loses it.
|
|
661
|
+
onConfig(_ctx, config) {
|
|
662
|
+
const messages = stripObservedToolCalls(config.messages)
|
|
663
|
+
if (messages.length === config.messages.length) return
|
|
664
|
+
return { messages }
|
|
665
|
+
},
|
|
666
|
+
|
|
667
|
+
// The engine re-syncs `middlewareCtx.messages` from its own array once per agent
|
|
668
|
+
// iteration, which drops whatever the recorder appended during the previous
|
|
669
|
+
// iteration's stream. Restoring it here — AFTER that sync — is what makes a
|
|
670
|
+
// multi-iteration run keep its full history without depending on where this
|
|
671
|
+
// middleware sits relative to persistence in the middleware array.
|
|
672
|
+
onIteration(ctx) {
|
|
673
|
+
runState.get(ctx)?.toolHistory.reconcile(ctx)
|
|
674
|
+
},
|
|
675
|
+
|
|
676
|
+
// Record the harness's own tool calls as transcript messages. Observe only:
|
|
677
|
+
// returning nothing passes every chunk through untouched.
|
|
678
|
+
onChunk(ctx, chunk) {
|
|
679
|
+
runState.get(ctx)?.toolHistory.observe(chunk, ctx)
|
|
284
680
|
},
|
|
285
681
|
|
|
286
682
|
async onFinish(ctx) {
|
|
@@ -288,13 +684,20 @@ export function withSandbox(
|
|
|
288
684
|
if (!state) return
|
|
289
685
|
const { handle, ensureCtx } = state
|
|
290
686
|
|
|
687
|
+
// Last chance before persistence writes the transcript. Only matters if a
|
|
688
|
+
// config sync landed after the final tool chunk; the recorder is idempotent, so
|
|
689
|
+
// in the normal case this changes nothing.
|
|
690
|
+
state.toolHistory.reconcile(ctx)
|
|
691
|
+
|
|
291
692
|
await drainWatcher(state, 'finish')
|
|
292
693
|
|
|
293
694
|
const lifecycle = definition.lifecycle
|
|
294
695
|
|
|
696
|
+
// `handle` is absent only if `setup` never got past `definition.ensure`, in
|
|
697
|
+
// which case there is no sandbox to snapshot.
|
|
295
698
|
if (
|
|
296
699
|
lifecycle?.snapshot === 'after-run' &&
|
|
297
|
-
handle
|
|
700
|
+
handle?.capabilities.snapshots &&
|
|
298
701
|
handle.snapshot
|
|
299
702
|
) {
|
|
300
703
|
const snapshot = await handle.snapshot(`after-run-${ctx.runId}`)
|
|
@@ -318,12 +721,49 @@ export function withSandbox(
|
|
|
318
721
|
}
|
|
319
722
|
},
|
|
320
723
|
|
|
321
|
-
async onAbort(ctx,
|
|
724
|
+
async onAbort(ctx, info: AbortInfo) {
|
|
322
725
|
const state = runState.get(ctx)
|
|
323
726
|
if (!state) return
|
|
324
727
|
|
|
728
|
+
// First on BOTH branches: a diff still in flight must be drained whether
|
|
729
|
+
// the sandbox is about to be destroyed or merely detached, or the final
|
|
730
|
+
// file's diff is dropped.
|
|
325
731
|
await drainWatcher(state, 'abort')
|
|
326
732
|
|
|
733
|
+
const durability = state.durability
|
|
734
|
+
const cancelled = await cancelIntent(
|
|
735
|
+
durability,
|
|
736
|
+
ctx.runId,
|
|
737
|
+
info.cancelRequested === true,
|
|
738
|
+
)
|
|
739
|
+
|
|
740
|
+
if (
|
|
741
|
+
durability !== undefined &&
|
|
742
|
+
!cancelled &&
|
|
743
|
+
durability.detachOnDisconnect
|
|
744
|
+
) {
|
|
745
|
+
// DETACH on the teardown path. Reached when the run is aborted for a
|
|
746
|
+
// reason that is NOT an out-of-band cancel while detachable — a genuine
|
|
747
|
+
// stop from elsewhere, or a host going down. The ordinary disconnect is
|
|
748
|
+
// handled by the disconnect subscriber in `setup`, which does not end the
|
|
749
|
+
// run at all.
|
|
750
|
+
//
|
|
751
|
+
// On a failed record write this branch is ABANDONED for the destroy one
|
|
752
|
+
// below, because a rejection here is the worst shape available: the
|
|
753
|
+
// verdict is unpublished, so core terminalizes the log and records a
|
|
754
|
+
// healthy detached run as failed; `detachedSince`/`sandboxKey` are
|
|
755
|
+
// unwritten, so `listReclaimable` can never surface the run and
|
|
756
|
+
// `reapDetachedRuns` can never reclaim it. A DESTROYED sandbox beats an
|
|
757
|
+
// unreachable one — the same reasoning `drainWatcher` applies to its own
|
|
758
|
+
// guarded `stop()`.
|
|
759
|
+
if (await recordDetach(definition, state, durability, ctx, 'abort')) {
|
|
760
|
+
return
|
|
761
|
+
}
|
|
762
|
+
await definition.destroy(state.ensureCtx)
|
|
763
|
+
await definition.hooks?.onDestroy?.()
|
|
764
|
+
return
|
|
765
|
+
}
|
|
766
|
+
|
|
327
767
|
// ALWAYS tear down on an explicit abort, regardless of `destroyOnComplete`.
|
|
328
768
|
// The in-sandbox agent process is not killed by closing its IO stream
|
|
329
769
|
// (e.g. a Docker exec survives client disconnect), so the only reliable way
|