@tanstack/ai-sandbox 0.2.3 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/dist/esm/agents-file.js +53 -34
  2. package/dist/esm/agents-file.js.map +1 -1
  3. package/dist/esm/align.d.ts +121 -0
  4. package/dist/esm/align.js +197 -0
  5. package/dist/esm/align.js.map +1 -0
  6. package/dist/esm/approvals.js +63 -29
  7. package/dist/esm/approvals.js.map +1 -1
  8. package/dist/esm/attach-preflight.d.ts +85 -0
  9. package/dist/esm/attach-preflight.js +189 -0
  10. package/dist/esm/attach-preflight.js.map +1 -0
  11. package/dist/esm/bootstrap.js +103 -117
  12. package/dist/esm/bootstrap.js.map +1 -1
  13. package/dist/esm/bridge-events.js +96 -71
  14. package/dist/esm/bridge-events.js.map +1 -1
  15. package/dist/esm/capabilities.d.ts +0 -5
  16. package/dist/esm/capabilities.js +32 -28
  17. package/dist/esm/capabilities.js.map +1 -1
  18. package/dist/esm/chunk-identity.d.ts +52 -0
  19. package/dist/esm/chunk-identity.js +102 -0
  20. package/dist/esm/chunk-identity.js.map +1 -0
  21. package/dist/esm/claim.d.ts +187 -0
  22. package/dist/esm/claim.js +349 -0
  23. package/dist/esm/claim.js.map +1 -0
  24. package/dist/esm/contracts.d.ts +13 -0
  25. package/dist/esm/driver.d.ts +83 -0
  26. package/dist/esm/driver.js +138 -0
  27. package/dist/esm/driver.js.map +1 -0
  28. package/dist/esm/durability.d.ts +263 -0
  29. package/dist/esm/durability.js +230 -0
  30. package/dist/esm/durability.js.map +1 -0
  31. package/dist/esm/errors.js +28 -24
  32. package/dist/esm/errors.js.map +1 -1
  33. package/dist/esm/file-diff.js +151 -135
  34. package/dist/esm/file-diff.js.map +1 -1
  35. package/dist/esm/git-exec.js +51 -62
  36. package/dist/esm/git-exec.js.map +1 -1
  37. package/dist/esm/harness-cwd.js +24 -19
  38. package/dist/esm/harness-cwd.js.map +1 -1
  39. package/dist/esm/index.d.ts +30 -8
  40. package/dist/esm/index.js +23 -91
  41. package/dist/esm/instance-store.d.ts +88 -0
  42. package/dist/esm/instance-store.js +67 -0
  43. package/dist/esm/instance-store.js.map +1 -0
  44. package/dist/esm/journal-bytes.d.ts +67 -0
  45. package/dist/esm/journal-bytes.js +110 -0
  46. package/dist/esm/journal-bytes.js.map +1 -0
  47. package/dist/esm/journal-reader.d.ts +66 -0
  48. package/dist/esm/journal-reader.js +228 -0
  49. package/dist/esm/journal-reader.js.map +1 -0
  50. package/dist/esm/journal-sweep.d.ts +113 -0
  51. package/dist/esm/journal-sweep.js +309 -0
  52. package/dist/esm/journal-sweep.js.map +1 -0
  53. package/dist/esm/journal.d.ts +542 -0
  54. package/dist/esm/journal.js +679 -0
  55. package/dist/esm/journal.js.map +1 -0
  56. package/dist/esm/key.js +36 -33
  57. package/dist/esm/key.js.map +1 -1
  58. package/dist/esm/middleware.d.ts +50 -2
  59. package/dist/esm/middleware.js +335 -208
  60. package/dist/esm/middleware.js.map +1 -1
  61. package/dist/esm/ngrok.js +75 -49
  62. package/dist/esm/ngrok.js.map +1 -1
  63. package/dist/esm/policy.js +43 -34
  64. package/dist/esm/policy.js.map +1 -1
  65. package/dist/esm/projection.js +16 -8
  66. package/dist/esm/projection.js.map +1 -1
  67. package/dist/esm/reap.d.ts +238 -0
  68. package/dist/esm/reap.js +355 -0
  69. package/dist/esm/reap.js.map +1 -0
  70. package/dist/esm/reclaim.d.ts +84 -0
  71. package/dist/esm/reclaim.js +106 -0
  72. package/dist/esm/reclaim.js.map +1 -0
  73. package/dist/esm/remote-tools.js +73 -62
  74. package/dist/esm/remote-tools.js.map +1 -1
  75. package/dist/esm/run.d.ts +93 -25
  76. package/dist/esm/run.js +274 -79
  77. package/dist/esm/run.js.map +1 -1
  78. package/dist/esm/runner.d.ts +119 -2
  79. package/dist/esm/runner.js +270 -51
  80. package/dist/esm/runner.js.map +1 -1
  81. package/dist/esm/sandbox.d.ts +3 -2
  82. package/dist/esm/sandbox.js +139 -123
  83. package/dist/esm/sandbox.js.map +1 -1
  84. package/dist/esm/secrets.js +39 -47
  85. package/dist/esm/secrets.js.map +1 -1
  86. package/dist/esm/setup-plan.js +22 -14
  87. package/dist/esm/setup-plan.js.map +1 -1
  88. package/dist/esm/shell.d.ts +8 -0
  89. package/dist/esm/shell.js +197 -158
  90. package/dist/esm/shell.js.map +1 -1
  91. package/dist/esm/testkit/conformance.d.ts +16 -0
  92. package/dist/esm/testkit/conformance.js +97 -0
  93. package/dist/esm/testkit/conformance.js.map +1 -0
  94. package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
  95. package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
  96. package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
  97. package/dist/esm/testkit/journal-conformance.d.ts +51 -0
  98. package/dist/esm/testkit/journal-conformance.js +378 -0
  99. package/dist/esm/testkit/journal-conformance.js.map +1 -0
  100. package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
  101. package/dist/esm/testkit/reaper-conformance.js +847 -0
  102. package/dist/esm/testkit/reaper-conformance.js.map +1 -0
  103. package/dist/esm/testkit/shell-spawn.d.ts +2 -0
  104. package/dist/esm/testkit/shell-spawn.js +60 -0
  105. package/dist/esm/testkit/shell-spawn.js.map +1 -0
  106. package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
  107. package/dist/esm/testkit/takeover-conformance.js +685 -0
  108. package/dist/esm/testkit/takeover-conformance.js.map +1 -0
  109. package/dist/esm/tool-bridge.js +227 -180
  110. package/dist/esm/tool-bridge.js.map +1 -1
  111. package/dist/esm/tool-history.d.ts +62 -0
  112. package/dist/esm/tool-history.js +171 -0
  113. package/dist/esm/tool-history.js.map +1 -0
  114. package/dist/esm/watch.js +310 -236
  115. package/dist/esm/watch.js.map +1 -1
  116. package/dist/esm/workspace.d.ts +1 -1
  117. package/dist/esm/workspace.js +49 -28
  118. package/dist/esm/workspace.js.map +1 -1
  119. package/package.json +16 -6
  120. package/skills/ai-sandbox/SKILL.md +658 -20
  121. package/src/align.ts +297 -0
  122. package/src/attach-preflight.ts +292 -0
  123. package/src/capabilities.ts +4 -13
  124. package/src/chunk-identity.ts +154 -0
  125. package/src/claim.ts +479 -0
  126. package/src/contracts.ts +13 -0
  127. package/src/driver.ts +205 -0
  128. package/src/durability.ts +380 -0
  129. package/src/index.ts +212 -27
  130. package/src/instance-store.ts +122 -0
  131. package/src/journal-bytes.ts +136 -0
  132. package/src/journal-reader.ts +359 -0
  133. package/src/journal-sweep.ts +406 -0
  134. package/src/journal.ts +875 -0
  135. package/src/middleware.ts +470 -30
  136. package/src/reap.ts +723 -0
  137. package/src/reclaim.ts +191 -0
  138. package/src/run.ts +365 -75
  139. package/src/runner.ts +347 -3
  140. package/src/sandbox.ts +38 -8
  141. package/src/shell.ts +106 -38
  142. package/src/testkit/conformance.ts +117 -0
  143. package/src/testkit/durable-run-fields-conformance.ts +147 -0
  144. package/src/testkit/journal-conformance.ts +676 -0
  145. package/src/testkit/reaper-conformance.ts +1201 -0
  146. package/src/testkit/shell-spawn.ts +67 -0
  147. package/src/testkit/takeover-conformance.ts +1040 -0
  148. package/src/tool-history.ts +245 -0
  149. package/src/workspace.ts +1 -1
  150. package/dist/esm/index.js.map +0 -1
  151. package/dist/esm/run-log.d.ts +0 -81
  152. package/dist/esm/run-log.js +0 -107
  153. package/dist/esm/run-log.js.map +0 -1
  154. package/dist/esm/store.d.ts +0 -53
  155. package/dist/esm/store.js +0 -34
  156. package/dist/esm/store.js.map +0 -1
  157. package/src/run-log.ts +0 -224
  158. package/src/store.ts +0 -83
package/src/middleware.ts CHANGED
@@ -1,12 +1,14 @@
1
1
  /**
2
- * `withSandbox(definition)` — the middleware that PROVIDES the
2
+ * `withSandbox(definition, options?)` — the middleware that PROVIDES the
3
3
  * {@link SandboxCapability} a harness adapter requires.
4
4
  *
5
5
  * - `setup`: resume-or-create the sandbox (via the definition's ensure
6
- * algorithm), provide the handle, using the optional SandboxStore/Locks
7
- * capabilities when a persistence middleware supplied them (in-memory
8
- * fallback otherwise). If `fileEvents` is not false, starts a watcher
9
- * that dispatches to sandbox-scoped hooks and forwards to the runtime sink.
6
+ * algorithm), provide the handle, using the durability seams from
7
+ * {@link SandboxMiddlewareOptions} (or, failing that, a bus-provided
8
+ * SandboxInstanceStoreCapability / LocksCapability, then an in-memory
9
+ * fallback). If `fileEvents` is not false, starts a
10
+ * watcher that dispatches to sandbox-scoped hooks and forwards to the runtime
11
+ * sink.
10
12
  * - `onFinish`/`onAbort`/`onError`: stop the watcher, snapshot (`after-run`)
11
13
  * and/or destroy per lifecycle.
12
14
  *
@@ -14,29 +16,54 @@
14
16
  * are emitted by the harness adapter's chatStream (which can yield CUSTOM
15
17
  * chunks), not from here — middleware setup runs before streaming begins.
16
18
  */
17
- import { defineChatMiddleware } from '@tanstack/ai'
18
- import { getSandboxRuntime } from '@tanstack/ai/adapter-internals'
19
19
  import {
20
- LocksCapability,
20
+ defineChatMiddleware,
21
+ provideDetachableRun,
22
+ provideRunDetached,
23
+ wasCancelRequested,
24
+ } from '@tanstack/ai'
25
+ import { InMemoryLockStore, LocksCapability } from '@tanstack/ai/locks'
26
+ import {
27
+ getPendingTurn,
28
+ getRunDisconnect,
29
+ getSandboxRuntime,
30
+ } from '@tanstack/ai/adapter-internals'
31
+ import {
21
32
  SandboxCapability,
22
- SandboxStoreCapability,
23
33
  provideSandbox,
24
34
  provideSandboxPolicy,
25
35
  } from './capabilities'
36
+ import {
37
+ provideSandboxDurability,
38
+ resolveSandboxDurability,
39
+ } from './durability'
40
+ import { SandboxInstanceStoreCapability } from './instance-store'
26
41
  import { computeWorkspaceHash } from './key'
27
42
  import { buildFileHookEvent, resolveFileEvents } from './file-diff'
28
43
  import { ProjectionCapability, provideWorkspaceProjection } from './projection'
29
44
  import { resolveSecret } from './secrets'
45
+ import {
46
+ createToolHistoryRecorder,
47
+ stripObservedToolCalls,
48
+ } from './tool-history'
30
49
  import { watchWorkspace } from './watch'
31
50
  import { DEFAULT_WORKSPACE_ROOT } from './bootstrap'
32
51
  import type { InternalLogger } from '@tanstack/ai/adapter-internals'
52
+ import type { LockStore } from '@tanstack/ai/locks'
33
53
  import type {
34
54
  AbortInfo,
35
55
  ChatMiddlewareContext,
36
56
  DefinedChatMiddleware,
57
+ RunStore,
37
58
  SandboxFileEvent,
38
59
  SandboxFileHookEvent,
39
60
  } from '@tanstack/ai'
61
+ import type {
62
+ SandboxDurabilityOptions,
63
+ SandboxRunDurability,
64
+ } from './durability'
65
+ import type { SandboxInstanceStore } from './instance-store'
66
+ import type { ToolHistoryRecorder } from './tool-history'
40
67
  import type { SandboxHandle } from './contracts'
41
68
  import type {
42
69
  SandboxDefinition,
@@ -47,7 +74,21 @@ import type { SandboxWatchHandle } from './watch'
47
74
 
48
75
  /** Per-request state we need to carry from `setup` to the terminal hooks. */
49
76
  interface SandboxRunState {
50
- handle: SandboxHandle
77
+ /**
78
+ * OPTIONAL because the state is registered BEFORE `definition.ensure()` is
79
+ * awaited, and `ensure` is the slowest thing in the whole run — cloning a repo
80
+ * into a fresh sandbox is minutes wide. That window is where the most common
81
+ * disconnect of all lands (a user starts a run and switches away while the UI
82
+ * still says "starting the sandbox"), so it is the one window the teardown and
83
+ * disconnect hooks most need to be able to act in. Registering only after the
84
+ * handle exists left exactly that window uncovered.
85
+ *
86
+ * Nothing the disconnect path does needs the handle: `detachedSince` and
87
+ * `sandboxKey` come from `ensureCtx`, which is built before `ensure` is called.
88
+ * Only `onFinish`'s snapshot needs it, and that cannot run before `setup` has
89
+ * completed.
90
+ */
91
+ handle?: SandboxHandle
51
92
  ensureCtx: SandboxEnsureContext
52
93
  watcher?: SandboxWatchHandle
53
94
  /** In-flight `enriched.diff()` promises queued by the `fileEvents.diff`
@@ -56,6 +97,18 @@ interface SandboxRunState {
56
97
  pendingDiffs: Array<Promise<void>>
57
98
  /** Logger captured at setup, so terminal hooks can log watcher teardown. */
58
99
  logger?: InternalLogger
100
+ /**
101
+ * Durability resolved once at setup (absent when the run is not durable), so
102
+ * `onAbort` cannot reach a different verdict than the one `setup` published
103
+ * on the capability bus.
104
+ */
105
+ durability?: SandboxRunDurability
106
+ /**
107
+ * Records the harness's own tool calls into the transcript, so a finished run
108
+ * restores its tool cards from the message store instead of only from the (live,
109
+ * rejoin-only) delivery log. See `./tool-history`.
110
+ */
111
+ toolHistory: ToolHistoryRecorder
59
112
  }
60
113
 
61
114
  const runState = new WeakMap<object, SandboxRunState>()
@@ -82,6 +135,84 @@ async function drainWatcher(
82
135
  if (state.watcher) state.logger?.sandbox('sandbox watcher stopped', { phase })
83
136
  }
84
137
 
138
+ /**
139
+ * Record the two facts a later attach and the reaper both need, then publish the
140
+ * detach verdict core reads.
141
+ *
142
+ * Shared by the DISCONNECT subscriber registered in `setup` (the run is still
143
+ * going — the normal case) and `onAbort`'s detach branch (the run is being torn
144
+ * down while detachable), so the two can never write a different shape of detach.
145
+ *
146
+ * GUARDED, and reports failure rather than throwing. `update` is a documented
147
+ * no-op for an unknown runId, so a vanished record does not turn teardown into a
148
+ * throw; a genuinely rejecting store is the caller's to react to — `onAbort` falls
149
+ * through to destroying the sandbox, because a DESTROYED sandbox beats an
150
+ * unreachable one, while the disconnect subscriber has nothing to fall back to
151
+ * (the run is alive and still using the sandbox) and simply leaves the verdict
152
+ * unpublished.
153
+ *
154
+ * The verdict is published ONLY on success. Publishing it after a failed record
155
+ * write would leave core holding the log open for a takeover that can never be
156
+ * found, since nothing in the store points at the run.
157
+ */
158
+ async function recordDetach(
159
+ definition: SandboxDefinition,
160
+ state: SandboxRunState,
161
+ durability: SandboxRunDurability,
162
+ ctx: ChatMiddlewareContext,
163
+ phase: 'disconnect' | 'abort',
164
+ ): Promise<boolean> {
165
+ try {
166
+ // The record already exists: `setup` pre-creates it for every durable run
167
+ // BEFORE `ensure`, precisely so this stamp cannot land on a runId the store has
168
+ // never heard of — `RunStore.update` is a documented no-op for an unknown
169
+ // runId, which is how the detach used to be lost silently (measured against the
170
+ // browser repro: `detached_since` and `sandbox_key` both stayed NULL for a run
171
+ // that had genuinely detached). If it has since vanished, that no-op is the
172
+ // correct outcome and this must not throw.
173
+ await durability.runs.update(ctx.runId, {
174
+ detachedSince: Date.now(),
175
+ sandboxKey: definition.key(state.ensureCtx),
176
+ })
177
+ } catch (error) {
178
+ state.logger?.warn('sandbox detach record write failed', {
179
+ runId: ctx.runId,
180
+ phase,
181
+ error,
182
+ })
183
+ return false
184
+ }
185
+ // Core's durable delivery sink reads this (see `RunDetachedCapability`) and
186
+ // leaves the run's log OPEN instead of appending a synthetic terminal
187
+ // `RUN_ERROR` and closing it — a terminalized log ends a later attach's replay
188
+ // at the prefix and diverges the takeover's journal replay, which recorded a
189
+ // healthy detached run as `'failed'`.
190
+ provideRunDetached(ctx, true)
191
+ return true
192
+ }
193
+
194
+ /**
195
+ * Whether an out-of-band cancel has been recorded for this run, in EITHER band.
196
+ * A user pressing Stop and a user closing the tab produce the IDENTICAL
197
+ * connection close, so intent is never inferred from the disconnect itself: it
198
+ * arrives in-process (the abort reason carried the cancel sentinel) or durably
199
+ * (another host recorded it on the run record).
200
+ */
201
+ async function cancelIntent(
202
+ durability: SandboxRunDurability | undefined,
203
+ runId: string,
204
+ inProcess: boolean,
205
+ ): Promise<boolean> {
206
+ if (inProcess) return true
207
+ if (durability === undefined) return false
208
+ // No guard needed here, and one would be dead code: `wasCancelRequested` already
209
+ // answers `false` for a store read that rejects. That matters on this path,
210
+ // because a rejection escaping into `onAbort` would skip BOTH of its branches at
211
+ // once, leaving a sandbox that is neither reclaimable nor destroyed. The test
212
+ // 'DETACHES when the cancel probe REJECTS' pins the composition.
213
+ return wasCancelRequested(durability.runs, runId)
214
+ }
215
+
85
216
  /** Defensively pull tenant scoping out of the runtime context, if present. */
86
217
  function tenantFrom(
87
218
  context: unknown,
@@ -94,12 +225,72 @@ function tenantFrom(
94
225
  return { userId, orgId }
95
226
  }
96
227
 
97
- function buildEnsureCtx(ctx: ChatMiddlewareContext): SandboxEnsureContext {
228
+ /**
229
+ * Durability seams for a sandboxed run. Both are optional; each independently
230
+ * falls back to a process-lifetime in-memory default, which is correct for a
231
+ * single process but NOT across replicas.
232
+ */
233
+ export interface SandboxMiddlewareOptions<TOffset extends string = string> {
234
+ /**
235
+ * Durable instance map (which provider sandbox to resume for a key). Pass
236
+ * your own store to make resume survive across processes/replicas.
237
+ *
238
+ * Takes precedence over a store provided on the capability bus (see
239
+ * `provideSandboxInstanceStore`), so the call site wins over ambient wiring.
240
+ */
241
+ instances?: SandboxInstanceStore
242
+ /**
243
+ * Distributed lock serializing resume-or-create for one key. Needed for
244
+ * multi-replica correctness so two concurrent runs don't both create.
245
+ *
246
+ * Prefer `withLocks` from `@tanstack/ai/locks` when other middleware also
247
+ * needs the lock; use this option to scope one to this sandbox. Takes
248
+ * precedence over a bus-provided lock.
249
+ */
250
+ locks?: LockStore
251
+ /**
252
+ * Run lifecycle records. Pair with `durability.adapter` to make a run
253
+ * DETACHABLE: a client disconnect then leaves the agent running and records
254
+ * `detachedSince` instead of destroying the sandbox.
255
+ *
256
+ * Pass the SAME store chat persistence uses (`persistence.stores.runs`) so
257
+ * one record describes the run instead of two that can disagree.
258
+ *
259
+ * Defaults to `undefined`: an app that passes neither this nor `durability`
260
+ * keeps today's destroy-on-disconnect behavior exactly.
261
+ */
262
+ runs?: RunStore
263
+ /**
264
+ * Delivery durability for the run's event log, plus the journal and detach
265
+ * knobs. Requires `runs`; either alone is not durable.
266
+ *
267
+ * `TOffset` is inferred from the adapter passed here, so a branded-cursor
268
+ * backend (`durableStream`) wires without a cast and without the call site
269
+ * ever naming the parameter.
270
+ */
271
+ durability?: SandboxDurabilityOptions<TOffset>
272
+ }
273
+
274
+ /**
275
+ * Resolve the ensure seams. Precedence is explicit option → capability bus →
276
+ * (in `ensure`) the in-memory fallback. The option wins because it is visible
277
+ * at the call site; the bus remains for platform/framework injection.
278
+ */
279
+ function buildEnsureCtx(
280
+ ctx: ChatMiddlewareContext,
281
+ // Narrowed to the two seams it reads rather than taking the whole options
282
+ // object: `SandboxMiddlewareOptions` is now generic in the durability offset,
283
+ // and `SandboxMiddlewareOptions<TOffset>` is not assignable to
284
+ // `SandboxMiddlewareOptions<string>`. Both members here are offset-free, so
285
+ // the narrowing keeps this helper independent of that parameter entirely.
286
+ options: Pick<SandboxMiddlewareOptions, 'instances' | 'locks'> | undefined,
287
+ ): SandboxEnsureContext {
98
288
  return {
99
289
  threadId: ctx.threadId,
100
290
  runId: ctx.runId,
101
- store: ctx.getOptional(SandboxStoreCapability),
102
- locks: ctx.getOptional(LocksCapability),
291
+ store:
292
+ options?.instances ?? ctx.getOptional(SandboxInstanceStoreCapability),
293
+ locks: options?.locks ?? ctx.getOptional(LocksCapability),
103
294
  tenant: tenantFrom(ctx.context),
104
295
  signal: ctx.signal,
105
296
  }
@@ -141,8 +332,9 @@ async function dispatchDefinitionHooks(
141
332
  }
142
333
  }
143
334
 
144
- export function withSandbox(
335
+ export function withSandbox<TOffset extends string = string>(
145
336
  definition: SandboxDefinition,
337
+ options?: SandboxMiddlewareOptions<TOffset>,
146
338
  ): DefinedChatMiddleware<
147
339
  unknown,
148
340
  readonly [],
@@ -153,14 +345,29 @@ export function withSandbox(
153
345
  provides: [SandboxCapability, ProjectionCapability],
154
346
  // SandboxPolicyCapability is provided conditionally (only when the
155
347
  // definition has a policy), so it is intentionally NOT declared here —
156
- // consumers read it via `getOptional`.
157
- optionalRequires: [SandboxStoreCapability, LocksCapability],
348
+ // consumers read it via `getOptional`. SandboxDurabilityCapability and
349
+ // DetachableRunCapability are conditional for the same reason (only when
350
+ // `runs` + `durability` are both wired), so they are intentionally NOT
351
+ // declared here either.
352
+ optionalRequires: [SandboxInstanceStoreCapability, LocksCapability],
158
353
 
159
354
  async setup(ctx) {
160
- const ensureCtx = buildEnsureCtx(ctx)
161
- const handle = await definition.ensure(ensureCtx)
162
- provideSandbox(ctx, handle)
163
- if (definition.policy) provideSandboxPolicy(ctx, definition.policy)
355
+ const ensureCtx = buildEnsureCtx(ctx, options)
356
+
357
+ // Resolving here (not lazily on the abort path) is what keeps `setup` and
358
+ // `onAbort` on one verdict: the payload the bus carries is the same object
359
+ // the teardown path consults.
360
+ // `TOffset` is passed explicitly: `options` is possibly `undefined` here,
361
+ // so inference has nothing to work from on that branch and would fall
362
+ // back to the `= string` default, re-erecting the very wall this
363
+ // parameter exists to remove.
364
+ const durability = resolveSandboxDurability<TOffset>(options)
365
+ if (durability !== undefined) {
366
+ provideSandboxDurability(ctx, durability)
367
+ // A neutral boolean core owns, so `@tanstack/ai-persistence` can ask
368
+ // "is this run detachable?" without depending on this package.
369
+ provideDetachableRun(ctx, true)
370
+ }
164
371
 
165
372
  // Pull the runtime (and its logger) up front so `baseSha` capture and
166
373
  // hook dispatch below can log through the same `sandbox`/`errors`
@@ -168,6 +375,166 @@ export function withSandbox(
168
375
  const runtime = getSandboxRuntime(ctx, { optional: true })
169
376
  const logger = runtime?.logger
170
377
 
378
+ // REGISTER THE RUN STATE NOW — before `definition.ensure()`, not merely
379
+ // before the end of `setup`.
380
+ //
381
+ // `onAbort` and the disconnect subscriber both need this state, so until
382
+ // this map is populated they are silent no-ops. `ensure` is the LONGEST
383
+ // await in the entire run (create a sandbox, clone a repo — minutes), and it
384
+ // is where the most common disconnect of all lands: a user starts a run and
385
+ // switches away while the UI still says "starting the sandbox". Registering
386
+ // after `ensure` returned still left that whole window uncovered.
387
+ //
388
+ // Leaving it uncovered loses every teardown behavior at once: no
389
+ // `detachedSince`/`sandboxKey`, so `listReclaimable` can never surface the
390
+ // run and the reaper can never reclaim it; no `definition.destroy`, so the
391
+ // sandbox leaks; and no detach verdict for core to read.
392
+ //
393
+ // Everything those hooks read is already resolved above: the ensure context
394
+ // (which is all `definition.key` needs), the durability verdict, and the
395
+ // logger. The fields discovered later (`handle`, `watcher`) are ASSIGNED onto
396
+ // this same object as they become available, so the teardown path always
397
+ // sees the most complete state that exists at the moment it runs.
398
+ const state: SandboxRunState = {
399
+ ensureCtx,
400
+ pendingDiffs: [],
401
+ toolHistory: createToolHistoryRecorder(),
402
+ ...(logger ? { logger } : {}),
403
+ ...(durability ? { durability } : {}),
404
+ }
405
+ runState.set(ctx, state)
406
+
407
+ // MAKE THE RUN FINDABLE BEFORE `ensure`, not after the run finally streams.
408
+ //
409
+ // Chat persistence creates the run record from `onConfig`, which runs after
410
+ // EVERY middleware `setup` — so for the whole of `definition.ensure` (create a
411
+ // sandbox, clone a repo: minutes) the run has no record at all, and
412
+ // `findActiveRun` answers "no active run" for a run that is demonstrably
413
+ // starting. Measured: a status sidebar read straight off `findActiveRun`
414
+ // reported `idle` for 6.5 minutes while the sandbox was being built, and a
415
+ // client returning to the thread in that window had nothing to tell it a run
416
+ // was in flight — so it rendered an empty pane instead of "starting sandbox".
417
+ //
418
+ // A crash in the same window is worse: no record means `listReclaimable` can
419
+ // never surface the run, so the sandbox leaks with no recovery path.
420
+ //
421
+ // `createOrResume` is idempotent and never resurrects a finished run, so
422
+ // persistence's own later call stays correct and simply finds this record.
423
+ if (durability !== undefined) {
424
+ try {
425
+ await durability.runs.createOrResume({
426
+ runId: ctx.runId,
427
+ threadId: ctx.threadId,
428
+ startedAt: Date.now(),
429
+ })
430
+ } catch (error) {
431
+ // Best-effort: a store blip must not stop a run that is otherwise fine.
432
+ // The run is simply invisible until persistence's own `onConfig` call.
433
+ logger?.warn('sandbox run record pre-create failed', {
434
+ runId: ctx.runId,
435
+ error,
436
+ })
437
+ }
438
+
439
+ // NO ATTACH MARKER HERE. A joiner does need a chunk in the log before the
440
+ // harness has emitted anything — an empty log fails every joiner's
441
+ // fast-fail (`memoryStream`'s first-chunk deadline, the client's rejoin
442
+ // connect deadline) and flushes no HTTP headers, so a reload during
443
+ // `ensure` reads a live run as gone. Core does it: a fresh durable producer
444
+ // appends `RUN_ACCEPTED_EVENT` before the producer stream is first pulled,
445
+ // for EVERY durable run rather than only sandboxed ones, and never on an
446
+ // attach. A second marker from here would only land mid-stream in a run
447
+ // that is already producing.
448
+
449
+ // STORE THE USER'S TURN NOW, before `ensure` takes minutes.
450
+ //
451
+ // Chat persistence stores it from `onStart`, which runs after every
452
+ // middleware `setup` — so without this the thread holds NOTHING for the
453
+ // whole sandbox build. Measured: a reload during the build asked the server
454
+ // for the conversation and got `{"messages":[],…}`, so the user saw no sign
455
+ // of the message they had just sent, and a second device saw an empty
456
+ // thread.
457
+ //
458
+ // The persistence layer owns WHAT gets stored (see `PendingTurnCapability`):
459
+ // `saveThread` replaces the thread, so deciding the list here would risk
460
+ // deleting the history. Absent when the app wires no persistence, which is
461
+ // simply a run with no transcript to store.
462
+ try {
463
+ await getPendingTurn(ctx, { optional: true })?.snapshot()
464
+ } catch (error) {
465
+ // Best-effort: the run is still worth doing, and `onStart` stores the
466
+ // turn again once setup completes.
467
+ logger?.warn('sandbox pending-turn snapshot failed', {
468
+ runId: ctx.runId,
469
+ error,
470
+ })
471
+ }
472
+ }
473
+
474
+ // SUBSCRIBE BEFORE `ensure`, for the same reason the state is registered
475
+ // before it: `ensure` is the minutes-wide await a disconnect actually lands
476
+ // in. Core calls back immediately if the socket has already closed, so
477
+ // subscribing here cannot miss a disconnect that beat us to it.
478
+ //
479
+ // This is what makes a durable run SURVIVE losing its viewer. The only route
480
+ // a disconnect previously had into this middleware was the application
481
+ // mirroring `request.signal` into `chat()`'s `abortController` — which aborts
482
+ // the run, so `chat()` returned right after this `setup` and the harness
483
+ // adapter's `chatStream` was never called: the agent in the sandbox we just
484
+ // spent minutes creating was NEVER LAUNCHED, and no takeover could recover it
485
+ // because an agent that never ran wrote no journal to replay.
486
+ if (durability !== undefined && durability.detachOnDisconnect) {
487
+ getRunDisconnect(ctx, { optional: true })?.subscribe(async () => {
488
+ // BOOKKEEPING ONLY — the run is still executing. Deliberately absent:
489
+ // `drainWatcher` (would blind a live agent's file events for the whole
490
+ // remainder) and `definition.destroy` (the run is still using the
491
+ // sandbox). Both belong to the terminal hooks, which still run exactly
492
+ // once afterwards.
493
+ //
494
+ // A run with a cancel already recorded is left alone: that is `onAbort`'s
495
+ // path, and stamping `detachedSince` on a deliberately-stopped run would
496
+ // hand it to the reaper as reclaimable work.
497
+ if (await cancelIntent(durability, ctx.runId, false)) return
498
+ if (
499
+ await recordDetach(definition, state, durability, ctx, 'disconnect')
500
+ ) {
501
+ state.logger?.sandbox(
502
+ 'sandbox run detached on disconnect; the run continues',
503
+ { runId: ctx.runId },
504
+ )
505
+ }
506
+ })
507
+ }
508
+
509
+ const handle = await definition.ensure(ensureCtx)
510
+ // MUTATE, don't re-`set`: a disconnect that landed during `ensure` already
511
+ // captured this object.
512
+ state.handle = handle
513
+ provideSandbox(ctx, handle)
514
+ if (definition.policy) provideSandboxPolicy(ctx, definition.policy)
515
+
516
+ // Deliberately placed AFTER `logger` is in scope rather than next to the
517
+ // `provideSandboxDurability` call above — there is no logger to warn
518
+ // through until the runtime has been read.
519
+ //
520
+ // `ensureCtx.locks === undefined` counts as in-memory: `defineSandbox`'s
521
+ // `ensure` falls back to a process-lifetime `InMemoryLockStore` when no
522
+ // lock is wired, so an unwired lock has exactly the deficiency being
523
+ // warned about — it is the MOST in-memory case, not an exempt one.
524
+ if (
525
+ durability !== undefined &&
526
+ (ensureCtx.locks === undefined ||
527
+ ensureCtx.locks instanceof InMemoryLockStore)
528
+ ) {
529
+ logger?.warn(
530
+ 'sandbox durability is wired over an InMemoryLockStore: run claims are ' +
531
+ 'serialized within this process only and the lease never signals loss, ' +
532
+ 'so two hosts can drive one run and duplicate its event log. Use a ' +
533
+ 'distributed LockStore via withLocks for any multi-replica deploy.',
534
+ { runId: ctx.runId },
535
+ )
536
+ }
537
+
171
538
  const watchRoot = definition.workspace?.root ?? DEFAULT_WORKSPACE_ROOT
172
539
  let baseSha = ''
173
540
  try {
@@ -229,7 +596,11 @@ export function withSandbox(
229
596
  await hooks?.onReady?.(handle)
230
597
 
231
598
  const fe = resolveFileEvents(definition.fileEvents)
232
- const pendingDiffs: Array<Promise<void>> = []
599
+ // THE SAME array the run state already holds, not a fresh one. The watcher
600
+ // callback below closes over this reference, and `drainWatcher` awaits
601
+ // `state.pendingDiffs` — a second array would silently drop every in-flight
602
+ // diff from the teardown drain.
603
+ const pendingDiffs = state.pendingDiffs
233
604
  let watcher: SandboxWatchHandle | undefined
234
605
  if (fe.enabled) {
235
606
  watcher = await watchWorkspace(handle, {
@@ -274,13 +645,38 @@ export function withSandbox(
274
645
  })
275
646
  }
276
647
 
277
- runState.set(ctx, {
278
- handle,
279
- ensureCtx,
280
- pendingDiffs,
281
- ...(watcher ? { watcher } : {}),
282
- ...(logger !== undefined ? { logger } : {}),
283
- })
648
+ // MUTATE the object registered above rather than `set`-ing a second one: an
649
+ // abort that landed mid-setup already captured a reference to it (and may
650
+ // already be draining `pendingDiffs`), so replacing the entry would hand the
651
+ // teardown path a different object than the watcher writes into.
652
+ // `pendingDiffs` needs no copying — it IS `state.pendingDiffs`.
653
+ if (watcher) state.watcher = watcher
654
+ },
655
+
656
+ // Keep the recorded tool history OUT of the request to the model. It is stored
657
+ // history for the next turn, it names tools the provider was never given, and one
658
+ // triage-sized run is hundreds of kilobytes — so replaying it is wasteful at best
659
+ // and rejected at worst. `ctx.messages` keeps it (that is what gets stored and
660
+ // rendered); only `config.messages` loses it.
661
+ onConfig(_ctx, config) {
662
+ const messages = stripObservedToolCalls(config.messages)
663
+ if (messages.length === config.messages.length) return
664
+ return { messages }
665
+ },
666
+
667
+ // The engine re-syncs `middlewareCtx.messages` from its own array once per agent
668
+ // iteration, which drops whatever the recorder appended during the previous
669
+ // iteration's stream. Restoring it here — AFTER that sync — is what makes a
670
+ // multi-iteration run keep its full history without depending on where this
671
+ // middleware sits relative to persistence in the middleware array.
672
+ onIteration(ctx) {
673
+ runState.get(ctx)?.toolHistory.reconcile(ctx)
674
+ },
675
+
676
+ // Record the harness's own tool calls as transcript messages. Observe only:
677
+ // returning nothing passes every chunk through untouched.
678
+ onChunk(ctx, chunk) {
679
+ runState.get(ctx)?.toolHistory.observe(chunk, ctx)
284
680
  },
285
681
 
286
682
  async onFinish(ctx) {
@@ -288,13 +684,20 @@ export function withSandbox(
288
684
  if (!state) return
289
685
  const { handle, ensureCtx } = state
290
686
 
687
+ // Last chance before persistence writes the transcript. Only matters if a
688
+ // config sync landed after the final tool chunk; the recorder is idempotent, so
689
+ // in the normal case this changes nothing.
690
+ state.toolHistory.reconcile(ctx)
691
+
291
692
  await drainWatcher(state, 'finish')
292
693
 
293
694
  const lifecycle = definition.lifecycle
294
695
 
696
+ // `handle` is absent only if `setup` never got past `definition.ensure`, in
697
+ // which case there is no sandbox to snapshot.
295
698
  if (
296
699
  lifecycle?.snapshot === 'after-run' &&
297
- handle.capabilities.snapshots &&
700
+ handle?.capabilities.snapshots &&
298
701
  handle.snapshot
299
702
  ) {
300
703
  const snapshot = await handle.snapshot(`after-run-${ctx.runId}`)
@@ -318,12 +721,49 @@ export function withSandbox(
318
721
  }
319
722
  },
320
723
 
321
- async onAbort(ctx, _info: AbortInfo) {
724
+ async onAbort(ctx, info: AbortInfo) {
322
725
  const state = runState.get(ctx)
323
726
  if (!state) return
324
727
 
728
+ // First on BOTH branches: a diff still in flight must be drained whether
729
+ // the sandbox is about to be destroyed or merely detached, or the final
730
+ // file's diff is dropped.
325
731
  await drainWatcher(state, 'abort')
326
732
 
733
+ const durability = state.durability
734
+ const cancelled = await cancelIntent(
735
+ durability,
736
+ ctx.runId,
737
+ info.cancelRequested === true,
738
+ )
739
+
740
+ if (
741
+ durability !== undefined &&
742
+ !cancelled &&
743
+ durability.detachOnDisconnect
744
+ ) {
745
+ // DETACH on the teardown path. Reached when the run is aborted for a
746
+ // reason that is NOT an out-of-band cancel while detachable — a genuine
747
+ // stop from elsewhere, or a host going down. The ordinary disconnect is
748
+ // handled by the disconnect subscriber in `setup`, which does not end the
749
+ // run at all.
750
+ //
751
+ // On a failed record write this branch is ABANDONED for the destroy one
752
+ // below, because a rejection here is the worst shape available: the
753
+ // verdict is unpublished, so core terminalizes the log and records a
754
+ // healthy detached run as failed; `detachedSince`/`sandboxKey` are
755
+ // unwritten, so `listReclaimable` can never surface the run and
756
+ // `reapDetachedRuns` can never reclaim it. A DESTROYED sandbox beats an
757
+ // unreachable one — the same reasoning `drainWatcher` applies to its own
758
+ // guarded `stop()`.
759
+ if (await recordDetach(definition, state, durability, ctx, 'abort')) {
760
+ return
761
+ }
762
+ await definition.destroy(state.ensureCtx)
763
+ await definition.hooks?.onDestroy?.()
764
+ return
765
+ }
766
+
327
767
  // ALWAYS tear down on an explicit abort, regardless of `destroyOnComplete`.
328
768
  // The in-sandbox agent process is not killed by closing its IO stream
329
769
  // (e.g. a Docker exec survives client disconnect), so the only reliable way