@tanstack/ai-sandbox 0.2.3 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/dist/esm/agents-file.js +53 -34
  2. package/dist/esm/agents-file.js.map +1 -1
  3. package/dist/esm/align.d.ts +121 -0
  4. package/dist/esm/align.js +197 -0
  5. package/dist/esm/align.js.map +1 -0
  6. package/dist/esm/approvals.js +63 -29
  7. package/dist/esm/approvals.js.map +1 -1
  8. package/dist/esm/attach-preflight.d.ts +85 -0
  9. package/dist/esm/attach-preflight.js +189 -0
  10. package/dist/esm/attach-preflight.js.map +1 -0
  11. package/dist/esm/bootstrap.js +103 -117
  12. package/dist/esm/bootstrap.js.map +1 -1
  13. package/dist/esm/bridge-events.js +96 -71
  14. package/dist/esm/bridge-events.js.map +1 -1
  15. package/dist/esm/capabilities.d.ts +0 -5
  16. package/dist/esm/capabilities.js +32 -28
  17. package/dist/esm/capabilities.js.map +1 -1
  18. package/dist/esm/chunk-identity.d.ts +52 -0
  19. package/dist/esm/chunk-identity.js +102 -0
  20. package/dist/esm/chunk-identity.js.map +1 -0
  21. package/dist/esm/claim.d.ts +187 -0
  22. package/dist/esm/claim.js +349 -0
  23. package/dist/esm/claim.js.map +1 -0
  24. package/dist/esm/contracts.d.ts +13 -0
  25. package/dist/esm/driver.d.ts +83 -0
  26. package/dist/esm/driver.js +138 -0
  27. package/dist/esm/driver.js.map +1 -0
  28. package/dist/esm/durability.d.ts +263 -0
  29. package/dist/esm/durability.js +230 -0
  30. package/dist/esm/durability.js.map +1 -0
  31. package/dist/esm/errors.js +28 -24
  32. package/dist/esm/errors.js.map +1 -1
  33. package/dist/esm/file-diff.js +151 -135
  34. package/dist/esm/file-diff.js.map +1 -1
  35. package/dist/esm/git-exec.js +51 -62
  36. package/dist/esm/git-exec.js.map +1 -1
  37. package/dist/esm/harness-cwd.js +24 -19
  38. package/dist/esm/harness-cwd.js.map +1 -1
  39. package/dist/esm/index.d.ts +30 -8
  40. package/dist/esm/index.js +23 -91
  41. package/dist/esm/instance-store.d.ts +88 -0
  42. package/dist/esm/instance-store.js +67 -0
  43. package/dist/esm/instance-store.js.map +1 -0
  44. package/dist/esm/journal-bytes.d.ts +67 -0
  45. package/dist/esm/journal-bytes.js +110 -0
  46. package/dist/esm/journal-bytes.js.map +1 -0
  47. package/dist/esm/journal-reader.d.ts +66 -0
  48. package/dist/esm/journal-reader.js +228 -0
  49. package/dist/esm/journal-reader.js.map +1 -0
  50. package/dist/esm/journal-sweep.d.ts +113 -0
  51. package/dist/esm/journal-sweep.js +309 -0
  52. package/dist/esm/journal-sweep.js.map +1 -0
  53. package/dist/esm/journal.d.ts +542 -0
  54. package/dist/esm/journal.js +679 -0
  55. package/dist/esm/journal.js.map +1 -0
  56. package/dist/esm/key.js +36 -33
  57. package/dist/esm/key.js.map +1 -1
  58. package/dist/esm/middleware.d.ts +50 -2
  59. package/dist/esm/middleware.js +335 -208
  60. package/dist/esm/middleware.js.map +1 -1
  61. package/dist/esm/ngrok.js +75 -49
  62. package/dist/esm/ngrok.js.map +1 -1
  63. package/dist/esm/policy.js +43 -34
  64. package/dist/esm/policy.js.map +1 -1
  65. package/dist/esm/projection.js +16 -8
  66. package/dist/esm/projection.js.map +1 -1
  67. package/dist/esm/reap.d.ts +238 -0
  68. package/dist/esm/reap.js +355 -0
  69. package/dist/esm/reap.js.map +1 -0
  70. package/dist/esm/reclaim.d.ts +84 -0
  71. package/dist/esm/reclaim.js +106 -0
  72. package/dist/esm/reclaim.js.map +1 -0
  73. package/dist/esm/remote-tools.js +73 -62
  74. package/dist/esm/remote-tools.js.map +1 -1
  75. package/dist/esm/run.d.ts +93 -25
  76. package/dist/esm/run.js +274 -79
  77. package/dist/esm/run.js.map +1 -1
  78. package/dist/esm/runner.d.ts +119 -2
  79. package/dist/esm/runner.js +270 -51
  80. package/dist/esm/runner.js.map +1 -1
  81. package/dist/esm/sandbox.d.ts +3 -2
  82. package/dist/esm/sandbox.js +139 -123
  83. package/dist/esm/sandbox.js.map +1 -1
  84. package/dist/esm/secrets.js +39 -47
  85. package/dist/esm/secrets.js.map +1 -1
  86. package/dist/esm/setup-plan.js +22 -14
  87. package/dist/esm/setup-plan.js.map +1 -1
  88. package/dist/esm/shell.d.ts +8 -0
  89. package/dist/esm/shell.js +197 -158
  90. package/dist/esm/shell.js.map +1 -1
  91. package/dist/esm/testkit/conformance.d.ts +16 -0
  92. package/dist/esm/testkit/conformance.js +97 -0
  93. package/dist/esm/testkit/conformance.js.map +1 -0
  94. package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
  95. package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
  96. package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
  97. package/dist/esm/testkit/journal-conformance.d.ts +51 -0
  98. package/dist/esm/testkit/journal-conformance.js +378 -0
  99. package/dist/esm/testkit/journal-conformance.js.map +1 -0
  100. package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
  101. package/dist/esm/testkit/reaper-conformance.js +847 -0
  102. package/dist/esm/testkit/reaper-conformance.js.map +1 -0
  103. package/dist/esm/testkit/shell-spawn.d.ts +2 -0
  104. package/dist/esm/testkit/shell-spawn.js +60 -0
  105. package/dist/esm/testkit/shell-spawn.js.map +1 -0
  106. package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
  107. package/dist/esm/testkit/takeover-conformance.js +685 -0
  108. package/dist/esm/testkit/takeover-conformance.js.map +1 -0
  109. package/dist/esm/tool-bridge.js +227 -180
  110. package/dist/esm/tool-bridge.js.map +1 -1
  111. package/dist/esm/tool-history.d.ts +62 -0
  112. package/dist/esm/tool-history.js +171 -0
  113. package/dist/esm/tool-history.js.map +1 -0
  114. package/dist/esm/watch.js +310 -236
  115. package/dist/esm/watch.js.map +1 -1
  116. package/dist/esm/workspace.d.ts +1 -1
  117. package/dist/esm/workspace.js +49 -28
  118. package/dist/esm/workspace.js.map +1 -1
  119. package/package.json +16 -6
  120. package/skills/ai-sandbox/SKILL.md +658 -20
  121. package/src/align.ts +297 -0
  122. package/src/attach-preflight.ts +292 -0
  123. package/src/capabilities.ts +4 -13
  124. package/src/chunk-identity.ts +154 -0
  125. package/src/claim.ts +479 -0
  126. package/src/contracts.ts +13 -0
  127. package/src/driver.ts +205 -0
  128. package/src/durability.ts +380 -0
  129. package/src/index.ts +212 -27
  130. package/src/instance-store.ts +122 -0
  131. package/src/journal-bytes.ts +136 -0
  132. package/src/journal-reader.ts +359 -0
  133. package/src/journal-sweep.ts +406 -0
  134. package/src/journal.ts +875 -0
  135. package/src/middleware.ts +470 -30
  136. package/src/reap.ts +723 -0
  137. package/src/reclaim.ts +191 -0
  138. package/src/run.ts +365 -75
  139. package/src/runner.ts +347 -3
  140. package/src/sandbox.ts +38 -8
  141. package/src/shell.ts +106 -38
  142. package/src/testkit/conformance.ts +117 -0
  143. package/src/testkit/durable-run-fields-conformance.ts +147 -0
  144. package/src/testkit/journal-conformance.ts +676 -0
  145. package/src/testkit/reaper-conformance.ts +1201 -0
  146. package/src/testkit/shell-spawn.ts +67 -0
  147. package/src/testkit/takeover-conformance.ts +1040 -0
  148. package/src/tool-history.ts +245 -0
  149. package/src/workspace.ts +1 -1
  150. package/dist/esm/index.js.map +0 -1
  151. package/dist/esm/run-log.d.ts +0 -81
  152. package/dist/esm/run-log.js +0 -107
  153. package/dist/esm/run-log.js.map +0 -1
  154. package/dist/esm/store.d.ts +0 -53
  155. package/dist/esm/store.js +0 -34
  156. package/dist/esm/store.js.map +0 -1
  157. package/src/run-log.ts +0 -224
  158. package/src/store.ts +0 -83
package/src/driver.ts ADDED
@@ -0,0 +1,205 @@
1
+ /**
2
+ * The convenience that turns core's *injected* takeover seams into this
3
+ * package's real ones.
4
+ *
5
+ * `@tanstack/ai`'s `RunDriverOptions` deliberately takes `claim` and `pipe` as
6
+ * functions instead of importing them: {@link withRunClaim} and
7
+ * {@link pipeToRunLog} live here, and core must not depend on this package to
8
+ * serve a plain chat run. {@link sandboxRunDriver} fills both in so an
9
+ * application writes four fields instead of six, and — more importantly — so
10
+ * the *fencing* is wired correctly by construction rather than by every caller
11
+ * remembering to.
12
+ *
13
+ * WHAT IS EASY TO GET WRONG HERE, and therefore what this module exists to
14
+ * make impossible:
15
+ *
16
+ * 1. **Carrying the real epoch into `pipe`.** Core's `pipe` receives only
17
+ * `{ runId, threadId, signal }` — no epoch — because core has no concept of
18
+ * one. But {@link fenceDurability} needs the epoch this driver actually
19
+ * acquired: a hardcoded epoch (say `0`) is not a weaker fence, it is a
20
+ * permanently *tripped* one, since `withRunClaim` bumps `driverEpoch` to at
21
+ * least `1` before `fn` ever runs, so `observed > claim.epoch` holds on the
22
+ * very first append and EVERY takeover fails. The claim is therefore
23
+ * captured in a closure by the `claim` wrapper and read back by `pipe`.
24
+ * 2. **Fencing `close()`.** {@link fenceDurability} wraps only `append` for the
25
+ * reason spelled out in `claim.ts`: `close()` runs on every teardown path,
26
+ * including the teardown caused by losing the claim, and a fenced `close`
27
+ * would wedge the record at `'running'` with every live tailer parked
28
+ * forever. This module must not add a second fence around it.
29
+ * 2b. **Fencing only ONE of the two authoritative seams.** A run's facts live in
30
+ * its log *and* in its record, and `pipeToRunLog` reacts to a refused append
31
+ * by writing a terminal record — so wrapping the log alone just moves the harm
32
+ * from "a dead host poisons the successor's stream" to "a dead host marks the
33
+ * successor's live run failed". {@link fenceRunStore} must be wired here too,
34
+ * over the SAME claim, which is what makes the two fences share one latch.
35
+ * 3. **Skipping quiescence.** The successor's first append must come after the
36
+ * stored log has stopped growing, so a predecessor still writing is observed
37
+ * rather than raced. The gate belongs inside `pipe`, before `pipeToRunLog`
38
+ * takes its first `snapshot`.
39
+ */
40
+ import { pipeToRunLog } from './run'
41
+ import {
42
+ DEFAULT_FENCE_QUIET_MS,
43
+ awaitLogQuiescence,
44
+ fenceDurability,
45
+ fenceRunStore,
46
+ withRunClaim,
47
+ } from './claim'
48
+ import type { RunClaim } from './claim'
49
+ import type { InternalLogger } from '@tanstack/ai/adapter-internals'
50
+ import type { LockStore } from '@tanstack/ai/locks'
51
+ import type {
52
+ RunDriverOptions,
53
+ RunStore,
54
+ StreamChunk,
55
+ StreamDurability,
56
+ } from '@tanstack/ai'
57
+
58
+ export interface SandboxRunDriverOptions<TOffset extends string = string> {
59
+ /** The attach request; core reads its run id with `resolveResumeRunId`. */
60
+ request: Request
61
+ runs: RunStore
62
+ locks: LockStore
63
+ /**
64
+ * Per-run event log factory, the same shape `RunDeps.durability` takes — a
65
+ * `StreamDurability` is bound to one run, so the log is resolved FROM the
66
+ * `runId` rather than handed in pre-bound.
67
+ *
68
+ * Generic in the offset type, defaulted to `string` so an existing call site
69
+ * needs no change. Hardcoding the default made a branded-cursor backend
70
+ * unusable here: `durableStream` returns
71
+ * `StreamDurability<DurableStreamOffset>`, which is not assignable to
72
+ * `StreamDurability<string>` because `read` is contravariant in its offset.
73
+ */
74
+ durability: (runId: string) => StreamDurability<TOffset>
75
+ /** Produce the run's remaining events. Called only once the claim is held. */
76
+ drive: (input: {
77
+ runId: string
78
+ threadId: string
79
+ signal: AbortSignal
80
+ }) => AsyncIterable<StreamChunk>
81
+ /** Quiescence window; defaults to {@link DEFAULT_FENCE_QUIET_MS}. */
82
+ fenceQuietMs?: number
83
+ /** Platform keep-alive (e.g. `ctx.waitUntil`) for the background drive. */
84
+ waitUntil?: (promise: Promise<unknown>) => void
85
+ logger?: InternalLogger
86
+ }
87
+
88
+ /**
89
+ * `pipe` ran without a held claim. Not a recoverable condition: it means the
90
+ * returned options object was taken apart and `pipe` called outside `claim`, so
91
+ * there is no epoch to fence with and no lease guaranteeing exclusivity. Any
92
+ * append made in that state is exactly the duplicate-write bug the claim exists
93
+ * to prevent, so this fails loudly rather than appending unfenced.
94
+ */
95
+ export class RunDriverPipeOutsideClaimError extends Error {
96
+ constructor(readonly runId: string) {
97
+ super(
98
+ `run ${runId}: sandboxRunDriver.pipe was called outside its claim, so the driver epoch is unknown; call it from within the claim callback`,
99
+ )
100
+ this.name = 'RunDriverPipeOutsideClaimError'
101
+ }
102
+ }
103
+
104
+ /**
105
+ * Fill in a core `driver` block with this package's claim and run log.
106
+ *
107
+ * `drive` receives an `AbortSignal` — the driver owns the abort, so it hands
108
+ * out a signal rather than a controller — but `chat()` takes an
109
+ * `AbortController`. Mirror one onto the other, exactly as
110
+ * {@link https://tanstack.com/ai/latest/docs/sandbox/takeover | Takeover & Detached Runs}'s
111
+ * `controllerFor` does, so a lost claim actually stops the drive.
112
+ *
113
+ * @example
114
+ * ```typescript
115
+ * function controllerFor(signal: AbortSignal): AbortController {
116
+ * const controller = new AbortController()
117
+ * const abort = (): void => controller.abort(signal.reason)
118
+ * if (signal.aborted) abort()
119
+ * else signal.addEventListener('abort', abort, { once: true })
120
+ * return controller
121
+ * }
122
+ *
123
+ * export async function GET(request: Request) {
124
+ * return resumeServerSentEventsResponse({
125
+ * adapter: memoryStream(request),
126
+ * driver: sandboxRunDriver({
127
+ * request,
128
+ * runs,
129
+ * locks,
130
+ * durability: (runId) => logFor(runId),
131
+ * drive: ({ runId, threadId, signal }) =>
132
+ * chat({
133
+ * ...config,
134
+ * runId,
135
+ * threadId,
136
+ * abortController: controllerFor(signal),
137
+ * }),
138
+ * }),
139
+ * })
140
+ * }
141
+ * ```
142
+ */
143
+ export function sandboxRunDriver<TOffset extends string = string>(
144
+ input: SandboxRunDriverOptions<TOffset>,
145
+ ): RunDriverOptions {
146
+ const fenceQuietMs = input.fenceQuietMs ?? DEFAULT_FENCE_QUIET_MS
147
+ // The seam between core's `claim` and core's `pipe`. One options object serves
148
+ // one attach request and therefore one run, so a single slot is enough; it is
149
+ // cleared on the way out so a `pipe` after the claim released cannot reuse a
150
+ // stale epoch.
151
+ let current: RunClaim | undefined
152
+
153
+ return {
154
+ request: input.request,
155
+ runs: input.runs,
156
+ locks: input.locks,
157
+ drive: input.drive,
158
+ claim: (claimInput, fn) =>
159
+ withRunClaim(
160
+ {
161
+ ...claimInput,
162
+ fenceQuietMs,
163
+ ...(input.logger === undefined ? {} : { logger: input.logger }),
164
+ },
165
+ async (claim) => {
166
+ const previous = current
167
+ current = claim
168
+ try {
169
+ return await fn(claim)
170
+ } finally {
171
+ current = previous
172
+ }
173
+ },
174
+ ),
175
+ pipe: async (stream, i) => {
176
+ const claim = current
177
+ if (claim === undefined) {
178
+ throw new RunDriverPipeOutsideClaimError(i.runId)
179
+ }
180
+ // Before the first append, never after: `pipeToRunLog` snapshots to align
181
+ // and a predecessor still writing must be observed, not raced.
182
+ await awaitLogQuiescence(input.durability(i.runId), fenceQuietMs)
183
+ return pipeToRunLog(stream, {
184
+ // BOTH authoritative seams are fenced at the epoch this driver actually
185
+ // acquired, and they must be: `pipeToRunLog` answers a refused append by
186
+ // recording a terminal record, so fencing only the log leaves a
187
+ // superseded host marking a live run `'failed'` (see `fenceRunStore`).
188
+ // Neither fence covers `close()` — that stays unfenced on purpose.
189
+ runs: fenceRunStore(input.runs, claim, {
190
+ ...(input.logger === undefined ? {} : { logger: input.logger }),
191
+ }),
192
+ durability: (runId) =>
193
+ fenceDurability(input.durability(runId), claim, {
194
+ runs: input.runs,
195
+ }),
196
+ runId: i.runId,
197
+ threadId: i.threadId,
198
+ signal: i.signal,
199
+ ...(input.logger === undefined ? {} : { logger: input.logger }),
200
+ })
201
+ },
202
+ ...(input.waitUntil === undefined ? {} : { waitUntil: input.waitUntil }),
203
+ ...(input.logger === undefined ? {} : { logger: input.logger }),
204
+ }
205
+ }
@@ -0,0 +1,380 @@
1
+ /**
2
+ * The durability seam for a sandboxed run: the option shape `withSandbox` takes,
3
+ * the capability harness adapters read, and the two guards that keep a
4
+ * "durable" run actually recoverable.
5
+ *
6
+ * A run is durable only when BOTH a `RunStore` and a `StreamDurability` are
7
+ * wired, because either alone is useless: a record with no event log cannot be
8
+ * replayed, and a log with no record cannot be found, claimed, or reaped. So the
9
+ * capability exists or it does not — there is no half-configured state, and
10
+ * every existing app (which wires neither) keeps today's behavior untouched.
11
+ */
12
+ import { createCapability } from '@tanstack/ai'
13
+ import { DEFAULT_JOURNAL_DIR } from './journal'
14
+ import { alignToStoredLog, isBridgeCustomChunk } from './align'
15
+ import type { JournalOptions } from './runner'
16
+ import type { InternalLogger } from '@tanstack/ai/adapter-internals'
17
+ import type { RunStore, StreamChunk, StreamDurability } from '@tanstack/ai'
18
+
19
+ /** `withSandbox(sandbox, { durability })`. */
20
+ export interface SandboxDurabilityOptions<TOffset extends string = string> {
21
+ /**
22
+ * Delivery-durable event log for the run. Same key and shape as the
23
+ * transport's `durability.adapter`, so one adapter instance can be handed to
24
+ * both `withSandbox` and `toServerSentEventsResponse`.
25
+ *
26
+ * Generic in the offset type, defaulted to `string`, for the same reason
27
+ * {@link SandboxRunDriverOptions} and {@link ReapOptions} are:
28
+ * `StreamDurability` is INVARIANT in `TOffset` (`read` takes an offset in),
29
+ * so a backend that brands its cursors — `@tanstack/ai-durable-stream`'s
30
+ * `durableStream`, the multi-host production backend the sandbox docs point
31
+ * at — is not assignable to `StreamDurability<string>`. Without the parameter
32
+ * the resume route could be wired with it and the route that STARTS the run
33
+ * could not.
34
+ */
35
+ adapter: StreamDurability<TOffset>
36
+ /** Journal directory inside the sandbox. Defaults to `/tmp/tanstack-runs`. */
37
+ journal?: string
38
+ /**
39
+ * Whether a client disconnect DETACHES (leave the agent running) instead of
40
+ * destroying the sandbox. Defaults to `true` whenever durability is wired,
41
+ * because that is the whole point of wiring it.
42
+ *
43
+ * Set `false` to keep today's destroy-on-disconnect cost profile while still
44
+ * getting resumable DELIVERY (a reload replays the log). An explicit cancel
45
+ * destroys either way.
46
+ */
47
+ detachOnDisconnect?: boolean
48
+ /**
49
+ * Read an EXISTING run's journal instead of starting a new agent. Set by the
50
+ * attach route's `drive()` callback, never by an application's POST handler.
51
+ *
52
+ * This is where `attach` lives, and deliberately NOT on `chat()`: `chat()` is
53
+ * core and must not gain sandbox vocabulary, and the provider options are
54
+ * per-model type state, not per-request lifecycle.
55
+ */
56
+ attach?: boolean
57
+ /** Journal poll interval for providers that cannot follow. */
58
+ pollIntervalMs?: number
59
+ /**
60
+ * How long an ATTACH waits for a live run's journal to appear before failing
61
+ * with a `JournalAttachUnavailableError`. Defaults to
62
+ * `DEFAULT_ATTACH_JOURNAL_WAIT_MS` (10s). Only the wait is configurable: an
63
+ * unknown or terminal runId fails immediately regardless, since no amount of
64
+ * waiting changes either verdict.
65
+ */
66
+ attachWaitMs?: number
67
+ }
68
+
69
+ /**
70
+ * The view of a caller's event log that the capability bus carries.
71
+ *
72
+ * Deliberately NOT the whole `StreamDurability`. `read` is the only member that
73
+ * takes an offset *in*, which is what makes `StreamDurability` invariant in
74
+ * `TOffset` and a branded-cursor backend unassignable to
75
+ * `StreamDurability<string>`. Every other member mentions the offset only in a
76
+ * return position, so this type is a genuine SUPERTYPE of
77
+ * `StreamDurability<TOffset>` for every `TOffset extends string` — which is the
78
+ * one property that lets a single concrete capability instantiation accept a
79
+ * branded backend. `createCapability<T>()` forces exactly one instantiation
80
+ * (the value type is a plain type argument, and TypeScript has no higher-kinded
81
+ * types), so the payload cannot be parameterized the way the *option* above is.
82
+ *
83
+ * Dropping `read` costs nothing, and that is a property of the seam rather than
84
+ * luck: the bus is the JOURNAL/ALIGNMENT seam, and alignment reads the stored
85
+ * prefix through `snapshot()` — never `read()`, which tails an open log forever
86
+ * (see `alignToStoredLog`). Replay *by offset* belongs to the delivery seam,
87
+ * and that seam (`toServerSentEventsResponse`, `sandboxRunDriver`) receives the
88
+ * application's own adapter directly, with its brand intact.
89
+ */
90
+ export type SandboxDurabilityLog = Omit<StreamDurability, 'read'>
91
+
92
+ /**
93
+ * Resolved durability, published on the capability bus by `withSandbox`.
94
+ *
95
+ * Deliberately carries NO detached-run TTL. The only actor that enforces one is
96
+ * `reapDetachedRuns`, which runs from a cron with no chat in flight — so it has
97
+ * no `CapabilityContext` and cannot read this bus at all. A TTL published here
98
+ * could therefore only ever be read by nobody, while the sweep took its own
99
+ * `ReapOptions.detachedRunTtlMs`; the two would silently disagree. The reaper's
100
+ * required option is the single source of truth.
101
+ */
102
+ export interface SandboxRunDurability {
103
+ runs: RunStore
104
+ adapter: SandboxDurabilityLog
105
+ journalDir: string
106
+ attach: boolean
107
+ detachOnDisconnect: boolean
108
+ pollIntervalMs?: number
109
+ attachWaitMs?: number
110
+ }
111
+
112
+ /**
113
+ * Provided by `withSandbox` only when a run is genuinely durable (both stores
114
+ * wired). Harness adapters read it with `getOptional` and treat its absence as
115
+ * "no journaling contract to honour", which is exactly today's behavior.
116
+ */
117
+ export const SandboxDurabilityCapability =
118
+ createCapability<SandboxRunDurability>()('sandbox-durability')
119
+
120
+ /** Destructured accessors, matching `./capabilities`. */
121
+ export const [getSandboxDurability, provideSandboxDurability] =
122
+ SandboxDurabilityCapability
123
+
124
+ /**
125
+ * A durable run was started without a caller-supplied `runId`.
126
+ *
127
+ * Thrown rather than defaulted because the failure is otherwise INVISIBLE: an
128
+ * adapter-generated id (`${name}-${Date.now()}-${Math.random()...}`) produces a
129
+ * journal path at `/tmp/tanstack-runs/<id>.ndjson` that no successor host can
130
+ * recompute, so the run streams normally, records normally, and is silently
131
+ * unrecoverable. A loud failure at the start of `chatStream` is strictly better
132
+ * than a run that only reveals itself as non-durable during an incident.
133
+ */
134
+ export class DurableRunIdRequiredError extends Error {
135
+ constructor(readonly adapter: string) {
136
+ super(
137
+ `${adapter}: a durable sandboxed run requires a caller-supplied \`runId\`. ` +
138
+ `The journal path and the deterministic message-id generator are both derived from it, ` +
139
+ `so a successor host can only resume a run whose \`runId\` it can recompute. ` +
140
+ `Pass \`runId\` to chat({ ... }), or drop \`runs\`/\`durability\` from withSandbox(...) to run non-durably.`,
141
+ )
142
+ this.name = 'DurableRunIdRequiredError'
143
+ }
144
+ }
145
+
146
+ /**
147
+ * Resolve the `runId` a harness adapter will journal under.
148
+ *
149
+ * Replaces the bare `options.runId ?? this.generateId()` in every harness
150
+ * adapter. The fallback is preserved for non-durable runs — several `chat()`
151
+ * paths pass `runId` as a conditional spread, so `undefined` is reachable and
152
+ * removing the fallback would break them for no benefit.
153
+ *
154
+ * The `durable` check runs BEFORE `fallback()`, and that ordering is load
155
+ * bearing: a generated id must never be minted for a durable run, not even one
156
+ * that is discarded, because the whole point is that no such id can exist.
157
+ */
158
+ export function resolveDurableRunId(
159
+ runId: string | undefined,
160
+ options: { durable: boolean; adapter: string; fallback: () => string },
161
+ ): string {
162
+ if (runId !== undefined && runId.length > 0) return runId
163
+ if (options.durable) throw new DurableRunIdRequiredError(options.adapter)
164
+ return options.fallback()
165
+ }
166
+
167
+ /**
168
+ * An ATTACHING durable run was driven without the run record's `threadId`.
169
+ *
170
+ * The sibling of {@link DurableRunIdRequiredError}, for the other id an attach
171
+ * cannot mint for itself. `threadId` lands in EVERY chunk a harness adapter
172
+ * emits (see each package's `stream/translate.ts`), so a replay that generates a
173
+ * fresh one produces a stream that differs from the stored log in its very first
174
+ * chunk. `alignToStoredLog` then fails at index 0 with a
175
+ * `JournalReplayThreadIdMismatchError` — mid-stream, after the takeover has
176
+ * already claimed the run. Refusing up front is strictly better, and mirrors
177
+ * what `resolveDurableRunId` does for an id whose absence is equally fatal.
178
+ *
179
+ * Core already does its part: `startRunDriver` reads the record and hands
180
+ * `active.threadId` to `drive({ runId, threadId, signal })`. This error exists
181
+ * for the one gap it cannot close — application `drive` code that forgets to
182
+ * forward it into `chat()`.
183
+ */
184
+ export class DurableThreadIdRequiredError extends Error {
185
+ constructor(readonly adapter: string) {
186
+ super(
187
+ `${adapter}: an ATTACHING durable sandboxed run requires the run record's \`threadId\`. ` +
188
+ `Every emitted chunk carries \`threadId\`, so an attach that generates a fresh one replays a stream whose first chunk ` +
189
+ `already differs from the stored log, and alignment fails at index 0 (\`JournalReplayThreadIdMismatchError\`) even though ` +
190
+ `the agent behaved identically. Forward the run record's \`threadId\` — the one \`sandboxRunDriver\` passes to ` +
191
+ `\`drive({ runId, threadId, signal })\` — into \`chat({ ... })\` on the attach route. ` +
192
+ `A durable FRESH run needs none: that run is what establishes the \`threadId\`.`,
193
+ )
194
+ this.name = 'DurableThreadIdRequiredError'
195
+ }
196
+ }
197
+
198
+ /**
199
+ * Resolve the `threadId` a harness adapter will stamp on every chunk.
200
+ *
201
+ * Replaces the bare `options.threadId ?? this.generateId()` in the journaling
202
+ * harness adapters. Only the durable-AND-attaching quadrant throws; the other
203
+ * three keep the generated fallback and are byte-identical to before:
204
+ *
205
+ * | durable | attaching | behavior |
206
+ * | ------- | --------- | --------------------------------------------------- |
207
+ * | no | no | fallback — a plain non-durable run |
208
+ * | no | yes | fallback — not reachable today, and harmless anyway |
209
+ * | yes | no | fallback — the FRESH run that ESTABLISHES the id |
210
+ * | yes | yes | throw {@link DurableThreadIdRequiredError} |
211
+ *
212
+ * The durable-fresh row is the load-bearing one. A fresh durable run legitimately
213
+ * mints its `threadId` (there is no record to reuse one from), so throwing on
214
+ * `durable` alone — the obvious over-simplification — would break every durable
215
+ * run that has ever worked. Only re-entering an existing run has an id it MUST
216
+ * reuse, which is exactly the condition `attach` already expresses.
217
+ *
218
+ * As in `resolveDurableRunId`, the guard runs BEFORE `fallback()`: a generated id
219
+ * must never be minted on this path, not even one that is then discarded.
220
+ */
221
+ export function resolveDurableThreadId(
222
+ threadId: string | undefined,
223
+ options: {
224
+ durable: boolean
225
+ attaching: boolean
226
+ adapter: string
227
+ fallback: () => string
228
+ },
229
+ ): string {
230
+ if (threadId !== undefined && threadId.length > 0) return threadId
231
+ if (options.durable && options.attaching) {
232
+ throw new DurableThreadIdRequiredError(options.adapter)
233
+ }
234
+ return options.fallback()
235
+ }
236
+
237
+ /**
238
+ * An ATTACH was driven into a code path that can never replay a run.
239
+ *
240
+ * The third sibling of {@link DurableRunIdRequiredError} and
241
+ * {@link DurableThreadIdRequiredError}, and the one that is not about a missing
242
+ * id: here every id is present and the path itself is the problem.
243
+ *
244
+ * `sandboxRunDriver`'s `drive()` re-invokes `chat()` with `attach: true`. On a
245
+ * JOURNALING path that is genuinely a replay — `spawnNdjson` tails the journal
246
+ * the previous host wrote, `awaitAttachableJournal` refuses a hopeless attach up
247
+ * front, and `alignedIfAttaching` suppresses the prefix already delivered. A
248
+ * protocol path with none of those three has no journal to tail and nothing to
249
+ * align against, so `attach: true` does not resume anything: it starts the agent
250
+ * over from scratch against the workspace the first attempt already mutated, and
251
+ * appends its entire output to a log that still holds the first attempt's.
252
+ *
253
+ * Deliberately NOT a `JournalAttachUnavailableError`. That error means "a
254
+ * journal that should exist has not appeared yet" — retryable, scoped to a wait
255
+ * (`attachWaitMs`). This condition is categorically different: the path cannot
256
+ * attach AT ALL, so telling a caller to wait would point it at something that is
257
+ * never coming. A 5xx/501-shaped refusal, not a 504.
258
+ *
259
+ * `reason` names the missing capability in the adapter's own vocabulary (which
260
+ * protocol, which spawn path), because the fix is always to change how the run
261
+ * is spawned or routed, never to retry.
262
+ */
263
+ export class DurableAttachNotSupportedError extends Error {
264
+ constructor(
265
+ readonly adapter: string,
266
+ readonly reason: string,
267
+ ) {
268
+ super(
269
+ `${adapter}: this code path cannot ATTACH to an existing durable run (${reason}). ` +
270
+ `It does not journal, so there is no stored output to replay and no alignment to suppress what was already delivered. ` +
271
+ `Proceeding would re-run the agent from scratch against the workspace the previous attempt already modified, and double-append its entire output to the run log. ` +
272
+ `Route the attach through a journaling spawn path, or drop \`runs\`/\`durability\` from withSandbox(...) so the run is never resumed in the first place. ` +
273
+ `This is not a transient condition — unlike \`JournalAttachUnavailableError\`, waiting and retrying can never make it succeed.`,
274
+ )
275
+ this.name = 'DurableAttachNotSupportedError'
276
+ }
277
+ }
278
+
279
+ /**
280
+ * Resolve `withSandbox`'s two durability options into the capability payload, or
281
+ * `undefined` when the app has not opted in.
282
+ *
283
+ * BOTH `runs` and `durability` are required. A half-configured app gets
284
+ * `undefined` **silently** rather than a warning: it has not asked for
285
+ * durability, so there is nothing to warn about, and the resulting behavior
286
+ * (destroy on disconnect, no journal) is exactly today's.
287
+ */
288
+ export function resolveSandboxDurability<TOffset extends string = string>(
289
+ options:
290
+ | { runs?: RunStore; durability?: SandboxDurabilityOptions<TOffset> }
291
+ | undefined,
292
+ ): SandboxRunDurability | undefined {
293
+ const runs = options?.runs
294
+ const durability = options?.durability
295
+ if (runs === undefined || durability === undefined) return undefined
296
+ return {
297
+ runs,
298
+ adapter: durability.adapter,
299
+ journalDir: durability.journal ?? DEFAULT_JOURNAL_DIR,
300
+ attach: durability.attach === true,
301
+ detachOnDisconnect: durability.detachOnDisconnect !== false,
302
+ ...(durability.pollIntervalMs === undefined
303
+ ? {}
304
+ : { pollIntervalMs: durability.pollIntervalMs }),
305
+ ...(durability.attachWaitMs === undefined
306
+ ? {}
307
+ : { attachWaitMs: durability.attachWaitMs }),
308
+ }
309
+ }
310
+
311
+ /**
312
+ * Build the `spawnNdjson` journal option for a run, or `undefined` when the run
313
+ * is not durable — in which case `spawnNdjson` takes its original, unjournaled
314
+ * path (`isJournaled` tests `options.journal !== undefined`, `runner.ts:70-72`)
315
+ * and behavior is byte-identical to a pre-durability run.
316
+ *
317
+ * `JournalOptions.dir` is optional, but this always supplies it: the resolved
318
+ * durability has already defaulted `journalDir`, and a successor host must
319
+ * recompute the same path rather than re-derive the default independently.
320
+ *
321
+ * `runs` and `attachWaitMs` are carried ONLY when attaching, and that is not a
322
+ * micro-optimization: they exist for `awaitAttachableJournal`, which the reader
323
+ * runs on the attach path alone. A fresh run has no journal yet BY DESIGN (its own
324
+ * `journaledCommand` spawn creates it moments later), so handing it a run store
325
+ * would only invite a future change to gate a path where absence proves nothing.
326
+ */
327
+ export function journalOptionsFor(
328
+ durability: SandboxRunDurability | undefined,
329
+ runId: string,
330
+ ): JournalOptions | undefined {
331
+ if (durability === undefined) return undefined
332
+ return {
333
+ runId,
334
+ dir: durability.journalDir,
335
+ attach: durability.attach,
336
+ ...(durability.pollIntervalMs === undefined
337
+ ? {}
338
+ : { pollIntervalMs: durability.pollIntervalMs }),
339
+ ...(durability.attach
340
+ ? {
341
+ runs: durability.runs,
342
+ ...(durability.attachWaitMs === undefined
343
+ ? {}
344
+ : { attachWaitMs: durability.attachWaitMs }),
345
+ }
346
+ : {}),
347
+ }
348
+ }
349
+
350
+ /**
351
+ * Align a harness stream against the run's stored log — but ONLY on an attach.
352
+ *
353
+ * The `attach` guard is not an optimization, it is a CORRECTNESS requirement.
354
+ * `alignToStoredLog` snapshots the log before the first chunk is pulled and
355
+ * treats everything in that snapshot as "already delivered". On a FRESH run that
356
+ * premise is false: if such a run were aligned against a log that already holds
357
+ * entries — a `runId` collision, a retried request — its own chunks would be
358
+ * matched against those entries and silently SUPPRESSED instead of delivered,
359
+ * which is silent data loss rather than a slow path. Aligning only when
360
+ * re-entering an existing run keeps the transform's premise ("this stream is a
361
+ * replay of what is already stored") actually true.
362
+ *
363
+ * `isBridgeCustomChunk` is passed because the stored log holds the previous
364
+ * host's MERGED output, including live bridged-tool CUSTOM events that a replay
365
+ * cannot reproduce; without it a bridged-tool run could not be taken over at
366
+ * all. Wrap the merge RESULT, never the pre-merge translator, or the comparison
367
+ * is against a stream the log never contained.
368
+ */
369
+ export function alignedIfAttaching(
370
+ chunks: AsyncIterable<StreamChunk>,
371
+ durability: SandboxRunDurability | undefined,
372
+ logger?: InternalLogger,
373
+ ): AsyncIterable<StreamChunk> {
374
+ if (durability === undefined || !durability.attach) return chunks
375
+ return alignToStoredLog(chunks, {
376
+ durability: durability.adapter,
377
+ isOutOfBand: isBridgeCustomChunk,
378
+ ...(logger === undefined ? {} : { logger }),
379
+ })
380
+ }