@tanstack/ai-sandbox 0.2.4 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/agents-file.js +53 -34
- package/dist/esm/agents-file.js.map +1 -1
- package/dist/esm/align.d.ts +121 -0
- package/dist/esm/align.js +197 -0
- package/dist/esm/align.js.map +1 -0
- package/dist/esm/approvals.js +63 -29
- package/dist/esm/approvals.js.map +1 -1
- package/dist/esm/attach-preflight.d.ts +85 -0
- package/dist/esm/attach-preflight.js +189 -0
- package/dist/esm/attach-preflight.js.map +1 -0
- package/dist/esm/bootstrap.js +103 -117
- package/dist/esm/bootstrap.js.map +1 -1
- package/dist/esm/bridge-events.js +96 -71
- package/dist/esm/bridge-events.js.map +1 -1
- package/dist/esm/capabilities.d.ts +0 -5
- package/dist/esm/capabilities.js +32 -28
- package/dist/esm/capabilities.js.map +1 -1
- package/dist/esm/chunk-identity.d.ts +52 -0
- package/dist/esm/chunk-identity.js +102 -0
- package/dist/esm/chunk-identity.js.map +1 -0
- package/dist/esm/claim.d.ts +187 -0
- package/dist/esm/claim.js +349 -0
- package/dist/esm/claim.js.map +1 -0
- package/dist/esm/contracts.d.ts +13 -0
- package/dist/esm/driver.d.ts +83 -0
- package/dist/esm/driver.js +138 -0
- package/dist/esm/driver.js.map +1 -0
- package/dist/esm/durability.d.ts +263 -0
- package/dist/esm/durability.js +230 -0
- package/dist/esm/durability.js.map +1 -0
- package/dist/esm/errors.js +28 -24
- package/dist/esm/errors.js.map +1 -1
- package/dist/esm/file-diff.js +151 -135
- package/dist/esm/file-diff.js.map +1 -1
- package/dist/esm/git-exec.js +51 -62
- package/dist/esm/git-exec.js.map +1 -1
- package/dist/esm/harness-cwd.js +24 -19
- package/dist/esm/harness-cwd.js.map +1 -1
- package/dist/esm/index.d.ts +30 -8
- package/dist/esm/index.js +23 -91
- package/dist/esm/instance-store.d.ts +88 -0
- package/dist/esm/instance-store.js +67 -0
- package/dist/esm/instance-store.js.map +1 -0
- package/dist/esm/journal-bytes.d.ts +67 -0
- package/dist/esm/journal-bytes.js +110 -0
- package/dist/esm/journal-bytes.js.map +1 -0
- package/dist/esm/journal-reader.d.ts +66 -0
- package/dist/esm/journal-reader.js +228 -0
- package/dist/esm/journal-reader.js.map +1 -0
- package/dist/esm/journal-sweep.d.ts +113 -0
- package/dist/esm/journal-sweep.js +309 -0
- package/dist/esm/journal-sweep.js.map +1 -0
- package/dist/esm/journal.d.ts +542 -0
- package/dist/esm/journal.js +679 -0
- package/dist/esm/journal.js.map +1 -0
- package/dist/esm/key.js +36 -33
- package/dist/esm/key.js.map +1 -1
- package/dist/esm/middleware.d.ts +50 -2
- package/dist/esm/middleware.js +335 -208
- package/dist/esm/middleware.js.map +1 -1
- package/dist/esm/ngrok.js +75 -49
- package/dist/esm/ngrok.js.map +1 -1
- package/dist/esm/policy.js +43 -34
- package/dist/esm/policy.js.map +1 -1
- package/dist/esm/projection.js +16 -8
- package/dist/esm/projection.js.map +1 -1
- package/dist/esm/reap.d.ts +238 -0
- package/dist/esm/reap.js +355 -0
- package/dist/esm/reap.js.map +1 -0
- package/dist/esm/reclaim.d.ts +84 -0
- package/dist/esm/reclaim.js +106 -0
- package/dist/esm/reclaim.js.map +1 -0
- package/dist/esm/remote-tools.js +73 -62
- package/dist/esm/remote-tools.js.map +1 -1
- package/dist/esm/run.d.ts +93 -25
- package/dist/esm/run.js +274 -79
- package/dist/esm/run.js.map +1 -1
- package/dist/esm/runner.d.ts +119 -2
- package/dist/esm/runner.js +270 -51
- package/dist/esm/runner.js.map +1 -1
- package/dist/esm/sandbox.d.ts +3 -2
- package/dist/esm/sandbox.js +139 -123
- package/dist/esm/sandbox.js.map +1 -1
- package/dist/esm/secrets.js +39 -47
- package/dist/esm/secrets.js.map +1 -1
- package/dist/esm/setup-plan.js +22 -14
- package/dist/esm/setup-plan.js.map +1 -1
- package/dist/esm/shell.d.ts +8 -0
- package/dist/esm/shell.js +197 -158
- package/dist/esm/shell.js.map +1 -1
- package/dist/esm/testkit/conformance.d.ts +16 -0
- package/dist/esm/testkit/conformance.js +97 -0
- package/dist/esm/testkit/conformance.js.map +1 -0
- package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
- package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
- package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
- package/dist/esm/testkit/journal-conformance.d.ts +51 -0
- package/dist/esm/testkit/journal-conformance.js +378 -0
- package/dist/esm/testkit/journal-conformance.js.map +1 -0
- package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
- package/dist/esm/testkit/reaper-conformance.js +847 -0
- package/dist/esm/testkit/reaper-conformance.js.map +1 -0
- package/dist/esm/testkit/shell-spawn.d.ts +2 -0
- package/dist/esm/testkit/shell-spawn.js +60 -0
- package/dist/esm/testkit/shell-spawn.js.map +1 -0
- package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
- package/dist/esm/testkit/takeover-conformance.js +685 -0
- package/dist/esm/testkit/takeover-conformance.js.map +1 -0
- package/dist/esm/tool-bridge.js +227 -180
- package/dist/esm/tool-bridge.js.map +1 -1
- package/dist/esm/tool-history.d.ts +62 -0
- package/dist/esm/tool-history.js +171 -0
- package/dist/esm/tool-history.js.map +1 -0
- package/dist/esm/watch.js +310 -236
- package/dist/esm/watch.js.map +1 -1
- package/dist/esm/workspace.d.ts +1 -1
- package/dist/esm/workspace.js +49 -28
- package/dist/esm/workspace.js.map +1 -1
- package/package.json +16 -6
- package/skills/ai-sandbox/SKILL.md +658 -20
- package/src/align.ts +297 -0
- package/src/attach-preflight.ts +292 -0
- package/src/capabilities.ts +4 -13
- package/src/chunk-identity.ts +154 -0
- package/src/claim.ts +479 -0
- package/src/contracts.ts +13 -0
- package/src/driver.ts +205 -0
- package/src/durability.ts +380 -0
- package/src/index.ts +212 -27
- package/src/instance-store.ts +122 -0
- package/src/journal-bytes.ts +136 -0
- package/src/journal-reader.ts +359 -0
- package/src/journal-sweep.ts +406 -0
- package/src/journal.ts +875 -0
- package/src/middleware.ts +470 -30
- package/src/reap.ts +723 -0
- package/src/reclaim.ts +191 -0
- package/src/run.ts +365 -75
- package/src/runner.ts +347 -3
- package/src/sandbox.ts +38 -8
- package/src/shell.ts +106 -38
- package/src/testkit/conformance.ts +117 -0
- package/src/testkit/durable-run-fields-conformance.ts +147 -0
- package/src/testkit/journal-conformance.ts +676 -0
- package/src/testkit/reaper-conformance.ts +1201 -0
- package/src/testkit/shell-spawn.ts +67 -0
- package/src/testkit/takeover-conformance.ts +1040 -0
- package/src/tool-history.ts +245 -0
- package/src/workspace.ts +1 -1
- package/dist/esm/index.js.map +0 -1
- package/dist/esm/run-log.d.ts +0 -81
- package/dist/esm/run-log.js +0 -107
- package/dist/esm/run-log.js.map +0 -1
- package/dist/esm/store.d.ts +0 -53
- package/dist/esm/store.js +0 -34
- package/dist/esm/store.js.map +0 -1
- package/src/run-log.ts +0 -224
- package/src/store.ts +0 -83
package/src/driver.ts
ADDED
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The convenience that turns core's *injected* takeover seams into this
|
|
3
|
+
* package's real ones.
|
|
4
|
+
*
|
|
5
|
+
* `@tanstack/ai`'s `RunDriverOptions` deliberately takes `claim` and `pipe` as
|
|
6
|
+
* functions instead of importing them: {@link withRunClaim} and
|
|
7
|
+
* {@link pipeToRunLog} live here, and core must not depend on this package to
|
|
8
|
+
* serve a plain chat run. {@link sandboxRunDriver} fills both in so an
|
|
9
|
+
* application writes four fields instead of six, and — more importantly — so
|
|
10
|
+
* the *fencing* is wired correctly by construction rather than by every caller
|
|
11
|
+
* remembering to.
|
|
12
|
+
*
|
|
13
|
+
* WHAT IS EASY TO GET WRONG HERE, and therefore what this module exists to
|
|
14
|
+
* make impossible:
|
|
15
|
+
*
|
|
16
|
+
* 1. **Carrying the real epoch into `pipe`.** Core's `pipe` receives only
|
|
17
|
+
* `{ runId, threadId, signal }` — no epoch — because core has no concept of
|
|
18
|
+
* one. But {@link fenceDurability} needs the epoch this driver actually
|
|
19
|
+
* acquired: a hardcoded epoch (say `0`) is not a weaker fence, it is a
|
|
20
|
+
* permanently *tripped* one, since `withRunClaim` bumps `driverEpoch` to at
|
|
21
|
+
* least `1` before `fn` ever runs, so `observed > claim.epoch` holds on the
|
|
22
|
+
* very first append and EVERY takeover fails. The claim is therefore
|
|
23
|
+
* captured in a closure by the `claim` wrapper and read back by `pipe`.
|
|
24
|
+
* 2. **Fencing `close()`.** {@link fenceDurability} wraps only `append` for the
|
|
25
|
+
* reason spelled out in `claim.ts`: `close()` runs on every teardown path,
|
|
26
|
+
* including the teardown caused by losing the claim, and a fenced `close`
|
|
27
|
+
* would wedge the record at `'running'` with every live tailer parked
|
|
28
|
+
* forever. This module must not add a second fence around it.
|
|
29
|
+
* 2b. **Fencing only ONE of the two authoritative seams.** A run's facts live in
|
|
30
|
+
* its log *and* in its record, and `pipeToRunLog` reacts to a refused append
|
|
31
|
+
* by writing a terminal record — so wrapping the log alone just moves the harm
|
|
32
|
+
* from "a dead host poisons the successor's stream" to "a dead host marks the
|
|
33
|
+
* successor's live run failed". {@link fenceRunStore} must be wired here too,
|
|
34
|
+
* over the SAME claim, which is what makes the two fences share one latch.
|
|
35
|
+
* 3. **Skipping quiescence.** The successor's first append must come after the
|
|
36
|
+
* stored log has stopped growing, so a predecessor still writing is observed
|
|
37
|
+
* rather than raced. The gate belongs inside `pipe`, before `pipeToRunLog`
|
|
38
|
+
* takes its first `snapshot`.
|
|
39
|
+
*/
|
|
40
|
+
import { pipeToRunLog } from './run'
|
|
41
|
+
import {
|
|
42
|
+
DEFAULT_FENCE_QUIET_MS,
|
|
43
|
+
awaitLogQuiescence,
|
|
44
|
+
fenceDurability,
|
|
45
|
+
fenceRunStore,
|
|
46
|
+
withRunClaim,
|
|
47
|
+
} from './claim'
|
|
48
|
+
import type { RunClaim } from './claim'
|
|
49
|
+
import type { InternalLogger } from '@tanstack/ai/adapter-internals'
|
|
50
|
+
import type { LockStore } from '@tanstack/ai/locks'
|
|
51
|
+
import type {
|
|
52
|
+
RunDriverOptions,
|
|
53
|
+
RunStore,
|
|
54
|
+
StreamChunk,
|
|
55
|
+
StreamDurability,
|
|
56
|
+
} from '@tanstack/ai'
|
|
57
|
+
|
|
58
|
+
export interface SandboxRunDriverOptions<TOffset extends string = string> {
|
|
59
|
+
/** The attach request; core reads its run id with `resolveResumeRunId`. */
|
|
60
|
+
request: Request
|
|
61
|
+
runs: RunStore
|
|
62
|
+
locks: LockStore
|
|
63
|
+
/**
|
|
64
|
+
* Per-run event log factory, the same shape `RunDeps.durability` takes — a
|
|
65
|
+
* `StreamDurability` is bound to one run, so the log is resolved FROM the
|
|
66
|
+
* `runId` rather than handed in pre-bound.
|
|
67
|
+
*
|
|
68
|
+
* Generic in the offset type, defaulted to `string` so an existing call site
|
|
69
|
+
* needs no change. Hardcoding the default made a branded-cursor backend
|
|
70
|
+
* unusable here: `durableStream` returns
|
|
71
|
+
* `StreamDurability<DurableStreamOffset>`, which is not assignable to
|
|
72
|
+
* `StreamDurability<string>` because `read` is contravariant in its offset.
|
|
73
|
+
*/
|
|
74
|
+
durability: (runId: string) => StreamDurability<TOffset>
|
|
75
|
+
/** Produce the run's remaining events. Called only once the claim is held. */
|
|
76
|
+
drive: (input: {
|
|
77
|
+
runId: string
|
|
78
|
+
threadId: string
|
|
79
|
+
signal: AbortSignal
|
|
80
|
+
}) => AsyncIterable<StreamChunk>
|
|
81
|
+
/** Quiescence window; defaults to {@link DEFAULT_FENCE_QUIET_MS}. */
|
|
82
|
+
fenceQuietMs?: number
|
|
83
|
+
/** Platform keep-alive (e.g. `ctx.waitUntil`) for the background drive. */
|
|
84
|
+
waitUntil?: (promise: Promise<unknown>) => void
|
|
85
|
+
logger?: InternalLogger
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* `pipe` ran without a held claim. Not a recoverable condition: it means the
|
|
90
|
+
* returned options object was taken apart and `pipe` called outside `claim`, so
|
|
91
|
+
* there is no epoch to fence with and no lease guaranteeing exclusivity. Any
|
|
92
|
+
* append made in that state is exactly the duplicate-write bug the claim exists
|
|
93
|
+
* to prevent, so this fails loudly rather than appending unfenced.
|
|
94
|
+
*/
|
|
95
|
+
export class RunDriverPipeOutsideClaimError extends Error {
|
|
96
|
+
constructor(readonly runId: string) {
|
|
97
|
+
super(
|
|
98
|
+
`run ${runId}: sandboxRunDriver.pipe was called outside its claim, so the driver epoch is unknown; call it from within the claim callback`,
|
|
99
|
+
)
|
|
100
|
+
this.name = 'RunDriverPipeOutsideClaimError'
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Fill in a core `driver` block with this package's claim and run log.
|
|
106
|
+
*
|
|
107
|
+
* `drive` receives an `AbortSignal` — the driver owns the abort, so it hands
|
|
108
|
+
* out a signal rather than a controller — but `chat()` takes an
|
|
109
|
+
* `AbortController`. Mirror one onto the other, exactly as
|
|
110
|
+
* {@link https://tanstack.com/ai/latest/docs/sandbox/takeover | Takeover & Detached Runs}'s
|
|
111
|
+
* `controllerFor` does, so a lost claim actually stops the drive.
|
|
112
|
+
*
|
|
113
|
+
* @example
|
|
114
|
+
* ```typescript
|
|
115
|
+
* function controllerFor(signal: AbortSignal): AbortController {
|
|
116
|
+
* const controller = new AbortController()
|
|
117
|
+
* const abort = (): void => controller.abort(signal.reason)
|
|
118
|
+
* if (signal.aborted) abort()
|
|
119
|
+
* else signal.addEventListener('abort', abort, { once: true })
|
|
120
|
+
* return controller
|
|
121
|
+
* }
|
|
122
|
+
*
|
|
123
|
+
* export async function GET(request: Request) {
|
|
124
|
+
* return resumeServerSentEventsResponse({
|
|
125
|
+
* adapter: memoryStream(request),
|
|
126
|
+
* driver: sandboxRunDriver({
|
|
127
|
+
* request,
|
|
128
|
+
* runs,
|
|
129
|
+
* locks,
|
|
130
|
+
* durability: (runId) => logFor(runId),
|
|
131
|
+
* drive: ({ runId, threadId, signal }) =>
|
|
132
|
+
* chat({
|
|
133
|
+
* ...config,
|
|
134
|
+
* runId,
|
|
135
|
+
* threadId,
|
|
136
|
+
* abortController: controllerFor(signal),
|
|
137
|
+
* }),
|
|
138
|
+
* }),
|
|
139
|
+
* })
|
|
140
|
+
* }
|
|
141
|
+
* ```
|
|
142
|
+
*/
|
|
143
|
+
export function sandboxRunDriver<TOffset extends string = string>(
|
|
144
|
+
input: SandboxRunDriverOptions<TOffset>,
|
|
145
|
+
): RunDriverOptions {
|
|
146
|
+
const fenceQuietMs = input.fenceQuietMs ?? DEFAULT_FENCE_QUIET_MS
|
|
147
|
+
// The seam between core's `claim` and core's `pipe`. One options object serves
|
|
148
|
+
// one attach request and therefore one run, so a single slot is enough; it is
|
|
149
|
+
// cleared on the way out so a `pipe` after the claim released cannot reuse a
|
|
150
|
+
// stale epoch.
|
|
151
|
+
let current: RunClaim | undefined
|
|
152
|
+
|
|
153
|
+
return {
|
|
154
|
+
request: input.request,
|
|
155
|
+
runs: input.runs,
|
|
156
|
+
locks: input.locks,
|
|
157
|
+
drive: input.drive,
|
|
158
|
+
claim: (claimInput, fn) =>
|
|
159
|
+
withRunClaim(
|
|
160
|
+
{
|
|
161
|
+
...claimInput,
|
|
162
|
+
fenceQuietMs,
|
|
163
|
+
...(input.logger === undefined ? {} : { logger: input.logger }),
|
|
164
|
+
},
|
|
165
|
+
async (claim) => {
|
|
166
|
+
const previous = current
|
|
167
|
+
current = claim
|
|
168
|
+
try {
|
|
169
|
+
return await fn(claim)
|
|
170
|
+
} finally {
|
|
171
|
+
current = previous
|
|
172
|
+
}
|
|
173
|
+
},
|
|
174
|
+
),
|
|
175
|
+
pipe: async (stream, i) => {
|
|
176
|
+
const claim = current
|
|
177
|
+
if (claim === undefined) {
|
|
178
|
+
throw new RunDriverPipeOutsideClaimError(i.runId)
|
|
179
|
+
}
|
|
180
|
+
// Before the first append, never after: `pipeToRunLog` snapshots to align
|
|
181
|
+
// and a predecessor still writing must be observed, not raced.
|
|
182
|
+
await awaitLogQuiescence(input.durability(i.runId), fenceQuietMs)
|
|
183
|
+
return pipeToRunLog(stream, {
|
|
184
|
+
// BOTH authoritative seams are fenced at the epoch this driver actually
|
|
185
|
+
// acquired, and they must be: `pipeToRunLog` answers a refused append by
|
|
186
|
+
// recording a terminal record, so fencing only the log leaves a
|
|
187
|
+
// superseded host marking a live run `'failed'` (see `fenceRunStore`).
|
|
188
|
+
// Neither fence covers `close()` — that stays unfenced on purpose.
|
|
189
|
+
runs: fenceRunStore(input.runs, claim, {
|
|
190
|
+
...(input.logger === undefined ? {} : { logger: input.logger }),
|
|
191
|
+
}),
|
|
192
|
+
durability: (runId) =>
|
|
193
|
+
fenceDurability(input.durability(runId), claim, {
|
|
194
|
+
runs: input.runs,
|
|
195
|
+
}),
|
|
196
|
+
runId: i.runId,
|
|
197
|
+
threadId: i.threadId,
|
|
198
|
+
signal: i.signal,
|
|
199
|
+
...(input.logger === undefined ? {} : { logger: input.logger }),
|
|
200
|
+
})
|
|
201
|
+
},
|
|
202
|
+
...(input.waitUntil === undefined ? {} : { waitUntil: input.waitUntil }),
|
|
203
|
+
...(input.logger === undefined ? {} : { logger: input.logger }),
|
|
204
|
+
}
|
|
205
|
+
}
|
|
@@ -0,0 +1,380 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The durability seam for a sandboxed run: the option shape `withSandbox` takes,
|
|
3
|
+
* the capability harness adapters read, and the two guards that keep a
|
|
4
|
+
* "durable" run actually recoverable.
|
|
5
|
+
*
|
|
6
|
+
* A run is durable only when BOTH a `RunStore` and a `StreamDurability` are
|
|
7
|
+
* wired, because either alone is useless: a record with no event log cannot be
|
|
8
|
+
* replayed, and a log with no record cannot be found, claimed, or reaped. So the
|
|
9
|
+
* capability exists or it does not — there is no half-configured state, and
|
|
10
|
+
* every existing app (which wires neither) keeps today's behavior untouched.
|
|
11
|
+
*/
|
|
12
|
+
import { createCapability } from '@tanstack/ai'
|
|
13
|
+
import { DEFAULT_JOURNAL_DIR } from './journal'
|
|
14
|
+
import { alignToStoredLog, isBridgeCustomChunk } from './align'
|
|
15
|
+
import type { JournalOptions } from './runner'
|
|
16
|
+
import type { InternalLogger } from '@tanstack/ai/adapter-internals'
|
|
17
|
+
import type { RunStore, StreamChunk, StreamDurability } from '@tanstack/ai'
|
|
18
|
+
|
|
19
|
+
/** `withSandbox(sandbox, { durability })`. */
|
|
20
|
+
export interface SandboxDurabilityOptions<TOffset extends string = string> {
|
|
21
|
+
/**
|
|
22
|
+
* Delivery-durable event log for the run. Same key and shape as the
|
|
23
|
+
* transport's `durability.adapter`, so one adapter instance can be handed to
|
|
24
|
+
* both `withSandbox` and `toServerSentEventsResponse`.
|
|
25
|
+
*
|
|
26
|
+
* Generic in the offset type, defaulted to `string`, for the same reason
|
|
27
|
+
* {@link SandboxRunDriverOptions} and {@link ReapOptions} are:
|
|
28
|
+
* `StreamDurability` is INVARIANT in `TOffset` (`read` takes an offset in),
|
|
29
|
+
* so a backend that brands its cursors — `@tanstack/ai-durable-stream`'s
|
|
30
|
+
* `durableStream`, the multi-host production backend the sandbox docs point
|
|
31
|
+
* at — is not assignable to `StreamDurability<string>`. Without the parameter
|
|
32
|
+
* the resume route could be wired with it and the route that STARTS the run
|
|
33
|
+
* could not.
|
|
34
|
+
*/
|
|
35
|
+
adapter: StreamDurability<TOffset>
|
|
36
|
+
/** Journal directory inside the sandbox. Defaults to `/tmp/tanstack-runs`. */
|
|
37
|
+
journal?: string
|
|
38
|
+
/**
|
|
39
|
+
* Whether a client disconnect DETACHES (leave the agent running) instead of
|
|
40
|
+
* destroying the sandbox. Defaults to `true` whenever durability is wired,
|
|
41
|
+
* because that is the whole point of wiring it.
|
|
42
|
+
*
|
|
43
|
+
* Set `false` to keep today's destroy-on-disconnect cost profile while still
|
|
44
|
+
* getting resumable DELIVERY (a reload replays the log). An explicit cancel
|
|
45
|
+
* destroys either way.
|
|
46
|
+
*/
|
|
47
|
+
detachOnDisconnect?: boolean
|
|
48
|
+
/**
|
|
49
|
+
* Read an EXISTING run's journal instead of starting a new agent. Set by the
|
|
50
|
+
* attach route's `drive()` callback, never by an application's POST handler.
|
|
51
|
+
*
|
|
52
|
+
* This is where `attach` lives, and deliberately NOT on `chat()`: `chat()` is
|
|
53
|
+
* core and must not gain sandbox vocabulary, and the provider options are
|
|
54
|
+
* per-model type state, not per-request lifecycle.
|
|
55
|
+
*/
|
|
56
|
+
attach?: boolean
|
|
57
|
+
/** Journal poll interval for providers that cannot follow. */
|
|
58
|
+
pollIntervalMs?: number
|
|
59
|
+
/**
|
|
60
|
+
* How long an ATTACH waits for a live run's journal to appear before failing
|
|
61
|
+
* with a `JournalAttachUnavailableError`. Defaults to
|
|
62
|
+
* `DEFAULT_ATTACH_JOURNAL_WAIT_MS` (10s). Only the wait is configurable: an
|
|
63
|
+
* unknown or terminal runId fails immediately regardless, since no amount of
|
|
64
|
+
* waiting changes either verdict.
|
|
65
|
+
*/
|
|
66
|
+
attachWaitMs?: number
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* The view of a caller's event log that the capability bus carries.
|
|
71
|
+
*
|
|
72
|
+
* Deliberately NOT the whole `StreamDurability`. `read` is the only member that
|
|
73
|
+
* takes an offset *in*, which is what makes `StreamDurability` invariant in
|
|
74
|
+
* `TOffset` and a branded-cursor backend unassignable to
|
|
75
|
+
* `StreamDurability<string>`. Every other member mentions the offset only in a
|
|
76
|
+
* return position, so this type is a genuine SUPERTYPE of
|
|
77
|
+
* `StreamDurability<TOffset>` for every `TOffset extends string` — which is the
|
|
78
|
+
* one property that lets a single concrete capability instantiation accept a
|
|
79
|
+
* branded backend. `createCapability<T>()` forces exactly one instantiation
|
|
80
|
+
* (the value type is a plain type argument, and TypeScript has no higher-kinded
|
|
81
|
+
* types), so the payload cannot be parameterized the way the *option* above is.
|
|
82
|
+
*
|
|
83
|
+
* Dropping `read` costs nothing, and that is a property of the seam rather than
|
|
84
|
+
* luck: the bus is the JOURNAL/ALIGNMENT seam, and alignment reads the stored
|
|
85
|
+
* prefix through `snapshot()` — never `read()`, which tails an open log forever
|
|
86
|
+
* (see `alignToStoredLog`). Replay *by offset* belongs to the delivery seam,
|
|
87
|
+
* and that seam (`toServerSentEventsResponse`, `sandboxRunDriver`) receives the
|
|
88
|
+
* application's own adapter directly, with its brand intact.
|
|
89
|
+
*/
|
|
90
|
+
export type SandboxDurabilityLog = Omit<StreamDurability, 'read'>
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Resolved durability, published on the capability bus by `withSandbox`.
|
|
94
|
+
*
|
|
95
|
+
* Deliberately carries NO detached-run TTL. The only actor that enforces one is
|
|
96
|
+
* `reapDetachedRuns`, which runs from a cron with no chat in flight — so it has
|
|
97
|
+
* no `CapabilityContext` and cannot read this bus at all. A TTL published here
|
|
98
|
+
* could therefore only ever be read by nobody, while the sweep took its own
|
|
99
|
+
* `ReapOptions.detachedRunTtlMs`; the two would silently disagree. The reaper's
|
|
100
|
+
* required option is the single source of truth.
|
|
101
|
+
*/
|
|
102
|
+
export interface SandboxRunDurability {
|
|
103
|
+
runs: RunStore
|
|
104
|
+
adapter: SandboxDurabilityLog
|
|
105
|
+
journalDir: string
|
|
106
|
+
attach: boolean
|
|
107
|
+
detachOnDisconnect: boolean
|
|
108
|
+
pollIntervalMs?: number
|
|
109
|
+
attachWaitMs?: number
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Provided by `withSandbox` only when a run is genuinely durable (both stores
|
|
114
|
+
* wired). Harness adapters read it with `getOptional` and treat its absence as
|
|
115
|
+
* "no journaling contract to honour", which is exactly today's behavior.
|
|
116
|
+
*/
|
|
117
|
+
export const SandboxDurabilityCapability =
|
|
118
|
+
createCapability<SandboxRunDurability>()('sandbox-durability')
|
|
119
|
+
|
|
120
|
+
/** Destructured accessors, matching `./capabilities`. */
|
|
121
|
+
export const [getSandboxDurability, provideSandboxDurability] =
|
|
122
|
+
SandboxDurabilityCapability
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* A durable run was started without a caller-supplied `runId`.
|
|
126
|
+
*
|
|
127
|
+
* Thrown rather than defaulted because the failure is otherwise INVISIBLE: an
|
|
128
|
+
* adapter-generated id (`${name}-${Date.now()}-${Math.random()...}`) produces a
|
|
129
|
+
* journal path at `/tmp/tanstack-runs/<id>.ndjson` that no successor host can
|
|
130
|
+
* recompute, so the run streams normally, records normally, and is silently
|
|
131
|
+
* unrecoverable. A loud failure at the start of `chatStream` is strictly better
|
|
132
|
+
* than a run that only reveals itself as non-durable during an incident.
|
|
133
|
+
*/
|
|
134
|
+
export class DurableRunIdRequiredError extends Error {
|
|
135
|
+
constructor(readonly adapter: string) {
|
|
136
|
+
super(
|
|
137
|
+
`${adapter}: a durable sandboxed run requires a caller-supplied \`runId\`. ` +
|
|
138
|
+
`The journal path and the deterministic message-id generator are both derived from it, ` +
|
|
139
|
+
`so a successor host can only resume a run whose \`runId\` it can recompute. ` +
|
|
140
|
+
`Pass \`runId\` to chat({ ... }), or drop \`runs\`/\`durability\` from withSandbox(...) to run non-durably.`,
|
|
141
|
+
)
|
|
142
|
+
this.name = 'DurableRunIdRequiredError'
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Resolve the `runId` a harness adapter will journal under.
|
|
148
|
+
*
|
|
149
|
+
* Replaces the bare `options.runId ?? this.generateId()` in every harness
|
|
150
|
+
* adapter. The fallback is preserved for non-durable runs — several `chat()`
|
|
151
|
+
* paths pass `runId` as a conditional spread, so `undefined` is reachable and
|
|
152
|
+
* removing the fallback would break them for no benefit.
|
|
153
|
+
*
|
|
154
|
+
* The `durable` check runs BEFORE `fallback()`, and that ordering is load
|
|
155
|
+
* bearing: a generated id must never be minted for a durable run, not even one
|
|
156
|
+
* that is discarded, because the whole point is that no such id can exist.
|
|
157
|
+
*/
|
|
158
|
+
export function resolveDurableRunId(
|
|
159
|
+
runId: string | undefined,
|
|
160
|
+
options: { durable: boolean; adapter: string; fallback: () => string },
|
|
161
|
+
): string {
|
|
162
|
+
if (runId !== undefined && runId.length > 0) return runId
|
|
163
|
+
if (options.durable) throw new DurableRunIdRequiredError(options.adapter)
|
|
164
|
+
return options.fallback()
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* An ATTACHING durable run was driven without the run record's `threadId`.
|
|
169
|
+
*
|
|
170
|
+
* The sibling of {@link DurableRunIdRequiredError}, for the other id an attach
|
|
171
|
+
* cannot mint for itself. `threadId` lands in EVERY chunk a harness adapter
|
|
172
|
+
* emits (see each package's `stream/translate.ts`), so a replay that generates a
|
|
173
|
+
* fresh one produces a stream that differs from the stored log in its very first
|
|
174
|
+
* chunk. `alignToStoredLog` then fails at index 0 with a
|
|
175
|
+
* `JournalReplayThreadIdMismatchError` — mid-stream, after the takeover has
|
|
176
|
+
* already claimed the run. Refusing up front is strictly better, and mirrors
|
|
177
|
+
* what `resolveDurableRunId` does for an id whose absence is equally fatal.
|
|
178
|
+
*
|
|
179
|
+
* Core already does its part: `startRunDriver` reads the record and hands
|
|
180
|
+
* `active.threadId` to `drive({ runId, threadId, signal })`. This error exists
|
|
181
|
+
* for the one gap it cannot close — application `drive` code that forgets to
|
|
182
|
+
* forward it into `chat()`.
|
|
183
|
+
*/
|
|
184
|
+
export class DurableThreadIdRequiredError extends Error {
|
|
185
|
+
constructor(readonly adapter: string) {
|
|
186
|
+
super(
|
|
187
|
+
`${adapter}: an ATTACHING durable sandboxed run requires the run record's \`threadId\`. ` +
|
|
188
|
+
`Every emitted chunk carries \`threadId\`, so an attach that generates a fresh one replays a stream whose first chunk ` +
|
|
189
|
+
`already differs from the stored log, and alignment fails at index 0 (\`JournalReplayThreadIdMismatchError\`) even though ` +
|
|
190
|
+
`the agent behaved identically. Forward the run record's \`threadId\` — the one \`sandboxRunDriver\` passes to ` +
|
|
191
|
+
`\`drive({ runId, threadId, signal })\` — into \`chat({ ... })\` on the attach route. ` +
|
|
192
|
+
`A durable FRESH run needs none: that run is what establishes the \`threadId\`.`,
|
|
193
|
+
)
|
|
194
|
+
this.name = 'DurableThreadIdRequiredError'
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* Resolve the `threadId` a harness adapter will stamp on every chunk.
|
|
200
|
+
*
|
|
201
|
+
* Replaces the bare `options.threadId ?? this.generateId()` in the journaling
|
|
202
|
+
* harness adapters. Only the durable-AND-attaching quadrant throws; the other
|
|
203
|
+
* three keep the generated fallback and are byte-identical to before:
|
|
204
|
+
*
|
|
205
|
+
* | durable | attaching | behavior |
|
|
206
|
+
* | ------- | --------- | --------------------------------------------------- |
|
|
207
|
+
* | no | no | fallback — a plain non-durable run |
|
|
208
|
+
* | no | yes | fallback — not reachable today, and harmless anyway |
|
|
209
|
+
* | yes | no | fallback — the FRESH run that ESTABLISHES the id |
|
|
210
|
+
* | yes | yes | throw {@link DurableThreadIdRequiredError} |
|
|
211
|
+
*
|
|
212
|
+
* The durable-fresh row is the load-bearing one. A fresh durable run legitimately
|
|
213
|
+
* mints its `threadId` (there is no record to reuse one from), so throwing on
|
|
214
|
+
* `durable` alone — the obvious over-simplification — would break every durable
|
|
215
|
+
* run that has ever worked. Only re-entering an existing run has an id it MUST
|
|
216
|
+
* reuse, which is exactly the condition `attach` already expresses.
|
|
217
|
+
*
|
|
218
|
+
* As in `resolveDurableRunId`, the guard runs BEFORE `fallback()`: a generated id
|
|
219
|
+
* must never be minted on this path, not even one that is then discarded.
|
|
220
|
+
*/
|
|
221
|
+
export function resolveDurableThreadId(
|
|
222
|
+
threadId: string | undefined,
|
|
223
|
+
options: {
|
|
224
|
+
durable: boolean
|
|
225
|
+
attaching: boolean
|
|
226
|
+
adapter: string
|
|
227
|
+
fallback: () => string
|
|
228
|
+
},
|
|
229
|
+
): string {
|
|
230
|
+
if (threadId !== undefined && threadId.length > 0) return threadId
|
|
231
|
+
if (options.durable && options.attaching) {
|
|
232
|
+
throw new DurableThreadIdRequiredError(options.adapter)
|
|
233
|
+
}
|
|
234
|
+
return options.fallback()
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
/**
|
|
238
|
+
* An ATTACH was driven into a code path that can never replay a run.
|
|
239
|
+
*
|
|
240
|
+
* The third sibling of {@link DurableRunIdRequiredError} and
|
|
241
|
+
* {@link DurableThreadIdRequiredError}, and the one that is not about a missing
|
|
242
|
+
* id: here every id is present and the path itself is the problem.
|
|
243
|
+
*
|
|
244
|
+
* `sandboxRunDriver`'s `drive()` re-invokes `chat()` with `attach: true`. On a
|
|
245
|
+
* JOURNALING path that is genuinely a replay — `spawnNdjson` tails the journal
|
|
246
|
+
* the previous host wrote, `awaitAttachableJournal` refuses a hopeless attach up
|
|
247
|
+
* front, and `alignedIfAttaching` suppresses the prefix already delivered. A
|
|
248
|
+
* protocol path with none of those three has no journal to tail and nothing to
|
|
249
|
+
* align against, so `attach: true` does not resume anything: it starts the agent
|
|
250
|
+
* over from scratch against the workspace the first attempt already mutated, and
|
|
251
|
+
* appends its entire output to a log that still holds the first attempt's.
|
|
252
|
+
*
|
|
253
|
+
* Deliberately NOT a `JournalAttachUnavailableError`. That error means "a
|
|
254
|
+
* journal that should exist has not appeared yet" — retryable, scoped to a wait
|
|
255
|
+
* (`attachWaitMs`). This condition is categorically different: the path cannot
|
|
256
|
+
* attach AT ALL, so telling a caller to wait would point it at something that is
|
|
257
|
+
* never coming. A 5xx/501-shaped refusal, not a 504.
|
|
258
|
+
*
|
|
259
|
+
* `reason` names the missing capability in the adapter's own vocabulary (which
|
|
260
|
+
* protocol, which spawn path), because the fix is always to change how the run
|
|
261
|
+
* is spawned or routed, never to retry.
|
|
262
|
+
*/
|
|
263
|
+
export class DurableAttachNotSupportedError extends Error {
|
|
264
|
+
constructor(
|
|
265
|
+
readonly adapter: string,
|
|
266
|
+
readonly reason: string,
|
|
267
|
+
) {
|
|
268
|
+
super(
|
|
269
|
+
`${adapter}: this code path cannot ATTACH to an existing durable run (${reason}). ` +
|
|
270
|
+
`It does not journal, so there is no stored output to replay and no alignment to suppress what was already delivered. ` +
|
|
271
|
+
`Proceeding would re-run the agent from scratch against the workspace the previous attempt already modified, and double-append its entire output to the run log. ` +
|
|
272
|
+
`Route the attach through a journaling spawn path, or drop \`runs\`/\`durability\` from withSandbox(...) so the run is never resumed in the first place. ` +
|
|
273
|
+
`This is not a transient condition — unlike \`JournalAttachUnavailableError\`, waiting and retrying can never make it succeed.`,
|
|
274
|
+
)
|
|
275
|
+
this.name = 'DurableAttachNotSupportedError'
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/**
|
|
280
|
+
* Resolve `withSandbox`'s two durability options into the capability payload, or
|
|
281
|
+
* `undefined` when the app has not opted in.
|
|
282
|
+
*
|
|
283
|
+
* BOTH `runs` and `durability` are required. A half-configured app gets
|
|
284
|
+
* `undefined` **silently** rather than a warning: it has not asked for
|
|
285
|
+
* durability, so there is nothing to warn about, and the resulting behavior
|
|
286
|
+
* (destroy on disconnect, no journal) is exactly today's.
|
|
287
|
+
*/
|
|
288
|
+
export function resolveSandboxDurability<TOffset extends string = string>(
|
|
289
|
+
options:
|
|
290
|
+
| { runs?: RunStore; durability?: SandboxDurabilityOptions<TOffset> }
|
|
291
|
+
| undefined,
|
|
292
|
+
): SandboxRunDurability | undefined {
|
|
293
|
+
const runs = options?.runs
|
|
294
|
+
const durability = options?.durability
|
|
295
|
+
if (runs === undefined || durability === undefined) return undefined
|
|
296
|
+
return {
|
|
297
|
+
runs,
|
|
298
|
+
adapter: durability.adapter,
|
|
299
|
+
journalDir: durability.journal ?? DEFAULT_JOURNAL_DIR,
|
|
300
|
+
attach: durability.attach === true,
|
|
301
|
+
detachOnDisconnect: durability.detachOnDisconnect !== false,
|
|
302
|
+
...(durability.pollIntervalMs === undefined
|
|
303
|
+
? {}
|
|
304
|
+
: { pollIntervalMs: durability.pollIntervalMs }),
|
|
305
|
+
...(durability.attachWaitMs === undefined
|
|
306
|
+
? {}
|
|
307
|
+
: { attachWaitMs: durability.attachWaitMs }),
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
/**
|
|
312
|
+
* Build the `spawnNdjson` journal option for a run, or `undefined` when the run
|
|
313
|
+
* is not durable — in which case `spawnNdjson` takes its original, unjournaled
|
|
314
|
+
* path (`isJournaled` tests `options.journal !== undefined`, `runner.ts:70-72`)
|
|
315
|
+
* and behavior is byte-identical to a pre-durability run.
|
|
316
|
+
*
|
|
317
|
+
* `JournalOptions.dir` is optional, but this always supplies it: the resolved
|
|
318
|
+
* durability has already defaulted `journalDir`, and a successor host must
|
|
319
|
+
* recompute the same path rather than re-derive the default independently.
|
|
320
|
+
*
|
|
321
|
+
* `runs` and `attachWaitMs` are carried ONLY when attaching, and that is not a
|
|
322
|
+
* micro-optimization: they exist for `awaitAttachableJournal`, which the reader
|
|
323
|
+
* runs on the attach path alone. A fresh run has no journal yet BY DESIGN (its own
|
|
324
|
+
* `journaledCommand` spawn creates it moments later), so handing it a run store
|
|
325
|
+
* would only invite a future change to gate a path where absence proves nothing.
|
|
326
|
+
*/
|
|
327
|
+
export function journalOptionsFor(
|
|
328
|
+
durability: SandboxRunDurability | undefined,
|
|
329
|
+
runId: string,
|
|
330
|
+
): JournalOptions | undefined {
|
|
331
|
+
if (durability === undefined) return undefined
|
|
332
|
+
return {
|
|
333
|
+
runId,
|
|
334
|
+
dir: durability.journalDir,
|
|
335
|
+
attach: durability.attach,
|
|
336
|
+
...(durability.pollIntervalMs === undefined
|
|
337
|
+
? {}
|
|
338
|
+
: { pollIntervalMs: durability.pollIntervalMs }),
|
|
339
|
+
...(durability.attach
|
|
340
|
+
? {
|
|
341
|
+
runs: durability.runs,
|
|
342
|
+
...(durability.attachWaitMs === undefined
|
|
343
|
+
? {}
|
|
344
|
+
: { attachWaitMs: durability.attachWaitMs }),
|
|
345
|
+
}
|
|
346
|
+
: {}),
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
/**
|
|
351
|
+
* Align a harness stream against the run's stored log — but ONLY on an attach.
|
|
352
|
+
*
|
|
353
|
+
* The `attach` guard is not an optimization, it is a CORRECTNESS requirement.
|
|
354
|
+
* `alignToStoredLog` snapshots the log before the first chunk is pulled and
|
|
355
|
+
* treats everything in that snapshot as "already delivered". On a FRESH run that
|
|
356
|
+
* premise is false: if such a run were aligned against a log that already holds
|
|
357
|
+
* entries — a `runId` collision, a retried request — its own chunks would be
|
|
358
|
+
* matched against those entries and silently SUPPRESSED instead of delivered,
|
|
359
|
+
* which is silent data loss rather than a slow path. Aligning only when
|
|
360
|
+
* re-entering an existing run keeps the transform's premise ("this stream is a
|
|
361
|
+
* replay of what is already stored") actually true.
|
|
362
|
+
*
|
|
363
|
+
* `isBridgeCustomChunk` is passed because the stored log holds the previous
|
|
364
|
+
* host's MERGED output, including live bridged-tool CUSTOM events that a replay
|
|
365
|
+
* cannot reproduce; without it a bridged-tool run could not be taken over at
|
|
366
|
+
* all. Wrap the merge RESULT, never the pre-merge translator, or the comparison
|
|
367
|
+
* is against a stream the log never contained.
|
|
368
|
+
*/
|
|
369
|
+
export function alignedIfAttaching(
|
|
370
|
+
chunks: AsyncIterable<StreamChunk>,
|
|
371
|
+
durability: SandboxRunDurability | undefined,
|
|
372
|
+
logger?: InternalLogger,
|
|
373
|
+
): AsyncIterable<StreamChunk> {
|
|
374
|
+
if (durability === undefined || !durability.attach) return chunks
|
|
375
|
+
return alignToStoredLog(chunks, {
|
|
376
|
+
durability: durability.adapter,
|
|
377
|
+
isOutOfBand: isBridgeCustomChunk,
|
|
378
|
+
...(logger === undefined ? {} : { logger }),
|
|
379
|
+
})
|
|
380
|
+
}
|