@tanstack/ai-sandbox 0.3.1 → 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -5
- package/dist/esm/agents-file.d.ts +15 -0
- package/dist/esm/agents-file.js +47 -1
- package/dist/esm/agents-file.js.map +1 -1
- package/dist/esm/bootstrap.js +2 -1
- package/dist/esm/bootstrap.js.map +1 -1
- package/dist/esm/contracts.d.ts +12 -8
- package/dist/esm/git-exec.js +2 -0
- package/dist/esm/git-exec.js.map +1 -1
- package/dist/esm/index.d.ts +2 -1
- package/dist/esm/index.js +3 -3
- package/dist/esm/journal-sweep.js +3 -2
- package/dist/esm/journal-sweep.js.map +1 -1
- package/dist/esm/middleware.js +5 -2
- package/dist/esm/middleware.js.map +1 -1
- package/dist/esm/reap.js +2 -1
- package/dist/esm/reap.js.map +1 -1
- package/dist/esm/sandbox.d.ts +2 -0
- package/dist/esm/sandbox.js +19 -2
- package/dist/esm/sandbox.js.map +1 -1
- package/dist/esm/shell.js +5 -4
- package/dist/esm/shell.js.map +1 -1
- package/dist/esm/testkit/conformance.js +4 -2
- package/dist/esm/testkit/conformance.js.map +1 -1
- package/dist/esm/testkit/journal-conformance.js +6 -3
- package/dist/esm/testkit/journal-conformance.js.map +1 -1
- package/dist/esm/testkit/reaper-conformance.js +13 -8
- package/dist/esm/testkit/reaper-conformance.js.map +1 -1
- package/dist/esm/testkit/takeover-conformance.js +4 -3
- package/dist/esm/testkit/takeover-conformance.js.map +1 -1
- package/dist/esm/tool-bridge.js +1 -1
- package/dist/esm/tool-bridge.js.map +1 -1
- package/package.json +14 -4
- package/skills/ai-sandbox/SKILL.md +13 -8
- package/src/agents-file.ts +65 -0
- package/src/bootstrap.ts +5 -1
- package/src/contracts.ts +12 -8
- package/src/git-exec.ts +7 -0
- package/src/index.ts +2 -0
- package/src/middleware.ts +4 -1
- package/src/sandbox.ts +23 -0
- package/src/tool-bridge.ts +3 -1
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"middleware.js","names":[],"sources":["../../src/middleware.ts"],"sourcesContent":["/**\n * `withSandbox(definition, options?)` — the middleware that PROVIDES the\n * {@link SandboxCapability} a harness adapter requires.\n *\n * - `setup`: resume-or-create the sandbox (via the definition's ensure\n * algorithm), provide the handle, using the durability seams from\n * {@link SandboxMiddlewareOptions} (or, failing that, a bus-provided\n * SandboxInstanceStoreCapability / LocksCapability, then an in-memory\n * fallback). If `fileEvents` is not false, starts a\n * watcher that dispatches to sandbox-scoped hooks and forwards to the runtime\n * sink.\n * - `onFinish`/`onAbort`/`onError`: stop the watcher, snapshot (`after-run`)\n * and/or destroy per lifecycle.\n *\n * NOTE: streamed sandbox lifecycle events (sandbox.created, workspace.setup.*)\n * are emitted by the harness adapter's chatStream (which can yield CUSTOM\n * chunks), not from here — middleware setup runs before streaming begins.\n */\nimport {\n defineChatMiddleware,\n provideDetachableRun,\n provideRunDetached,\n wasCancelRequested,\n} from '@tanstack/ai'\nimport { InMemoryLockStore, LocksCapability } from '@tanstack/ai/locks'\nimport {\n getPendingTurn,\n getRunDisconnect,\n getSandboxRuntime,\n} from '@tanstack/ai/adapter-internals'\nimport {\n SandboxCapability,\n provideSandbox,\n provideSandboxPolicy,\n} from './capabilities'\nimport {\n provideSandboxDurability,\n resolveSandboxDurability,\n} from './durability'\nimport { SandboxInstanceStoreCapability } from './instance-store'\nimport { computeWorkspaceHash } from './key'\nimport { buildFileHookEvent, resolveFileEvents } from './file-diff'\nimport { ProjectionCapability, provideWorkspaceProjection } from './projection'\nimport { resolveSecret } from './secrets'\nimport {\n createToolHistoryRecorder,\n stripObservedToolCalls,\n} from './tool-history'\nimport { watchWorkspace } from './watch'\nimport { DEFAULT_WORKSPACE_ROOT } from './bootstrap'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { LockStore } from '@tanstack/ai/locks'\nimport type {\n AbortInfo,\n ChatMiddlewareContext,\n DefinedChatMiddleware,\n RunStore,\n SandboxFileEvent,\n SandboxFileHookEvent,\n} from '@tanstack/ai'\nimport type {\n SandboxDurabilityOptions,\n SandboxRunDurability,\n} from './durability'\nimport type { SandboxInstanceStore } from './instance-store'\nimport type { ToolHistoryRecorder } from './tool-history'\nimport type { SandboxHandle } from './contracts'\nimport type {\n SandboxDefinition,\n SandboxEnsureContext,\n SandboxHooks,\n} from './sandbox'\nimport type { SandboxWatchHandle } from './watch'\n\n/** Per-request state we need to carry from `setup` to the terminal hooks. */\ninterface SandboxRunState {\n /**\n * OPTIONAL because the state is registered BEFORE `definition.ensure()` is\n * awaited, and `ensure` is the slowest thing in the whole run — cloning a repo\n * into a fresh sandbox is minutes wide. That window is where the most common\n * disconnect of all lands (a user starts a run and switches away while the UI\n * still says \"starting the sandbox\"), so it is the one window the teardown and\n * disconnect hooks most need to be able to act in. Registering only after the\n * handle exists left exactly that window uncovered.\n *\n * Nothing the disconnect path does needs the handle: `detachedSince` and\n * `sandboxKey` come from `ensureCtx`, which is built before `ensure` is called.\n * Only `onFinish`'s snapshot needs it, and that cannot run before `setup` has\n * completed.\n */\n handle?: SandboxHandle\n ensureCtx: SandboxEnsureContext\n watcher?: SandboxWatchHandle\n /** In-flight `enriched.diff()` promises queued by the `fileEvents.diff`\n * watcher callback, awaited before teardown so a pending diff isn't\n * dropped when the run finishes/aborts/errors mid-computation. */\n pendingDiffs: Array<Promise<void>>\n /** Logger captured at setup, so terminal hooks can log watcher teardown. */\n logger?: InternalLogger\n /**\n * Durability resolved once at setup (absent when the run is not durable), so\n * `onAbort` cannot reach a different verdict than the one `setup` published\n * on the capability bus.\n */\n durability?: SandboxRunDurability\n /**\n * Records the harness's own tool calls into the transcript, so a finished run\n * restores its tool cards from the message store instead of only from the (live,\n * rejoin-only) delivery log. See `./tool-history`.\n */\n toolHistory: ToolHistoryRecorder\n}\n\nconst runState = new WeakMap<object, SandboxRunState>()\n\n/**\n * Stop the watcher and drain any in-flight `diff()` promises before teardown,\n * so the final file's diff isn't dropped when a run finishes/aborts/errors\n * mid-computation. The `pendingDiffs` await is the load-bearing line — without\n * it a deferred diff resolves after the run is gone and its chunk is lost.\n */\nasync function drainWatcher(\n state: SandboxRunState,\n phase: 'finish' | 'abort' | 'error',\n): Promise<void> {\n // Guard `stop()`: a rejecting watcher teardown must NOT propagate out of\n // here, or the caller skips the `definition.destroy(...)` that follows —\n // leaking the sandbox on exactly the abort path that must ALWAYS tear down.\n try {\n await state.watcher?.stop()\n } catch (error) {\n state.logger?.warn('sandbox watcher stop failed', { phase, error })\n }\n await Promise.allSettled(state.pendingDiffs)\n if (state.watcher) state.logger?.sandbox('sandbox watcher stopped', { phase })\n}\n\n/**\n * Record the two facts a later attach and the reaper both need, then publish the\n * detach verdict core reads.\n *\n * Shared by the DISCONNECT subscriber registered in `setup` (the run is still\n * going — the normal case) and `onAbort`'s detach branch (the run is being torn\n * down while detachable), so the two can never write a different shape of detach.\n *\n * GUARDED, and reports failure rather than throwing. `update` is a documented\n * no-op for an unknown runId, so a vanished record does not turn teardown into a\n * throw; a genuinely rejecting store is the caller's to react to — `onAbort` falls\n * through to destroying the sandbox, because a DESTROYED sandbox beats an\n * unreachable one, while the disconnect subscriber has nothing to fall back to\n * (the run is alive and still using the sandbox) and simply leaves the verdict\n * unpublished.\n *\n * The verdict is published ONLY on success. Publishing it after a failed record\n * write would leave core holding the log open for a takeover that can never be\n * found, since nothing in the store points at the run.\n */\nasync function recordDetach(\n definition: SandboxDefinition,\n state: SandboxRunState,\n durability: SandboxRunDurability,\n ctx: ChatMiddlewareContext,\n phase: 'disconnect' | 'abort',\n): Promise<boolean> {\n try {\n // The record already exists: `setup` pre-creates it for every durable run\n // BEFORE `ensure`, precisely so this stamp cannot land on a runId the store has\n // never heard of — `RunStore.update` is a documented no-op for an unknown\n // runId, which is how the detach used to be lost silently (measured against the\n // browser repro: `detached_since` and `sandbox_key` both stayed NULL for a run\n // that had genuinely detached). If it has since vanished, that no-op is the\n // correct outcome and this must not throw.\n await durability.runs.update(ctx.runId, {\n detachedSince: Date.now(),\n sandboxKey: definition.key(state.ensureCtx),\n })\n } catch (error) {\n state.logger?.warn('sandbox detach record write failed', {\n runId: ctx.runId,\n phase,\n error,\n })\n return false\n }\n // Core's durable delivery sink reads this (see `RunDetachedCapability`) and\n // leaves the run's log OPEN instead of appending a synthetic terminal\n // `RUN_ERROR` and closing it — a terminalized log ends a later attach's replay\n // at the prefix and diverges the takeover's journal replay, which recorded a\n // healthy detached run as `'failed'`.\n provideRunDetached(ctx, true)\n return true\n}\n\n/**\n * Whether an out-of-band cancel has been recorded for this run, in EITHER band.\n * A user pressing Stop and a user closing the tab produce the IDENTICAL\n * connection close, so intent is never inferred from the disconnect itself: it\n * arrives in-process (the abort reason carried the cancel sentinel) or durably\n * (another host recorded it on the run record).\n */\nasync function cancelIntent(\n durability: SandboxRunDurability | undefined,\n runId: string,\n inProcess: boolean,\n): Promise<boolean> {\n if (inProcess) return true\n if (durability === undefined) return false\n // No guard needed here, and one would be dead code: `wasCancelRequested` already\n // answers `false` for a store read that rejects. That matters on this path,\n // because a rejection escaping into `onAbort` would skip BOTH of its branches at\n // once, leaving a sandbox that is neither reclaimable nor destroyed. The test\n // 'DETACHES when the cancel probe REJECTS' pins the composition.\n return wasCancelRequested(durability.runs, runId)\n}\n\n/** Defensively pull tenant scoping out of the runtime context, if present. */\nfunction tenantFrom(\n context: unknown,\n): { userId?: string; orgId?: string } | undefined {\n if (context === null || typeof context !== 'object') return undefined\n const c = context as Record<string, unknown>\n const userId = typeof c.userId === 'string' ? c.userId : undefined\n const orgId = typeof c.orgId === 'string' ? c.orgId : undefined\n if (userId === undefined && orgId === undefined) return undefined\n return { userId, orgId }\n}\n\n/**\n * Durability seams for a sandboxed run. Both are optional; each independently\n * falls back to a process-lifetime in-memory default, which is correct for a\n * single process but NOT across replicas.\n */\nexport interface SandboxMiddlewareOptions<TOffset extends string = string> {\n /**\n * Durable instance map (which provider sandbox to resume for a key). Pass\n * your own store to make resume survive across processes/replicas.\n *\n * Takes precedence over a store provided on the capability bus (see\n * `provideSandboxInstanceStore`), so the call site wins over ambient wiring.\n */\n instances?: SandboxInstanceStore\n /**\n * Distributed lock serializing resume-or-create for one key. Needed for\n * multi-replica correctness so two concurrent runs don't both create.\n *\n * Prefer `withLocks` from `@tanstack/ai/locks` when other middleware also\n * needs the lock; use this option to scope one to this sandbox. Takes\n * precedence over a bus-provided lock.\n */\n locks?: LockStore\n /**\n * Run lifecycle records. Pair with `durability.adapter` to make a run\n * DETACHABLE: a client disconnect then leaves the agent running and records\n * `detachedSince` instead of destroying the sandbox.\n *\n * Pass the SAME store chat persistence uses (`persistence.stores.runs`) so\n * one record describes the run instead of two that can disagree.\n *\n * Defaults to `undefined`: an app that passes neither this nor `durability`\n * keeps today's destroy-on-disconnect behavior exactly.\n */\n runs?: RunStore\n /**\n * Delivery durability for the run's event log, plus the journal and detach\n * knobs. Requires `runs`; either alone is not durable.\n *\n * `TOffset` is inferred from the adapter passed here, so a branded-cursor\n * backend (`durableStream`) wires without a cast and without the call site\n * ever naming the parameter.\n */\n durability?: SandboxDurabilityOptions<TOffset>\n}\n\n/**\n * Resolve the ensure seams. Precedence is explicit option → capability bus →\n * (in `ensure`) the in-memory fallback. The option wins because it is visible\n * at the call site; the bus remains for platform/framework injection.\n */\nfunction buildEnsureCtx(\n ctx: ChatMiddlewareContext,\n // Narrowed to the two seams it reads rather than taking the whole options\n // object: `SandboxMiddlewareOptions` is now generic in the durability offset,\n // and `SandboxMiddlewareOptions<TOffset>` is not assignable to\n // `SandboxMiddlewareOptions<string>`. Both members here are offset-free, so\n // the narrowing keeps this helper independent of that parameter entirely.\n options: Pick<SandboxMiddlewareOptions, 'instances' | 'locks'> | undefined,\n): SandboxEnsureContext {\n return {\n threadId: ctx.threadId,\n runId: ctx.runId,\n store:\n options?.instances ?? ctx.getOptional(SandboxInstanceStoreCapability),\n locks: options?.locks ?? ctx.getOptional(LocksCapability),\n tenant: tenantFrom(ctx.context),\n signal: ctx.signal,\n }\n}\n\n/**\n * Dispatch a sandbox file event to the per-type hooks declared on the\n * definition. Errors in individual hooks are swallowed so one bad hook\n * cannot break the run — but are logged under the `errors` category first, so\n * a throwing hook is observable (matching the run-scoped path in the engine\n * and the behavior the observability docs promise).\n */\nasync function dispatchDefinitionHooks(\n hooks: SandboxHooks | undefined,\n event: SandboxFileHookEvent,\n logger?: InternalLogger,\n): Promise<void> {\n if (!hooks) return\n const typed = (\n {\n create: 'onFileCreate',\n change: 'onFileChange',\n delete: 'onFileDelete',\n } as const\n )[event.type]\n for (const fn of [hooks.onFile, hooks[typed]]) {\n if (!fn) continue\n try {\n await fn(event)\n } catch (error) {\n // swallowed — one bad hook must not break the run — but logged so the\n // failure isn't invisible.\n logger?.errors('sandbox file hook failed', {\n path: event.path,\n type: event.type,\n error,\n })\n }\n }\n}\n\nexport function withSandbox<TOffset extends string = string>(\n definition: SandboxDefinition,\n options?: SandboxMiddlewareOptions<TOffset>,\n): DefinedChatMiddleware<\n unknown,\n readonly [],\n readonly [typeof SandboxCapability, typeof ProjectionCapability]\n> {\n return defineChatMiddleware({\n name: 'sandbox',\n provides: [SandboxCapability, ProjectionCapability],\n // SandboxPolicyCapability is provided conditionally (only when the\n // definition has a policy), so it is intentionally NOT declared here —\n // consumers read it via `getOptional`. SandboxDurabilityCapability and\n // DetachableRunCapability are conditional for the same reason (only when\n // `runs` + `durability` are both wired), so they are intentionally NOT\n // declared here either.\n optionalRequires: [SandboxInstanceStoreCapability, LocksCapability],\n\n async setup(ctx) {\n const ensureCtx = buildEnsureCtx(ctx, options)\n\n // Resolving here (not lazily on the abort path) is what keeps `setup` and\n // `onAbort` on one verdict: the payload the bus carries is the same object\n // the teardown path consults.\n // `TOffset` is passed explicitly: `options` is possibly `undefined` here,\n // so inference has nothing to work from on that branch and would fall\n // back to the `= string` default, re-erecting the very wall this\n // parameter exists to remove.\n const durability = resolveSandboxDurability<TOffset>(options)\n if (durability !== undefined) {\n provideSandboxDurability(ctx, durability)\n // A neutral boolean core owns, so `@tanstack/ai-persistence` can ask\n // \"is this run detachable?\" without depending on this package.\n provideDetachableRun(ctx, true)\n }\n\n // Pull the runtime (and its logger) up front so `baseSha` capture and\n // hook dispatch below can log through the same `sandbox`/`errors`\n // categories the engine uses.\n const runtime = getSandboxRuntime(ctx, { optional: true })\n const logger = runtime?.logger\n\n // REGISTER THE RUN STATE NOW — before `definition.ensure()`, not merely\n // before the end of `setup`.\n //\n // `onAbort` and the disconnect subscriber both need this state, so until\n // this map is populated they are silent no-ops. `ensure` is the LONGEST\n // await in the entire run (create a sandbox, clone a repo — minutes), and it\n // is where the most common disconnect of all lands: a user starts a run and\n // switches away while the UI still says \"starting the sandbox\". Registering\n // after `ensure` returned still left that whole window uncovered.\n //\n // Leaving it uncovered loses every teardown behavior at once: no\n // `detachedSince`/`sandboxKey`, so `listReclaimable` can never surface the\n // run and the reaper can never reclaim it; no `definition.destroy`, so the\n // sandbox leaks; and no detach verdict for core to read.\n //\n // Everything those hooks read is already resolved above: the ensure context\n // (which is all `definition.key` needs), the durability verdict, and the\n // logger. The fields discovered later (`handle`, `watcher`) are ASSIGNED onto\n // this same object as they become available, so the teardown path always\n // sees the most complete state that exists at the moment it runs.\n const state: SandboxRunState = {\n ensureCtx,\n pendingDiffs: [],\n toolHistory: createToolHistoryRecorder(),\n ...(logger ? { logger } : {}),\n ...(durability ? { durability } : {}),\n }\n runState.set(ctx, state)\n\n // MAKE THE RUN FINDABLE BEFORE `ensure`, not after the run finally streams.\n //\n // Chat persistence creates the run record from `onConfig`, which runs after\n // EVERY middleware `setup` — so for the whole of `definition.ensure` (create a\n // sandbox, clone a repo: minutes) the run has no record at all, and\n // `findActiveRun` answers \"no active run\" for a run that is demonstrably\n // starting. Measured: a status sidebar read straight off `findActiveRun`\n // reported `idle` for 6.5 minutes while the sandbox was being built, and a\n // client returning to the thread in that window had nothing to tell it a run\n // was in flight — so it rendered an empty pane instead of \"starting sandbox\".\n //\n // A crash in the same window is worse: no record means `listReclaimable` can\n // never surface the run, so the sandbox leaks with no recovery path.\n //\n // `createOrResume` is idempotent and never resurrects a finished run, so\n // persistence's own later call stays correct and simply finds this record.\n if (durability !== undefined) {\n try {\n await durability.runs.createOrResume({\n runId: ctx.runId,\n threadId: ctx.threadId,\n startedAt: Date.now(),\n })\n } catch (error) {\n // Best-effort: a store blip must not stop a run that is otherwise fine.\n // The run is simply invisible until persistence's own `onConfig` call.\n logger?.warn('sandbox run record pre-create failed', {\n runId: ctx.runId,\n error,\n })\n }\n\n // NO ATTACH MARKER HERE. A joiner does need a chunk in the log before the\n // harness has emitted anything — an empty log fails every joiner's\n // fast-fail (`memoryStream`'s first-chunk deadline, the client's rejoin\n // connect deadline) and flushes no HTTP headers, so a reload during\n // `ensure` reads a live run as gone. Core does it: a fresh durable producer\n // appends `RUN_ACCEPTED_EVENT` before the producer stream is first pulled,\n // for EVERY durable run rather than only sandboxed ones, and never on an\n // attach. A second marker from here would only land mid-stream in a run\n // that is already producing.\n\n // STORE THE USER'S TURN NOW, before `ensure` takes minutes.\n //\n // Chat persistence stores it from `onStart`, which runs after every\n // middleware `setup` — so without this the thread holds NOTHING for the\n // whole sandbox build. Measured: a reload during the build asked the server\n // for the conversation and got `{\"messages\":[],…}`, so the user saw no sign\n // of the message they had just sent, and a second device saw an empty\n // thread.\n //\n // The persistence layer owns WHAT gets stored (see `PendingTurnCapability`):\n // `saveThread` replaces the thread, so deciding the list here would risk\n // deleting the history. Absent when the app wires no persistence, which is\n // simply a run with no transcript to store.\n try {\n await getPendingTurn(ctx, { optional: true })?.snapshot()\n } catch (error) {\n // Best-effort: the run is still worth doing, and `onStart` stores the\n // turn again once setup completes.\n logger?.warn('sandbox pending-turn snapshot failed', {\n runId: ctx.runId,\n error,\n })\n }\n }\n\n // SUBSCRIBE BEFORE `ensure`, for the same reason the state is registered\n // before it: `ensure` is the minutes-wide await a disconnect actually lands\n // in. Core calls back immediately if the socket has already closed, so\n // subscribing here cannot miss a disconnect that beat us to it.\n //\n // This is what makes a durable run SURVIVE losing its viewer. The only route\n // a disconnect previously had into this middleware was the application\n // mirroring `request.signal` into `chat()`'s `abortController` — which aborts\n // the run, so `chat()` returned right after this `setup` and the harness\n // adapter's `chatStream` was never called: the agent in the sandbox we just\n // spent minutes creating was NEVER LAUNCHED, and no takeover could recover it\n // because an agent that never ran wrote no journal to replay.\n if (durability !== undefined && durability.detachOnDisconnect) {\n getRunDisconnect(ctx, { optional: true })?.subscribe(async () => {\n // BOOKKEEPING ONLY — the run is still executing. Deliberately absent:\n // `drainWatcher` (would blind a live agent's file events for the whole\n // remainder) and `definition.destroy` (the run is still using the\n // sandbox). Both belong to the terminal hooks, which still run exactly\n // once afterwards.\n //\n // A run with a cancel already recorded is left alone: that is `onAbort`'s\n // path, and stamping `detachedSince` on a deliberately-stopped run would\n // hand it to the reaper as reclaimable work.\n if (await cancelIntent(durability, ctx.runId, false)) return\n if (\n await recordDetach(definition, state, durability, ctx, 'disconnect')\n ) {\n state.logger?.sandbox(\n 'sandbox run detached on disconnect; the run continues',\n { runId: ctx.runId },\n )\n }\n })\n }\n\n const handle = await definition.ensure(ensureCtx)\n // MUTATE, don't re-`set`: a disconnect that landed during `ensure` already\n // captured this object.\n state.handle = handle\n provideSandbox(ctx, handle)\n if (definition.policy) provideSandboxPolicy(ctx, definition.policy)\n\n // Deliberately placed AFTER `logger` is in scope rather than next to the\n // `provideSandboxDurability` call above — there is no logger to warn\n // through until the runtime has been read.\n //\n // `ensureCtx.locks === undefined` counts as in-memory: `defineSandbox`'s\n // `ensure` falls back to a process-lifetime `InMemoryLockStore` when no\n // lock is wired, so an unwired lock has exactly the deficiency being\n // warned about — it is the MOST in-memory case, not an exempt one.\n if (\n durability !== undefined &&\n (ensureCtx.locks === undefined ||\n ensureCtx.locks instanceof InMemoryLockStore)\n ) {\n logger?.warn(\n 'sandbox durability is wired over an InMemoryLockStore: run claims are ' +\n 'serialized within this process only and the lease never signals loss, ' +\n 'so two hosts can drive one run and duplicate its event log. Use a ' +\n 'distributed LockStore via withLocks for any multi-replica deploy.',\n { runId: ctx.runId },\n )\n }\n\n const watchRoot = definition.workspace?.root ?? DEFAULT_WORKSPACE_ROOT\n let baseSha = ''\n try {\n const shaRes = await handle.process.exec('git rev-parse HEAD', {\n cwd: watchRoot,\n })\n if (shaRes.exitCode === 0) {\n baseSha = shaRes.stdout.trim()\n logger?.sandbox('sandbox git baseline captured', {\n root: watchRoot,\n baseSha,\n })\n } else {\n // Non-zero exit: either not a git repository (non-git workspace) or a\n // repo with no commits (no HEAD). Expected, but it silently degrades\n // every subsequent diff to a full-file add-patch, so surface it\n // under `sandbox` (with stderr) rather than leaving nothing to grep.\n logger?.sandbox('sandbox git baseline unavailable (non-zero exit)', {\n root: watchRoot,\n exitCode: shaRes.exitCode,\n stderr: shaRes.stderr,\n })\n }\n } catch (error) {\n // exec rejected (git not on PATH, exec seam broken) → baseSha stays ''\n // and accessors fall back, but this is a real anomaly, not a plain\n // non-git workspace, so warn.\n logger?.warn('sandbox git baseline capture failed', {\n root: watchRoot,\n error,\n })\n }\n\n const workspace = definition.workspace\n if (workspace !== undefined) {\n const root = workspace.root ?? DEFAULT_WORKSPACE_ROOT\n const workspaceHash = computeWorkspaceHash(workspace)\n const secrets = workspace.secrets\n provideWorkspaceProjection(ctx, {\n skills: workspace.skills ?? [],\n plugins: workspace.plugins ?? [],\n resolveSecret: (ref) => {\n if (secrets === undefined) {\n throw new Error(\n `resolveSecret: no secrets defined on this workspace (ref: \"${ref.__secretName}\")`,\n )\n }\n return resolveSecret(secrets, ref)\n },\n markerPath: `${root}/.tanstack-projected-${workspaceHash}`,\n root,\n ...(workspace.scripts !== undefined\n ? { scripts: workspace.scripts }\n : {}),\n })\n }\n\n const hooks = definition.hooks\n await hooks?.onReady?.(handle)\n\n const fe = resolveFileEvents(definition.fileEvents)\n // THE SAME array the run state already holds, not a fresh one. The watcher\n // callback below closes over this reference, and `drainWatcher` awaits\n // `state.pendingDiffs` — a second array would silently drop every in-flight\n // diff from the teardown drain.\n const pendingDiffs = state.pendingDiffs\n let watcher: SandboxWatchHandle | undefined\n if (fe.enabled) {\n watcher = await watchWorkspace(handle, {\n onEvent: (event: SandboxFileEvent) => {\n const enriched = buildFileHookEvent(\n handle,\n watchRoot,\n baseSha,\n event,\n logger,\n )\n void dispatchDefinitionHooks(hooks, enriched, logger)\n runtime?.emit(enriched)\n if (fe.diff) {\n pendingDiffs.push(\n enriched\n .diff()\n .then((diff) => {\n runtime?.emitFileDiff({ path: event.path, diff })\n })\n .catch((error: unknown) => {\n logger?.warn('sandbox file diff emit failed', {\n path: event.path,\n error,\n })\n }),\n )\n }\n },\n // Watch the SAME root the enrichment layer relativizes against\n // (`buildFileHookEvent(handle, watchRoot, …)` and the `baseSha`\n // capture). Without this the watcher defaults to `/workspace` while\n // enrichment uses `watchRoot`, so a custom `workspace.root` makes the\n // two look at different directories and git pathspecs break.\n root: watchRoot,\n ...(ctx.signal !== undefined ? { signal: ctx.signal } : {}),\n ...(logger !== undefined ? { logger } : {}),\n })\n logger?.sandbox('sandbox watcher started', {\n root: watchRoot,\n diff: fe.diff,\n })\n }\n\n // MUTATE the object registered above rather than `set`-ing a second one: an\n // abort that landed mid-setup already captured a reference to it (and may\n // already be draining `pendingDiffs`), so replacing the entry would hand the\n // teardown path a different object than the watcher writes into.\n // `pendingDiffs` needs no copying — it IS `state.pendingDiffs`.\n if (watcher) state.watcher = watcher\n },\n\n // Keep the recorded tool history OUT of the request to the model. It is stored\n // history for the next turn, it names tools the provider was never given, and one\n // triage-sized run is hundreds of kilobytes — so replaying it is wasteful at best\n // and rejected at worst. `ctx.messages` keeps it (that is what gets stored and\n // rendered); only `config.messages` loses it.\n onConfig(_ctx, config) {\n const messages = stripObservedToolCalls(config.messages)\n if (messages.length === config.messages.length) return\n return { messages }\n },\n\n // The engine re-syncs `middlewareCtx.messages` from its own array once per agent\n // iteration, which drops whatever the recorder appended during the previous\n // iteration's stream. Restoring it here — AFTER that sync — is what makes a\n // multi-iteration run keep its full history without depending on where this\n // middleware sits relative to persistence in the middleware array.\n onIteration(ctx) {\n runState.get(ctx)?.toolHistory.reconcile(ctx)\n },\n\n // Record the harness's own tool calls as transcript messages. Observe only:\n // returning nothing passes every chunk through untouched.\n onChunk(ctx, chunk) {\n runState.get(ctx)?.toolHistory.observe(chunk, ctx)\n },\n\n async onFinish(ctx) {\n const state = runState.get(ctx)\n if (!state) return\n const { handle, ensureCtx } = state\n\n // Last chance before persistence writes the transcript. Only matters if a\n // config sync landed after the final tool chunk; the recorder is idempotent, so\n // in the normal case this changes nothing.\n state.toolHistory.reconcile(ctx)\n\n await drainWatcher(state, 'finish')\n\n const lifecycle = definition.lifecycle\n\n // `handle` is absent only if `setup` never got past `definition.ensure`, in\n // which case there is no sandbox to snapshot.\n if (\n lifecycle?.snapshot === 'after-run' &&\n handle?.capabilities.snapshots &&\n handle.snapshot\n ) {\n const snapshot = await handle.snapshot(`after-run-${ctx.runId}`)\n const store = ensureCtx.store\n if (store) {\n const key = definition.key(ensureCtx)\n const existing = await store.get(key)\n if (existing) {\n await store.upsert({\n ...existing,\n latestSnapshotId: snapshot.id,\n updatedAt: Date.now(),\n })\n }\n }\n }\n\n if (lifecycle?.destroyOnComplete) {\n await definition.destroy(ensureCtx)\n await definition.hooks?.onDestroy?.()\n }\n },\n\n async onAbort(ctx, info: AbortInfo) {\n const state = runState.get(ctx)\n if (!state) return\n\n // First on BOTH branches: a diff still in flight must be drained whether\n // the sandbox is about to be destroyed or merely detached, or the final\n // file's diff is dropped.\n await drainWatcher(state, 'abort')\n\n const durability = state.durability\n const cancelled = await cancelIntent(\n durability,\n ctx.runId,\n info.cancelRequested === true,\n )\n\n if (\n durability !== undefined &&\n !cancelled &&\n durability.detachOnDisconnect\n ) {\n // DETACH on the teardown path. Reached when the run is aborted for a\n // reason that is NOT an out-of-band cancel while detachable — a genuine\n // stop from elsewhere, or a host going down. The ordinary disconnect is\n // handled by the disconnect subscriber in `setup`, which does not end the\n // run at all.\n //\n // On a failed record write this branch is ABANDONED for the destroy one\n // below, because a rejection here is the worst shape available: the\n // verdict is unpublished, so core terminalizes the log and records a\n // healthy detached run as failed; `detachedSince`/`sandboxKey` are\n // unwritten, so `listReclaimable` can never surface the run and\n // `reapDetachedRuns` can never reclaim it. A DESTROYED sandbox beats an\n // unreachable one — the same reasoning `drainWatcher` applies to its own\n // guarded `stop()`.\n if (await recordDetach(definition, state, durability, ctx, 'abort')) {\n return\n }\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n return\n }\n\n // ALWAYS tear down on an explicit abort, regardless of `destroyOnComplete`.\n // The in-sandbox agent process is not killed by closing its IO stream\n // (e.g. a Docker exec survives client disconnect), so the only reliable way\n // to stop it — and the token/cost drain of its ongoing API calls — is to\n // destroy the sandbox (stop the container/VM). `keepAlive` /\n // `destroyOnComplete:false` governs *successful completion*, never cancel.\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n },\n\n async onError(ctx, info) {\n const state = runState.get(ctx)\n if (!state) return\n\n await drainWatcher(state, 'error')\n await definition.hooks?.onError?.(info.error)\n\n // On failure, only tear down when the lifecycle says so; otherwise leave\n // the sandbox for a resumed retry.\n if (definition.lifecycle?.destroyOnComplete) {\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n }\n },\n })\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAiHA,IAAM,2BAAW,IAAI,QAAiC;;;;;;;AAQtD,eAAe,aACb,OACA,OACe;CAIf,IAAI;EACF,MAAM,MAAM,SAAS,KAAK;CAC5B,SAAS,OAAO;EACd,MAAM,QAAQ,KAAK,+BAA+B;GAAE;GAAO;EAAM,CAAC;CACpE;CACA,MAAM,QAAQ,WAAW,MAAM,YAAY;CAC3C,IAAI,MAAM,SAAS,MAAM,QAAQ,QAAQ,2BAA2B,EAAE,MAAM,CAAC;AAC/E;;;;;;;;;;;;;;;;;;;;;AAsBA,eAAe,aACb,YACA,OACA,YACA,KACA,OACkB;CAClB,IAAI;EAQF,MAAM,WAAW,KAAK,OAAO,IAAI,OAAO;GACtC,eAAe,KAAK,IAAI;GACxB,YAAY,WAAW,IAAI,MAAM,SAAS;EAC5C,CAAC;CACH,SAAS,OAAO;EACd,MAAM,QAAQ,KAAK,sCAAsC;GACvD,OAAO,IAAI;GACX;GACA;EACF,CAAC;EACD,OAAO;CACT;CAMA,mBAAmB,KAAK,IAAI;CAC5B,OAAO;AACT;;;;;;;;AASA,eAAe,aACb,YACA,OACA,WACkB;CAClB,IAAI,WAAW,OAAO;CACtB,IAAI,eAAe,KAAA,GAAW,OAAO;CAMrC,OAAO,mBAAmB,WAAW,MAAM,KAAK;AAClD;;AAGA,SAAS,WACP,SACiD;CACjD,IAAI,YAAY,QAAQ,OAAO,YAAY,UAAU,OAAO,KAAA;CAC5D,MAAM,IAAI;CACV,MAAM,SAAS,OAAO,EAAE,WAAW,WAAW,EAAE,SAAS,KAAA;CACzD,MAAM,QAAQ,OAAO,EAAE,UAAU,WAAW,EAAE,QAAQ,KAAA;CACtD,IAAI,WAAW,KAAA,KAAa,UAAU,KAAA,GAAW,OAAO,KAAA;CACxD,OAAO;EAAE;EAAQ;CAAM;AACzB;;;;;;AAqDA,SAAS,eACP,KAMA,SACsB;CACtB,OAAO;EACL,UAAU,IAAI;EACd,OAAO,IAAI;EACX,OACE,SAAS,aAAa,IAAI,YAAY,8BAA8B;EACtE,OAAO,SAAS,SAAS,IAAI,YAAY,eAAe;EACxD,QAAQ,WAAW,IAAI,OAAO;EAC9B,QAAQ,IAAI;CACd;AACF;;;;;;;;AASA,eAAe,wBACb,OACA,OACA,QACe;CACf,IAAI,CAAC,OAAO;CACZ,MAAM,QACJ;EACE,QAAQ;EACR,QAAQ;EACR,QAAQ;CACV,EACA,MAAM;CACR,KAAK,MAAM,MAAM,CAAC,MAAM,QAAQ,MAAM,MAAM,GAAG;EAC7C,IAAI,CAAC,IAAI;EACT,IAAI;GACF,MAAM,GAAG,KAAK;EAChB,SAAS,OAAO;GAGd,QAAQ,OAAO,4BAA4B;IACzC,MAAM,MAAM;IACZ,MAAM,MAAM;IACZ;GACF,CAAC;EACH;CACF;AACF;AAEA,SAAgB,YACd,YACA,SAKA;CACA,OAAO,qBAAqB;EAC1B,MAAM;EACN,UAAU,CAAC,mBAAmB,oBAAoB;EAOlD,kBAAkB,CAAC,gCAAgC,eAAe;EAElE,MAAM,MAAM,KAAK;GACf,MAAM,YAAY,eAAe,KAAK,OAAO;GAS7C,MAAM,aAAa,yBAAkC,OAAO;GAC5D,IAAI,eAAe,KAAA,GAAW;IAC5B,yBAAyB,KAAK,UAAU;IAGxC,qBAAqB,KAAK,IAAI;GAChC;GAKA,MAAM,UAAU,kBAAkB,KAAK,EAAE,UAAU,KAAK,CAAC;GACzD,MAAM,SAAS,SAAS;GAsBxB,MAAM,QAAyB;IAC7B;IACA,cAAc,CAAC;IACf,aAAa,0BAA0B;IACvC,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC;IAC3B,GAAI,aAAa,EAAE,WAAW,IAAI,CAAC;GACrC;GACA,SAAS,IAAI,KAAK,KAAK;GAkBvB,IAAI,eAAe,KAAA,GAAW;IAC5B,IAAI;KACF,MAAM,WAAW,KAAK,eAAe;MACnC,OAAO,IAAI;MACX,UAAU,IAAI;MACd,WAAW,KAAK,IAAI;KACtB,CAAC;IACH,SAAS,OAAO;KAGd,QAAQ,KAAK,wCAAwC;MACnD,OAAO,IAAI;MACX;KACF,CAAC;IACH;IAyBA,IAAI;KACF,MAAM,eAAe,KAAK,EAAE,UAAU,KAAK,CAAC,CAAC,EAAE,SAAS;IAC1D,SAAS,OAAO;KAGd,QAAQ,KAAK,wCAAwC;MACnD,OAAO,IAAI;MACX;KACF,CAAC;IACH;GACF;GAcA,IAAI,eAAe,KAAA,KAAa,WAAW,oBACzC,iBAAiB,KAAK,EAAE,UAAU,KAAK,CAAC,CAAC,EAAE,UAAU,YAAY;IAU/D,IAAI,MAAM,aAAa,YAAY,IAAI,OAAO,KAAK,GAAG;IACtD,IACE,MAAM,aAAa,YAAY,OAAO,YAAY,KAAK,YAAY,GAEnE,MAAM,QAAQ,QACZ,yDACA,EAAE,OAAO,IAAI,MAAM,CACrB;GAEJ,CAAC;GAGH,MAAM,SAAS,MAAM,WAAW,OAAO,SAAS;GAGhD,MAAM,SAAS;GACf,eAAe,KAAK,MAAM;GAC1B,IAAI,WAAW,QAAQ,qBAAqB,KAAK,WAAW,MAAM;GAUlE,IACE,eAAe,KAAA,MACd,UAAU,UAAU,KAAA,KACnB,UAAU,iBAAiB,oBAE7B,QAAQ,KACN,mRAIA,EAAE,OAAO,IAAI,MAAM,CACrB;GAGF,MAAM,YAAY,WAAW,WAAW,QAAA;GACxC,IAAI,UAAU;GACd,IAAI;IACF,MAAM,SAAS,MAAM,OAAO,QAAQ,KAAK,sBAAsB,EAC7D,KAAK,UACP,CAAC;IACD,IAAI,OAAO,aAAa,GAAG;KACzB,UAAU,OAAO,OAAO,KAAK;KAC7B,QAAQ,QAAQ,iCAAiC;MAC/C,MAAM;MACN;KACF,CAAC;IACH,OAKE,QAAQ,QAAQ,oDAAoD;KAClE,MAAM;KACN,UAAU,OAAO;KACjB,QAAQ,OAAO;IACjB,CAAC;GAEL,SAAS,OAAO;IAId,QAAQ,KAAK,uCAAuC;KAClD,MAAM;KACN;IACF,CAAC;GACH;GAEA,MAAM,YAAY,WAAW;GAC7B,IAAI,cAAc,KAAA,GAAW;IAC3B,MAAM,OAAO,UAAU,QAAA;IACvB,MAAM,gBAAgB,qBAAqB,SAAS;IACpD,MAAM,UAAU,UAAU;IAC1B,2BAA2B,KAAK;KAC9B,QAAQ,UAAU,UAAU,CAAC;KAC7B,SAAS,UAAU,WAAW,CAAC;KAC/B,gBAAgB,QAAQ;MACtB,IAAI,YAAY,KAAA,GACd,MAAM,IAAI,MACR,8DAA8D,IAAI,aAAa,GACjF;MAEF,OAAO,cAAc,SAAS,GAAG;KACnC;KACA,YAAY,GAAG,KAAK,uBAAuB;KAC3C;KACA,GAAI,UAAU,YAAY,KAAA,IACtB,EAAE,SAAS,UAAU,QAAQ,IAC7B,CAAC;IACP,CAAC;GACH;GAEA,MAAM,QAAQ,WAAW;GACzB,MAAM,OAAO,UAAU,MAAM;GAE7B,MAAM,KAAK,kBAAkB,WAAW,UAAU;GAKlD,MAAM,eAAe,MAAM;GAC3B,IAAI;GACJ,IAAI,GAAG,SAAS;IACd,UAAU,MAAM,eAAe,QAAQ;KACrC,UAAU,UAA4B;MACpC,MAAM,WAAW,mBACf,QACA,WACA,SACA,OACA,MACF;MACA,wBAA6B,OAAO,UAAU,MAAM;MACpD,SAAS,KAAK,QAAQ;MACtB,IAAI,GAAG,MACL,aAAa,KACX,SACG,KAAK,CAAC,CACN,MAAM,SAAS;OACd,SAAS,aAAa;QAAE,MAAM,MAAM;QAAM;OAAK,CAAC;MAClD,CAAC,CAAC,CACD,OAAO,UAAmB;OACzB,QAAQ,KAAK,iCAAiC;QAC5C,MAAM,MAAM;QACZ;OACF,CAAC;MACH,CAAC,CACL;KAEJ;KAMA,MAAM;KACN,GAAI,IAAI,WAAW,KAAA,IAAY,EAAE,QAAQ,IAAI,OAAO,IAAI,CAAC;KACzD,GAAI,WAAW,KAAA,IAAY,EAAE,OAAO,IAAI,CAAC;IAC3C,CAAC;IACD,QAAQ,QAAQ,2BAA2B;KACzC,MAAM;KACN,MAAM,GAAG;IACX,CAAC;GACH;GAOA,IAAI,SAAS,MAAM,UAAU;EAC/B;EAOA,SAAS,MAAM,QAAQ;GACrB,MAAM,WAAW,uBAAuB,OAAO,QAAQ;GACvD,IAAI,SAAS,WAAW,OAAO,SAAS,QAAQ;GAChD,OAAO,EAAE,SAAS;EACpB;EAOA,YAAY,KAAK;GACf,SAAS,IAAI,GAAG,CAAC,EAAE,YAAY,UAAU,GAAG;EAC9C;EAIA,QAAQ,KAAK,OAAO;GAClB,SAAS,IAAI,GAAG,CAAC,EAAE,YAAY,QAAQ,OAAO,GAAG;EACnD;EAEA,MAAM,SAAS,KAAK;GAClB,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GACZ,MAAM,EAAE,QAAQ,cAAc;GAK9B,MAAM,YAAY,UAAU,GAAG;GAE/B,MAAM,aAAa,OAAO,QAAQ;GAElC,MAAM,YAAY,WAAW;GAI7B,IACE,WAAW,aAAa,eACxB,QAAQ,aAAa,aACrB,OAAO,UACP;IACA,MAAM,WAAW,MAAM,OAAO,SAAS,aAAa,IAAI,OAAO;IAC/D,MAAM,QAAQ,UAAU;IACxB,IAAI,OAAO;KACT,MAAM,MAAM,WAAW,IAAI,SAAS;KACpC,MAAM,WAAW,MAAM,MAAM,IAAI,GAAG;KACpC,IAAI,UACF,MAAM,MAAM,OAAO;MACjB,GAAG;MACH,kBAAkB,SAAS;MAC3B,WAAW,KAAK,IAAI;KACtB,CAAC;IAEL;GACF;GAEA,IAAI,WAAW,mBAAmB;IAChC,MAAM,WAAW,QAAQ,SAAS;IAClC,MAAM,WAAW,OAAO,YAAY;GACtC;EACF;EAEA,MAAM,QAAQ,KAAK,MAAiB;GAClC,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GAKZ,MAAM,aAAa,OAAO,OAAO;GAEjC,MAAM,aAAa,MAAM;GACzB,MAAM,YAAY,MAAM,aACtB,YACA,IAAI,OACJ,KAAK,oBAAoB,IAC3B;GAEA,IACE,eAAe,KAAA,KACf,CAAC,aACD,WAAW,oBACX;IAeA,IAAI,MAAM,aAAa,YAAY,OAAO,YAAY,KAAK,OAAO,GAChE;IAEF,MAAM,WAAW,QAAQ,MAAM,SAAS;IACxC,MAAM,WAAW,OAAO,YAAY;IACpC;GACF;GAQA,MAAM,WAAW,QAAQ,MAAM,SAAS;GACxC,MAAM,WAAW,OAAO,YAAY;EACtC;EAEA,MAAM,QAAQ,KAAK,MAAM;GACvB,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GAEZ,MAAM,aAAa,OAAO,OAAO;GACjC,MAAM,WAAW,OAAO,UAAU,KAAK,KAAK;GAI5C,IAAI,WAAW,WAAW,mBAAmB;IAC3C,MAAM,WAAW,QAAQ,MAAM,SAAS;IACxC,MAAM,WAAW,OAAO,YAAY;GACtC;EACF;CACF,CAAC;AACH"}
|
|
1
|
+
{"version":3,"file":"middleware.js","names":[],"sources":["../../src/middleware.ts"],"sourcesContent":["/**\n * `withSandbox(definition, options?)` — the middleware that PROVIDES the\n * {@link SandboxCapability} a harness adapter requires.\n *\n * - `setup`: resume-or-create the sandbox (via the definition's ensure\n * algorithm), provide the handle, using the durability seams from\n * {@link SandboxMiddlewareOptions} (or, failing that, a bus-provided\n * SandboxInstanceStoreCapability / LocksCapability, then an in-memory\n * fallback). If `fileEvents` is not false, starts a\n * watcher that dispatches to sandbox-scoped hooks and forwards to the runtime\n * sink.\n * - `onFinish`/`onAbort`/`onError`: stop the watcher, snapshot (`after-run`)\n * and/or destroy per lifecycle.\n *\n * NOTE: streamed sandbox lifecycle events (sandbox.created, workspace.setup.*)\n * are emitted by the harness adapter's chatStream (which can yield CUSTOM\n * chunks), not from here — middleware setup runs before streaming begins.\n */\nimport {\n defineChatMiddleware,\n provideDetachableRun,\n provideRunDetached,\n wasCancelRequested,\n} from '@tanstack/ai'\nimport { InMemoryLockStore, LocksCapability } from '@tanstack/ai/locks'\nimport {\n getPendingTurn,\n getRunDisconnect,\n getSandboxRuntime,\n} from '@tanstack/ai/adapter-internals'\nimport {\n SandboxCapability,\n provideSandbox,\n provideSandboxPolicy,\n} from './capabilities'\nimport {\n provideSandboxDurability,\n resolveSandboxDurability,\n} from './durability'\nimport { SandboxInstanceStoreCapability } from './instance-store'\nimport { computeWorkspaceHash } from './key'\nimport { buildFileHookEvent, resolveFileEvents } from './file-diff'\nimport { ProjectionCapability, provideWorkspaceProjection } from './projection'\nimport { resolveSecret } from './secrets'\nimport {\n createToolHistoryRecorder,\n stripObservedToolCalls,\n} from './tool-history'\nimport { watchWorkspace } from './watch'\nimport { DEFAULT_WORKSPACE_ROOT } from './bootstrap'\nimport { resolveHarnessCwd } from './harness-cwd'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { LockStore } from '@tanstack/ai/locks'\nimport type {\n AbortInfo,\n ChatMiddlewareContext,\n DefinedChatMiddleware,\n RunStore,\n SandboxFileEvent,\n SandboxFileHookEvent,\n} from '@tanstack/ai'\nimport type {\n SandboxDurabilityOptions,\n SandboxRunDurability,\n} from './durability'\nimport type { SandboxInstanceStore } from './instance-store'\nimport type { ToolHistoryRecorder } from './tool-history'\nimport type { SandboxHandle } from './contracts'\nimport type {\n SandboxDefinition,\n SandboxEnsureContext,\n SandboxHooks,\n} from './sandbox'\nimport type { SandboxWatchHandle } from './watch'\n\n/** Per-request state we need to carry from `setup` to the terminal hooks. */\ninterface SandboxRunState {\n /**\n * OPTIONAL because the state is registered BEFORE `definition.ensure()` is\n * awaited, and `ensure` is the slowest thing in the whole run — cloning a repo\n * into a fresh sandbox is minutes wide. That window is where the most common\n * disconnect of all lands (a user starts a run and switches away while the UI\n * still says \"starting the sandbox\"), so it is the one window the teardown and\n * disconnect hooks most need to be able to act in. Registering only after the\n * handle exists left exactly that window uncovered.\n *\n * Nothing the disconnect path does needs the handle: `detachedSince` and\n * `sandboxKey` come from `ensureCtx`, which is built before `ensure` is called.\n * Only `onFinish`'s snapshot needs it, and that cannot run before `setup` has\n * completed.\n */\n handle?: SandboxHandle\n ensureCtx: SandboxEnsureContext\n watcher?: SandboxWatchHandle\n /** In-flight `enriched.diff()` promises queued by the `fileEvents.diff`\n * watcher callback, awaited before teardown so a pending diff isn't\n * dropped when the run finishes/aborts/errors mid-computation. */\n pendingDiffs: Array<Promise<void>>\n /** Logger captured at setup, so terminal hooks can log watcher teardown. */\n logger?: InternalLogger\n /**\n * Durability resolved once at setup (absent when the run is not durable), so\n * `onAbort` cannot reach a different verdict than the one `setup` published\n * on the capability bus.\n */\n durability?: SandboxRunDurability\n /**\n * Records the harness's own tool calls into the transcript, so a finished run\n * restores its tool cards from the message store instead of only from the (live,\n * rejoin-only) delivery log. See `./tool-history`.\n */\n toolHistory: ToolHistoryRecorder\n}\n\nconst runState = new WeakMap<object, SandboxRunState>()\n\n/**\n * Stop the watcher and drain any in-flight `diff()` promises before teardown,\n * so the final file's diff isn't dropped when a run finishes/aborts/errors\n * mid-computation. The `pendingDiffs` await is the load-bearing line — without\n * it a deferred diff resolves after the run is gone and its chunk is lost.\n */\nasync function drainWatcher(\n state: SandboxRunState,\n phase: 'finish' | 'abort' | 'error',\n): Promise<void> {\n // Guard `stop()`: a rejecting watcher teardown must NOT propagate out of\n // here, or the caller skips the `definition.destroy(...)` that follows —\n // leaking the sandbox on exactly the abort path that must ALWAYS tear down.\n try {\n await state.watcher?.stop()\n } catch (error) {\n state.logger?.warn('sandbox watcher stop failed', { phase, error })\n }\n await Promise.allSettled(state.pendingDiffs)\n if (state.watcher) state.logger?.sandbox('sandbox watcher stopped', { phase })\n}\n\n/**\n * Record the two facts a later attach and the reaper both need, then publish the\n * detach verdict core reads.\n *\n * Shared by the DISCONNECT subscriber registered in `setup` (the run is still\n * going — the normal case) and `onAbort`'s detach branch (the run is being torn\n * down while detachable), so the two can never write a different shape of detach.\n *\n * GUARDED, and reports failure rather than throwing. `update` is a documented\n * no-op for an unknown runId, so a vanished record does not turn teardown into a\n * throw; a genuinely rejecting store is the caller's to react to — `onAbort` falls\n * through to destroying the sandbox, because a DESTROYED sandbox beats an\n * unreachable one, while the disconnect subscriber has nothing to fall back to\n * (the run is alive and still using the sandbox) and simply leaves the verdict\n * unpublished.\n *\n * The verdict is published ONLY on success. Publishing it after a failed record\n * write would leave core holding the log open for a takeover that can never be\n * found, since nothing in the store points at the run.\n */\nasync function recordDetach(\n definition: SandboxDefinition,\n state: SandboxRunState,\n durability: SandboxRunDurability,\n ctx: ChatMiddlewareContext,\n phase: 'disconnect' | 'abort',\n): Promise<boolean> {\n try {\n // The record already exists: `setup` pre-creates it for every durable run\n // BEFORE `ensure`, precisely so this stamp cannot land on a runId the store has\n // never heard of — `RunStore.update` is a documented no-op for an unknown\n // runId, which is how the detach used to be lost silently (measured against the\n // browser repro: `detached_since` and `sandbox_key` both stayed NULL for a run\n // that had genuinely detached). If it has since vanished, that no-op is the\n // correct outcome and this must not throw.\n await durability.runs.update(ctx.runId, {\n detachedSince: Date.now(),\n sandboxKey: definition.key(state.ensureCtx),\n })\n } catch (error) {\n state.logger?.warn('sandbox detach record write failed', {\n runId: ctx.runId,\n phase,\n error,\n })\n return false\n }\n // Core's durable delivery sink reads this (see `RunDetachedCapability`) and\n // leaves the run's log OPEN instead of appending a synthetic terminal\n // `RUN_ERROR` and closing it — a terminalized log ends a later attach's replay\n // at the prefix and diverges the takeover's journal replay, which recorded a\n // healthy detached run as `'failed'`.\n provideRunDetached(ctx, true)\n return true\n}\n\n/**\n * Whether an out-of-band cancel has been recorded for this run, in EITHER band.\n * A user pressing Stop and a user closing the tab produce the IDENTICAL\n * connection close, so intent is never inferred from the disconnect itself: it\n * arrives in-process (the abort reason carried the cancel sentinel) or durably\n * (another host recorded it on the run record).\n */\nasync function cancelIntent(\n durability: SandboxRunDurability | undefined,\n runId: string,\n inProcess: boolean,\n): Promise<boolean> {\n if (inProcess) return true\n if (durability === undefined) return false\n // No guard needed here, and one would be dead code: `wasCancelRequested` already\n // answers `false` for a store read that rejects. That matters on this path,\n // because a rejection escaping into `onAbort` would skip BOTH of its branches at\n // once, leaving a sandbox that is neither reclaimable nor destroyed. The test\n // 'DETACHES when the cancel probe REJECTS' pins the composition.\n return wasCancelRequested(durability.runs, runId)\n}\n\n/** Defensively pull tenant scoping out of the runtime context, if present. */\nfunction tenantFrom(\n context: unknown,\n): { userId?: string; orgId?: string } | undefined {\n if (context === null || typeof context !== 'object') return undefined\n const c = context as Record<string, unknown>\n const userId = typeof c.userId === 'string' ? c.userId : undefined\n const orgId = typeof c.orgId === 'string' ? c.orgId : undefined\n if (userId === undefined && orgId === undefined) return undefined\n return { userId, orgId }\n}\n\n/**\n * Durability seams for a sandboxed run. Both are optional; each independently\n * falls back to a process-lifetime in-memory default, which is correct for a\n * single process but NOT across replicas.\n */\nexport interface SandboxMiddlewareOptions<TOffset extends string = string> {\n /**\n * Durable instance map (which provider sandbox to resume for a key). Pass\n * your own store to make resume survive across processes/replicas.\n *\n * Takes precedence over a store provided on the capability bus (see\n * `provideSandboxInstanceStore`), so the call site wins over ambient wiring.\n */\n instances?: SandboxInstanceStore\n /**\n * Distributed lock serializing resume-or-create for one key. Needed for\n * multi-replica correctness so two concurrent runs don't both create.\n *\n * Prefer `withLocks` from `@tanstack/ai/locks` when other middleware also\n * needs the lock; use this option to scope one to this sandbox. Takes\n * precedence over a bus-provided lock.\n */\n locks?: LockStore\n /**\n * Run lifecycle records. Pair with `durability.adapter` to make a run\n * DETACHABLE: a client disconnect then leaves the agent running and records\n * `detachedSince` instead of destroying the sandbox.\n *\n * Pass the SAME store chat persistence uses (`persistence.stores.runs`) so\n * one record describes the run instead of two that can disagree.\n *\n * Defaults to `undefined`: an app that passes neither this nor `durability`\n * keeps today's destroy-on-disconnect behavior exactly.\n */\n runs?: RunStore\n /**\n * Delivery durability for the run's event log, plus the journal and detach\n * knobs. Requires `runs`; either alone is not durable.\n *\n * `TOffset` is inferred from the adapter passed here, so a branded-cursor\n * backend (`durableStream`) wires without a cast and without the call site\n * ever naming the parameter.\n */\n durability?: SandboxDurabilityOptions<TOffset>\n}\n\n/**\n * Resolve the ensure seams. Precedence is explicit option → capability bus →\n * (in `ensure`) the in-memory fallback. The option wins because it is visible\n * at the call site; the bus remains for platform/framework injection.\n */\nfunction buildEnsureCtx(\n ctx: ChatMiddlewareContext,\n // Narrowed to the two seams it reads rather than taking the whole options\n // object: `SandboxMiddlewareOptions` is now generic in the durability offset,\n // and `SandboxMiddlewareOptions<TOffset>` is not assignable to\n // `SandboxMiddlewareOptions<string>`. Both members here are offset-free, so\n // the narrowing keeps this helper independent of that parameter entirely.\n options: Pick<SandboxMiddlewareOptions, 'instances' | 'locks'> | undefined,\n): SandboxEnsureContext {\n return {\n threadId: ctx.threadId,\n runId: ctx.runId,\n store:\n options?.instances ?? ctx.getOptional(SandboxInstanceStoreCapability),\n locks: options?.locks ?? ctx.getOptional(LocksCapability),\n tenant: tenantFrom(ctx.context),\n signal: ctx.signal,\n adapterName: ctx.provider,\n }\n}\n\n/**\n * Dispatch a sandbox file event to the per-type hooks declared on the\n * definition. Errors in individual hooks are swallowed so one bad hook\n * cannot break the run — but are logged under the `errors` category first, so\n * a throwing hook is observable (matching the run-scoped path in the engine\n * and the behavior the observability docs promise).\n */\nasync function dispatchDefinitionHooks(\n hooks: SandboxHooks | undefined,\n event: SandboxFileHookEvent,\n logger?: InternalLogger,\n): Promise<void> {\n if (!hooks) return\n const typed = (\n {\n create: 'onFileCreate',\n change: 'onFileChange',\n delete: 'onFileDelete',\n } as const\n )[event.type]\n for (const fn of [hooks.onFile, hooks[typed]]) {\n if (!fn) continue\n try {\n await fn(event)\n } catch (error) {\n // swallowed — one bad hook must not break the run — but logged so the\n // failure isn't invisible.\n logger?.errors('sandbox file hook failed', {\n path: event.path,\n type: event.type,\n error,\n })\n }\n }\n}\n\nexport function withSandbox<TOffset extends string = string>(\n definition: SandboxDefinition,\n options?: SandboxMiddlewareOptions<TOffset>,\n): DefinedChatMiddleware<\n unknown,\n readonly [],\n readonly [typeof SandboxCapability, typeof ProjectionCapability]\n> {\n return defineChatMiddleware({\n name: 'sandbox',\n provides: [SandboxCapability, ProjectionCapability],\n // SandboxPolicyCapability is provided conditionally (only when the\n // definition has a policy), so it is intentionally NOT declared here —\n // consumers read it via `getOptional`. SandboxDurabilityCapability and\n // DetachableRunCapability are conditional for the same reason (only when\n // `runs` + `durability` are both wired), so they are intentionally NOT\n // declared here either.\n optionalRequires: [SandboxInstanceStoreCapability, LocksCapability],\n\n async setup(ctx) {\n const ensureCtx = buildEnsureCtx(ctx, options)\n\n // Resolving here (not lazily on the abort path) is what keeps `setup` and\n // `onAbort` on one verdict: the payload the bus carries is the same object\n // the teardown path consults.\n // `TOffset` is passed explicitly: `options` is possibly `undefined` here,\n // so inference has nothing to work from on that branch and would fall\n // back to the `= string` default, re-erecting the very wall this\n // parameter exists to remove.\n const durability = resolveSandboxDurability<TOffset>(options)\n if (durability !== undefined) {\n provideSandboxDurability(ctx, durability)\n // A neutral boolean core owns, so `@tanstack/ai-persistence` can ask\n // \"is this run detachable?\" without depending on this package.\n provideDetachableRun(ctx, true)\n }\n\n // Pull the runtime (and its logger) up front so `baseSha` capture and\n // hook dispatch below can log through the same `sandbox`/`errors`\n // categories the engine uses.\n const runtime = getSandboxRuntime(ctx, { optional: true })\n const logger = runtime?.logger\n\n // REGISTER THE RUN STATE NOW — before `definition.ensure()`, not merely\n // before the end of `setup`.\n //\n // `onAbort` and the disconnect subscriber both need this state, so until\n // this map is populated they are silent no-ops. `ensure` is the LONGEST\n // await in the entire run (create a sandbox, clone a repo — minutes), and it\n // is where the most common disconnect of all lands: a user starts a run and\n // switches away while the UI still says \"starting the sandbox\". Registering\n // after `ensure` returned still left that whole window uncovered.\n //\n // Leaving it uncovered loses every teardown behavior at once: no\n // `detachedSince`/`sandboxKey`, so `listReclaimable` can never surface the\n // run and the reaper can never reclaim it; no `definition.destroy`, so the\n // sandbox leaks; and no detach verdict for core to read.\n //\n // Everything those hooks read is already resolved above: the ensure context\n // (which is all `definition.key` needs), the durability verdict, and the\n // logger. The fields discovered later (`handle`, `watcher`) are ASSIGNED onto\n // this same object as they become available, so the teardown path always\n // sees the most complete state that exists at the moment it runs.\n const state: SandboxRunState = {\n ensureCtx,\n pendingDiffs: [],\n toolHistory: createToolHistoryRecorder(),\n ...(logger ? { logger } : {}),\n ...(durability ? { durability } : {}),\n }\n runState.set(ctx, state)\n\n // MAKE THE RUN FINDABLE BEFORE `ensure`, not after the run finally streams.\n //\n // Chat persistence creates the run record from `onConfig`, which runs after\n // EVERY middleware `setup` — so for the whole of `definition.ensure` (create a\n // sandbox, clone a repo: minutes) the run has no record at all, and\n // `findActiveRun` answers \"no active run\" for a run that is demonstrably\n // starting. Measured: a status sidebar read straight off `findActiveRun`\n // reported `idle` for 6.5 minutes while the sandbox was being built, and a\n // client returning to the thread in that window had nothing to tell it a run\n // was in flight — so it rendered an empty pane instead of \"starting sandbox\".\n //\n // A crash in the same window is worse: no record means `listReclaimable` can\n // never surface the run, so the sandbox leaks with no recovery path.\n //\n // `createOrResume` is idempotent and never resurrects a finished run, so\n // persistence's own later call stays correct and simply finds this record.\n if (durability !== undefined) {\n try {\n await durability.runs.createOrResume({\n runId: ctx.runId,\n threadId: ctx.threadId,\n startedAt: Date.now(),\n })\n } catch (error) {\n // Best-effort: a store blip must not stop a run that is otherwise fine.\n // The run is simply invisible until persistence's own `onConfig` call.\n logger?.warn('sandbox run record pre-create failed', {\n runId: ctx.runId,\n error,\n })\n }\n\n // NO ATTACH MARKER HERE. A joiner does need a chunk in the log before the\n // harness has emitted anything — an empty log fails every joiner's\n // fast-fail (`memoryStream`'s first-chunk deadline, the client's rejoin\n // connect deadline) and flushes no HTTP headers, so a reload during\n // `ensure` reads a live run as gone. Core does it: a fresh durable producer\n // appends `RUN_ACCEPTED_EVENT` before the producer stream is first pulled,\n // for EVERY durable run rather than only sandboxed ones, and never on an\n // attach. A second marker from here would only land mid-stream in a run\n // that is already producing.\n\n // STORE THE USER'S TURN NOW, before `ensure` takes minutes.\n //\n // Chat persistence stores it from `onStart`, which runs after every\n // middleware `setup` — so without this the thread holds NOTHING for the\n // whole sandbox build. Measured: a reload during the build asked the server\n // for the conversation and got `{\"messages\":[],…}`, so the user saw no sign\n // of the message they had just sent, and a second device saw an empty\n // thread.\n //\n // The persistence layer owns WHAT gets stored (see `PendingTurnCapability`):\n // `saveThread` replaces the thread, so deciding the list here would risk\n // deleting the history. Absent when the app wires no persistence, which is\n // simply a run with no transcript to store.\n try {\n await getPendingTurn(ctx, { optional: true })?.snapshot()\n } catch (error) {\n // Best-effort: the run is still worth doing, and `onStart` stores the\n // turn again once setup completes.\n logger?.warn('sandbox pending-turn snapshot failed', {\n runId: ctx.runId,\n error,\n })\n }\n }\n\n // SUBSCRIBE BEFORE `ensure`, for the same reason the state is registered\n // before it: `ensure` is the minutes-wide await a disconnect actually lands\n // in. Core calls back immediately if the socket has already closed, so\n // subscribing here cannot miss a disconnect that beat us to it.\n //\n // This is what makes a durable run SURVIVE losing its viewer. The only route\n // a disconnect previously had into this middleware was the application\n // mirroring `request.signal` into `chat()`'s `abortController` — which aborts\n // the run, so `chat()` returned right after this `setup` and the harness\n // adapter's `chatStream` was never called: the agent in the sandbox we just\n // spent minutes creating was NEVER LAUNCHED, and no takeover could recover it\n // because an agent that never ran wrote no journal to replay.\n if (durability !== undefined && durability.detachOnDisconnect) {\n getRunDisconnect(ctx, { optional: true })?.subscribe(async () => {\n // BOOKKEEPING ONLY — the run is still executing. Deliberately absent:\n // `drainWatcher` (would blind a live agent's file events for the whole\n // remainder) and `definition.destroy` (the run is still using the\n // sandbox). Both belong to the terminal hooks, which still run exactly\n // once afterwards.\n //\n // A run with a cancel already recorded is left alone: that is `onAbort`'s\n // path, and stamping `detachedSince` on a deliberately-stopped run would\n // hand it to the reaper as reclaimable work.\n if (await cancelIntent(durability, ctx.runId, false)) return\n if (\n await recordDetach(definition, state, durability, ctx, 'disconnect')\n ) {\n state.logger?.sandbox(\n 'sandbox run detached on disconnect; the run continues',\n { runId: ctx.runId },\n )\n }\n })\n }\n\n const handle = await definition.ensure(ensureCtx)\n // MUTATE, don't re-`set`: a disconnect that landed during `ensure` already\n // captured this object.\n state.handle = handle\n provideSandbox(ctx, handle)\n if (definition.policy) provideSandboxPolicy(ctx, definition.policy)\n\n // Deliberately placed AFTER `logger` is in scope rather than next to the\n // `provideSandboxDurability` call above — there is no logger to warn\n // through until the runtime has been read.\n //\n // `ensureCtx.locks === undefined` counts as in-memory: `defineSandbox`'s\n // `ensure` falls back to a process-lifetime `InMemoryLockStore` when no\n // lock is wired, so an unwired lock has exactly the deficiency being\n // warned about — it is the MOST in-memory case, not an exempt one.\n if (\n durability !== undefined &&\n (ensureCtx.locks === undefined ||\n ensureCtx.locks instanceof InMemoryLockStore)\n ) {\n logger?.warn(\n 'sandbox durability is wired over an InMemoryLockStore: run claims are ' +\n 'serialized within this process only and the lease never signals loss, ' +\n 'so two hosts can drive one run and duplicate its event log. Use a ' +\n 'distributed LockStore via withLocks for any multi-replica deploy.',\n { runId: ctx.runId },\n )\n }\n\n const watchRoot = definition.workspace?.root ?? DEFAULT_WORKSPACE_ROOT\n let baseSha = ''\n try {\n const shaRes = await handle.process.exec('git rev-parse HEAD', {\n cwd: watchRoot,\n })\n if (shaRes.exitCode === 0) {\n baseSha = shaRes.stdout.trim()\n logger?.sandbox('sandbox git baseline captured', {\n root: watchRoot,\n baseSha,\n })\n } else {\n // Non-zero exit: either not a git repository (non-git workspace) or a\n // repo with no commits (no HEAD). Expected, but it silently degrades\n // every subsequent diff to a full-file add-patch, so surface it\n // under `sandbox` (with stderr) rather than leaving nothing to grep.\n logger?.sandbox('sandbox git baseline unavailable (non-zero exit)', {\n root: watchRoot,\n exitCode: shaRes.exitCode,\n stderr: shaRes.stderr,\n })\n }\n } catch (error) {\n // exec rejected (git not on PATH, exec seam broken) → baseSha stays ''\n // and accessors fall back, but this is a real anomaly, not a plain\n // non-git workspace, so warn.\n logger?.warn('sandbox git baseline capture failed', {\n root: watchRoot,\n error,\n })\n }\n\n const workspace = definition.workspace\n if (workspace !== undefined) {\n const virtualRoot = workspace.root ?? DEFAULT_WORKSPACE_ROOT\n const root = resolveHarnessCwd(handle, virtualRoot)\n const workspaceHash = computeWorkspaceHash(workspace)\n const secrets = workspace.secrets\n provideWorkspaceProjection(ctx, {\n skills: workspace.skills ?? [],\n plugins: workspace.plugins ?? [],\n resolveSecret: (ref) => {\n if (secrets === undefined) {\n throw new Error(\n `resolveSecret: no secrets defined on this workspace (ref: \"${ref.__secretName}\")`,\n )\n }\n return resolveSecret(secrets, ref)\n },\n markerPath: `${root}/.tanstack-projected-${workspaceHash}`,\n root,\n ...(workspace.scripts !== undefined\n ? { scripts: workspace.scripts }\n : {}),\n })\n }\n\n const hooks = definition.hooks\n await hooks?.onReady?.(handle)\n\n const fe = resolveFileEvents(definition.fileEvents)\n // THE SAME array the run state already holds, not a fresh one. The watcher\n // callback below closes over this reference, and `drainWatcher` awaits\n // `state.pendingDiffs` — a second array would silently drop every in-flight\n // diff from the teardown drain.\n const pendingDiffs = state.pendingDiffs\n let watcher: SandboxWatchHandle | undefined\n if (fe.enabled) {\n watcher = await watchWorkspace(handle, {\n onEvent: (event: SandboxFileEvent) => {\n const enriched = buildFileHookEvent(\n handle,\n watchRoot,\n baseSha,\n event,\n logger,\n )\n void dispatchDefinitionHooks(hooks, enriched, logger)\n runtime?.emit(enriched)\n if (fe.diff) {\n pendingDiffs.push(\n enriched\n .diff()\n .then((diff) => {\n runtime?.emitFileDiff({ path: event.path, diff })\n })\n .catch((error: unknown) => {\n logger?.warn('sandbox file diff emit failed', {\n path: event.path,\n error,\n })\n }),\n )\n }\n },\n // Watch the SAME root the enrichment layer relativizes against\n // (`buildFileHookEvent(handle, watchRoot, …)` and the `baseSha`\n // capture). Without this the watcher defaults to `/workspace` while\n // enrichment uses `watchRoot`, so a custom `workspace.root` makes the\n // two look at different directories and git pathspecs break.\n root: watchRoot,\n ...(ctx.signal !== undefined ? { signal: ctx.signal } : {}),\n ...(logger !== undefined ? { logger } : {}),\n })\n logger?.sandbox('sandbox watcher started', {\n root: watchRoot,\n diff: fe.diff,\n })\n }\n\n // MUTATE the object registered above rather than `set`-ing a second one: an\n // abort that landed mid-setup already captured a reference to it (and may\n // already be draining `pendingDiffs`), so replacing the entry would hand the\n // teardown path a different object than the watcher writes into.\n // `pendingDiffs` needs no copying — it IS `state.pendingDiffs`.\n if (watcher) state.watcher = watcher\n },\n\n // Keep the recorded tool history OUT of the request to the model. It is stored\n // history for the next turn, it names tools the provider was never given, and one\n // triage-sized run is hundreds of kilobytes — so replaying it is wasteful at best\n // and rejected at worst. `ctx.messages` keeps it (that is what gets stored and\n // rendered); only `config.messages` loses it.\n onConfig(_ctx, config) {\n const messages = stripObservedToolCalls(config.messages)\n if (messages.length === config.messages.length) return\n return { messages }\n },\n\n // The engine re-syncs `middlewareCtx.messages` from its own array once per agent\n // iteration, which drops whatever the recorder appended during the previous\n // iteration's stream. Restoring it here — AFTER that sync — is what makes a\n // multi-iteration run keep its full history without depending on where this\n // middleware sits relative to persistence in the middleware array.\n onIteration(ctx) {\n runState.get(ctx)?.toolHistory.reconcile(ctx)\n },\n\n // Record the harness's own tool calls as transcript messages. Observe only:\n // returning nothing passes every chunk through untouched.\n onChunk(ctx, chunk) {\n runState.get(ctx)?.toolHistory.observe(chunk, ctx)\n },\n\n async onFinish(ctx) {\n const state = runState.get(ctx)\n if (!state) return\n const { handle, ensureCtx } = state\n\n // Last chance before persistence writes the transcript. Only matters if a\n // config sync landed after the final tool chunk; the recorder is idempotent, so\n // in the normal case this changes nothing.\n state.toolHistory.reconcile(ctx)\n\n await drainWatcher(state, 'finish')\n\n const lifecycle = definition.lifecycle\n\n // `handle` is absent only if `setup` never got past `definition.ensure`, in\n // which case there is no sandbox to snapshot.\n if (\n lifecycle?.snapshot === 'after-run' &&\n handle?.capabilities.snapshots &&\n handle.snapshot\n ) {\n const snapshot = await handle.snapshot(`after-run-${ctx.runId}`)\n const store = ensureCtx.store\n if (store) {\n const key = definition.key(ensureCtx)\n const existing = await store.get(key)\n if (existing) {\n await store.upsert({\n ...existing,\n latestSnapshotId: snapshot.id,\n updatedAt: Date.now(),\n })\n }\n }\n }\n\n if (lifecycle?.destroyOnComplete) {\n await definition.destroy(ensureCtx)\n await definition.hooks?.onDestroy?.()\n }\n },\n\n async onAbort(ctx, info: AbortInfo) {\n const state = runState.get(ctx)\n if (!state) return\n\n // First on BOTH branches: a diff still in flight must be drained whether\n // the sandbox is about to be destroyed or merely detached, or the final\n // file's diff is dropped.\n await drainWatcher(state, 'abort')\n\n const durability = state.durability\n const cancelled = await cancelIntent(\n durability,\n ctx.runId,\n info.cancelRequested === true,\n )\n\n if (\n durability !== undefined &&\n !cancelled &&\n durability.detachOnDisconnect\n ) {\n // DETACH on the teardown path. Reached when the run is aborted for a\n // reason that is NOT an out-of-band cancel while detachable — a genuine\n // stop from elsewhere, or a host going down. The ordinary disconnect is\n // handled by the disconnect subscriber in `setup`, which does not end the\n // run at all.\n //\n // On a failed record write this branch is ABANDONED for the destroy one\n // below, because a rejection here is the worst shape available: the\n // verdict is unpublished, so core terminalizes the log and records a\n // healthy detached run as failed; `detachedSince`/`sandboxKey` are\n // unwritten, so `listReclaimable` can never surface the run and\n // `reapDetachedRuns` can never reclaim it. A DESTROYED sandbox beats an\n // unreachable one — the same reasoning `drainWatcher` applies to its own\n // guarded `stop()`.\n if (await recordDetach(definition, state, durability, ctx, 'abort')) {\n return\n }\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n return\n }\n\n // ALWAYS tear down on an explicit abort, regardless of `destroyOnComplete`.\n // The in-sandbox agent process is not killed by closing its IO stream\n // (e.g. a Docker exec survives client disconnect), so the only reliable way\n // to stop it — and the token/cost drain of its ongoing API calls — is to\n // destroy the sandbox (stop the container/VM). `keepAlive` /\n // `destroyOnComplete:false` governs *successful completion*, never cancel.\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n },\n\n async onError(ctx, info) {\n const state = runState.get(ctx)\n if (!state) return\n\n await drainWatcher(state, 'error')\n await definition.hooks?.onError?.(info.error)\n\n // On failure, only tear down when the lifecycle says so; otherwise leave\n // the sandbox for a resumed retry.\n if (definition.lifecycle?.destroyOnComplete) {\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n }\n },\n })\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAkHA,IAAM,2BAAW,IAAI,QAAiC;;;;;;;AAQtD,eAAe,aACb,OACA,OACe;CAIf,IAAI;EACF,MAAM,MAAM,SAAS,KAAK;CAC5B,SAAS,OAAO;EACd,MAAM,QAAQ,KAAK,+BAA+B;GAAE;GAAO;EAAM,CAAC;CACpE;CACA,MAAM,QAAQ,WAAW,MAAM,YAAY;CAC3C,IAAI,MAAM,SAAS,MAAM,QAAQ,QAAQ,2BAA2B,EAAE,MAAM,CAAC;AAC/E;;;;;;;;;;;;;;;;;;;;;AAsBA,eAAe,aACb,YACA,OACA,YACA,KACA,OACkB;CAClB,IAAI;EAQF,MAAM,WAAW,KAAK,OAAO,IAAI,OAAO;GACtC,eAAe,KAAK,IAAI;GACxB,YAAY,WAAW,IAAI,MAAM,SAAS;EAC5C,CAAC;CACH,SAAS,OAAO;EACd,MAAM,QAAQ,KAAK,sCAAsC;GACvD,OAAO,IAAI;GACX;GACA;EACF,CAAC;EACD,OAAO;CACT;CAMA,mBAAmB,KAAK,IAAI;CAC5B,OAAO;AACT;;;;;;;;AASA,eAAe,aACb,YACA,OACA,WACkB;CAClB,IAAI,WAAW,OAAO;CACtB,IAAI,eAAe,KAAA,GAAW,OAAO;CAMrC,OAAO,mBAAmB,WAAW,MAAM,KAAK;AAClD;;AAGA,SAAS,WACP,SACiD;CACjD,IAAI,YAAY,QAAQ,OAAO,YAAY,UAAU,OAAO,KAAA;CAC5D,MAAM,IAAI;CACV,MAAM,SAAS,OAAO,EAAE,WAAW,WAAW,EAAE,SAAS,KAAA;CACzD,MAAM,QAAQ,OAAO,EAAE,UAAU,WAAW,EAAE,QAAQ,KAAA;CACtD,IAAI,WAAW,KAAA,KAAa,UAAU,KAAA,GAAW,OAAO,KAAA;CACxD,OAAO;EAAE;EAAQ;CAAM;AACzB;;;;;;AAqDA,SAAS,eACP,KAMA,SACsB;CACtB,OAAO;EACL,UAAU,IAAI;EACd,OAAO,IAAI;EACX,OACE,SAAS,aAAa,IAAI,YAAY,8BAA8B;EACtE,OAAO,SAAS,SAAS,IAAI,YAAY,eAAe;EACxD,QAAQ,WAAW,IAAI,OAAO;EAC9B,QAAQ,IAAI;EACZ,aAAa,IAAI;CACnB;AACF;;;;;;;;AASA,eAAe,wBACb,OACA,OACA,QACe;CACf,IAAI,CAAC,OAAO;CACZ,MAAM,QACJ;EACE,QAAQ;EACR,QAAQ;EACR,QAAQ;CACV,EACA,MAAM;CACR,KAAK,MAAM,MAAM,CAAC,MAAM,QAAQ,MAAM,MAAM,GAAG;EAC7C,IAAI,CAAC,IAAI;EACT,IAAI;GACF,MAAM,GAAG,KAAK;EAChB,SAAS,OAAO;GAGd,QAAQ,OAAO,4BAA4B;IACzC,MAAM,MAAM;IACZ,MAAM,MAAM;IACZ;GACF,CAAC;EACH;CACF;AACF;AAEA,SAAgB,YACd,YACA,SAKA;CACA,OAAO,qBAAqB;EAC1B,MAAM;EACN,UAAU,CAAC,mBAAmB,oBAAoB;EAOlD,kBAAkB,CAAC,gCAAgC,eAAe;EAElE,MAAM,MAAM,KAAK;GACf,MAAM,YAAY,eAAe,KAAK,OAAO;GAS7C,MAAM,aAAa,yBAAkC,OAAO;GAC5D,IAAI,eAAe,KAAA,GAAW;IAC5B,yBAAyB,KAAK,UAAU;IAGxC,qBAAqB,KAAK,IAAI;GAChC;GAKA,MAAM,UAAU,kBAAkB,KAAK,EAAE,UAAU,KAAK,CAAC;GACzD,MAAM,SAAS,SAAS;GAsBxB,MAAM,QAAyB;IAC7B;IACA,cAAc,CAAC;IACf,aAAa,0BAA0B;IACvC,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC;IAC3B,GAAI,aAAa,EAAE,WAAW,IAAI,CAAC;GACrC;GACA,SAAS,IAAI,KAAK,KAAK;GAkBvB,IAAI,eAAe,KAAA,GAAW;IAC5B,IAAI;KACF,MAAM,WAAW,KAAK,eAAe;MACnC,OAAO,IAAI;MACX,UAAU,IAAI;MACd,WAAW,KAAK,IAAI;KACtB,CAAC;IACH,SAAS,OAAO;KAGd,QAAQ,KAAK,wCAAwC;MACnD,OAAO,IAAI;MACX;KACF,CAAC;IACH;IAyBA,IAAI;KACF,MAAM,eAAe,KAAK,EAAE,UAAU,KAAK,CAAC,CAAC,EAAE,SAAS;IAC1D,SAAS,OAAO;KAGd,QAAQ,KAAK,wCAAwC;MACnD,OAAO,IAAI;MACX;KACF,CAAC;IACH;GACF;GAcA,IAAI,eAAe,KAAA,KAAa,WAAW,oBACzC,iBAAiB,KAAK,EAAE,UAAU,KAAK,CAAC,CAAC,EAAE,UAAU,YAAY;IAU/D,IAAI,MAAM,aAAa,YAAY,IAAI,OAAO,KAAK,GAAG;IACtD,IACE,MAAM,aAAa,YAAY,OAAO,YAAY,KAAK,YAAY,GAEnE,MAAM,QAAQ,QACZ,yDACA,EAAE,OAAO,IAAI,MAAM,CACrB;GAEJ,CAAC;GAGH,MAAM,SAAS,MAAM,WAAW,OAAO,SAAS;GAGhD,MAAM,SAAS;GACf,eAAe,KAAK,MAAM;GAC1B,IAAI,WAAW,QAAQ,qBAAqB,KAAK,WAAW,MAAM;GAUlE,IACE,eAAe,KAAA,MACd,UAAU,UAAU,KAAA,KACnB,UAAU,iBAAiB,oBAE7B,QAAQ,KACN,mRAIA,EAAE,OAAO,IAAI,MAAM,CACrB;GAGF,MAAM,YAAY,WAAW,WAAW,QAAA;GACxC,IAAI,UAAU;GACd,IAAI;IACF,MAAM,SAAS,MAAM,OAAO,QAAQ,KAAK,sBAAsB,EAC7D,KAAK,UACP,CAAC;IACD,IAAI,OAAO,aAAa,GAAG;KACzB,UAAU,OAAO,OAAO,KAAK;KAC7B,QAAQ,QAAQ,iCAAiC;MAC/C,MAAM;MACN;KACF,CAAC;IACH,OAKE,QAAQ,QAAQ,oDAAoD;KAClE,MAAM;KACN,UAAU,OAAO;KACjB,QAAQ,OAAO;IACjB,CAAC;GAEL,SAAS,OAAO;IAId,QAAQ,KAAK,uCAAuC;KAClD,MAAM;KACN;IACF,CAAC;GACH;GAEA,MAAM,YAAY,WAAW;GAC7B,IAAI,cAAc,KAAA,GAAW;IAC3B,MAAM,cAAc,UAAU,QAAA;IAC9B,MAAM,OAAO,kBAAkB,QAAQ,WAAW;IAClD,MAAM,gBAAgB,qBAAqB,SAAS;IACpD,MAAM,UAAU,UAAU;IAC1B,2BAA2B,KAAK;KAC9B,QAAQ,UAAU,UAAU,CAAC;KAC7B,SAAS,UAAU,WAAW,CAAC;KAC/B,gBAAgB,QAAQ;MACtB,IAAI,YAAY,KAAA,GACd,MAAM,IAAI,MACR,8DAA8D,IAAI,aAAa,GACjF;MAEF,OAAO,cAAc,SAAS,GAAG;KACnC;KACA,YAAY,GAAG,KAAK,uBAAuB;KAC3C;KACA,GAAI,UAAU,YAAY,KAAA,IACtB,EAAE,SAAS,UAAU,QAAQ,IAC7B,CAAC;IACP,CAAC;GACH;GAEA,MAAM,QAAQ,WAAW;GACzB,MAAM,OAAO,UAAU,MAAM;GAE7B,MAAM,KAAK,kBAAkB,WAAW,UAAU;GAKlD,MAAM,eAAe,MAAM;GAC3B,IAAI;GACJ,IAAI,GAAG,SAAS;IACd,UAAU,MAAM,eAAe,QAAQ;KACrC,UAAU,UAA4B;MACpC,MAAM,WAAW,mBACf,QACA,WACA,SACA,OACA,MACF;MACA,wBAA6B,OAAO,UAAU,MAAM;MACpD,SAAS,KAAK,QAAQ;MACtB,IAAI,GAAG,MACL,aAAa,KACX,SACG,KAAK,CAAC,CACN,MAAM,SAAS;OACd,SAAS,aAAa;QAAE,MAAM,MAAM;QAAM;OAAK,CAAC;MAClD,CAAC,CAAC,CACD,OAAO,UAAmB;OACzB,QAAQ,KAAK,iCAAiC;QAC5C,MAAM,MAAM;QACZ;OACF,CAAC;MACH,CAAC,CACL;KAEJ;KAMA,MAAM;KACN,GAAI,IAAI,WAAW,KAAA,IAAY,EAAE,QAAQ,IAAI,OAAO,IAAI,CAAC;KACzD,GAAI,WAAW,KAAA,IAAY,EAAE,OAAO,IAAI,CAAC;IAC3C,CAAC;IACD,QAAQ,QAAQ,2BAA2B;KACzC,MAAM;KACN,MAAM,GAAG;IACX,CAAC;GACH;GAOA,IAAI,SAAS,MAAM,UAAU;EAC/B;EAOA,SAAS,MAAM,QAAQ;GACrB,MAAM,WAAW,uBAAuB,OAAO,QAAQ;GACvD,IAAI,SAAS,WAAW,OAAO,SAAS,QAAQ;GAChD,OAAO,EAAE,SAAS;EACpB;EAOA,YAAY,KAAK;GACf,SAAS,IAAI,GAAG,CAAC,EAAE,YAAY,UAAU,GAAG;EAC9C;EAIA,QAAQ,KAAK,OAAO;GAClB,SAAS,IAAI,GAAG,CAAC,EAAE,YAAY,QAAQ,OAAO,GAAG;EACnD;EAEA,MAAM,SAAS,KAAK;GAClB,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GACZ,MAAM,EAAE,QAAQ,cAAc;GAK9B,MAAM,YAAY,UAAU,GAAG;GAE/B,MAAM,aAAa,OAAO,QAAQ;GAElC,MAAM,YAAY,WAAW;GAI7B,IACE,WAAW,aAAa,eACxB,QAAQ,aAAa,aACrB,OAAO,UACP;IACA,MAAM,WAAW,MAAM,OAAO,SAAS,aAAa,IAAI,OAAO;IAC/D,MAAM,QAAQ,UAAU;IACxB,IAAI,OAAO;KACT,MAAM,MAAM,WAAW,IAAI,SAAS;KACpC,MAAM,WAAW,MAAM,MAAM,IAAI,GAAG;KACpC,IAAI,UACF,MAAM,MAAM,OAAO;MACjB,GAAG;MACH,kBAAkB,SAAS;MAC3B,WAAW,KAAK,IAAI;KACtB,CAAC;IAEL;GACF;GAEA,IAAI,WAAW,mBAAmB;IAChC,MAAM,WAAW,QAAQ,SAAS;IAClC,MAAM,WAAW,OAAO,YAAY;GACtC;EACF;EAEA,MAAM,QAAQ,KAAK,MAAiB;GAClC,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GAKZ,MAAM,aAAa,OAAO,OAAO;GAEjC,MAAM,aAAa,MAAM;GACzB,MAAM,YAAY,MAAM,aACtB,YACA,IAAI,OACJ,KAAK,oBAAoB,IAC3B;GAEA,IACE,eAAe,KAAA,KACf,CAAC,aACD,WAAW,oBACX;IAeA,IAAI,MAAM,aAAa,YAAY,OAAO,YAAY,KAAK,OAAO,GAChE;IAEF,MAAM,WAAW,QAAQ,MAAM,SAAS;IACxC,MAAM,WAAW,OAAO,YAAY;IACpC;GACF;GAQA,MAAM,WAAW,QAAQ,MAAM,SAAS;GACxC,MAAM,WAAW,OAAO,YAAY;EACtC;EAEA,MAAM,QAAQ,KAAK,MAAM;GACvB,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GAEZ,MAAM,aAAa,OAAO,OAAO;GACjC,MAAM,WAAW,OAAO,UAAU,KAAK,KAAK;GAI5C,IAAI,WAAW,WAAW,mBAAmB;IAC3C,MAAM,WAAW,QAAQ,MAAM,SAAS;IACxC,MAAM,WAAW,OAAO,YAAY;GACtC;EACF;CACF,CAAC;AACH"}
|
package/dist/esm/reap.js
CHANGED
|
@@ -118,7 +118,8 @@ async function decodeFrame(stdout) {
|
|
|
118
118
|
async function probeRunExit(input) {
|
|
119
119
|
try {
|
|
120
120
|
const paths = journalPaths(input.runId, input.dir);
|
|
121
|
-
const
|
|
121
|
+
const result = await input.handle.process.exec(journalExitProbeCommand(paths, input.maxBytes ?? 4096));
|
|
122
|
+
const exitCode = parseJournalExit(await decodeFrame(result.stdout), paths);
|
|
122
123
|
return exitCode === null ? { state: "producing" } : {
|
|
123
124
|
state: "finished",
|
|
124
125
|
exitCode
|
package/dist/esm/reap.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"reap.js","names":[],"sources":["../../src/reap.ts"],"sourcesContent":["/**\n * The sweep `RunStore.listReclaimable` was always missing a consumer for: take a\n * detached run whose viewer never came back, save its transcript, terminalize its\n * record, and tear its sandbox down.\n *\n * THE ONE RULE THAT SHAPES EVERYTHING HERE: **never drive a run to find out\n * whether it finished.**\n *\n * The obvious design — hand the run to `pipeToRunLog` under a short\n * `runBudgetMs` and see whether it terminalizes — was measured and is broken.\n * `pipeToRunLog` is total by construction: it ALWAYS writes a terminal status and\n * ALWAYS calls `durability.close()`. Against a run that has not finished, all\n * three producer shapes are destructive:\n *\n * | producer's reaction to the budget signal | stored status | `close()` |\n * | ---------------------------------------- | ------------- | --------- |\n * | ignores it and keeps producing | `aborted` | called |\n * | returns on abort (the realistic `drive`) | `aborted` | called |\n * | throws an AbortError | `failed` | called |\n *\n * The middle row USED to read `completed`, which was the fatal one: a signal-aware\n * producer exits its loop NORMALLY, and `pipeToRunLog` only checked its signal\n * per chunk, so a healthy mid-flight run was recorded as `'completed'` with a\n * `finishedAt` — a false transcript. That gap is fixed (`run.ts` re-checks the\n * signal after the loop), so the status is now honest on all three rows. The rule\n * above is UNCHANGED, because the status was never the whole harm: every row\n * writes a terminal record and closes a log that commit `5a1f821c9` deliberately\n * leaves OPEN for takeover (ending every attached client's stream), and a terminal\n * record drops out of `listReclaimable` forever, so TTL expiry can never reclaim\n * that run's sandbox. A cost leak with no recovery path. There is therefore no\n * \"still running\" outcome in {@link ReapRunOutcome}: it is unreachable by\n * construction, not merely unlikely.\n *\n * So sentinel-reached is detected OUT OF BAND, through the in-sandbox journal\n * ({@link probeRunExit}), and `pipeToRunLog` is entered only for a run already\n * KNOWN to have finished, or for one whose TTL has expired (terminal either way).\n * On the FINALIZATION path `runBudgetMs` therefore degrades from a load-bearing\n * mechanism into a safety net whose expiry is a genuine anomaly — see\n * `'budget-exceeded'`. On the EXPIRY path it stays load-bearing: nothing polls the\n * cancel this module records, so the budget is what ends the drive of an expired\n * run whose agent is still producing, and its expiry there is the designed path.\n *\n * WHY THE PROBE IS INJECTED (`ReapOptions.hasFinished`) rather than resolved\n * here, exactly like `ReapOptions.reclaim`:\n *\n * - It cannot read `durability.snapshot()`. After a detach nothing appends to the\n * delivery log — the host that would have appended is the host that left — so\n * the log is frozen at the last delivered chunk while the JOURNAL keeps\n * growing. The log can only ever say \"no news\".\n * - It cannot resolve a `SandboxHandle` either. `SandboxInstanceStore` is\n * `get`/`upsert`/`delete` with no `list` (see `reclaim.ts` for why that is\n * deliberate), and only the application maps a `sandboxKey` to a live handle.\n *\n * NEVER REJECTS. This runs from a cron, an `alarm()`, or a `waitUntil` with\n * nobody to catch it, so every per-run failure is logged and folded into\n * {@link ReapResult} rather than escaping.\n *\n * NEVER CLEARS `detachedSince`. That field is what the reaper SELECTS on, and\n * `packages/ai/src/stream-to-response.ts`'s `startRunDriver` clears it because a\n * real viewer stopping the TTL clock is the opposite job. Its comment there names\n * borrowing that path \"the single most likely bug in this phase\"; clearing the\n * marker would reset the TTL on every sweep and a detached run would never\n * expire.\n */\nimport { isTerminalRunStatus, requestRunCancel } from '@tanstack/ai'\nimport {\n DEFAULT_FENCE_QUIET_MS,\n RunClaimLostError,\n // Thrown, not merely caught: the expiry re-derivation under the lock refuses\n // its own claim when the run's viewer has come back.\n RunClaimNotAcquiredError,\n awaitLogQuiescence,\n fenceDurability,\n fenceRunStore,\n withRunClaim,\n} from './claim'\nimport { pipeToRunLog } from './run'\nimport {\n journalExitProbeCommand,\n journalPaths,\n parseJournalExit,\n} from './journal'\nimport { decodeBase64Stream } from './journal-bytes'\nimport type { SandboxHandle } from './contracts'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { LockStore } from '@tanstack/ai/locks'\nimport type {\n RunRecord,\n RunStatus,\n RunStore,\n StreamChunk,\n StreamDurability,\n} from '@tanstack/ai'\n\n/**\n * Safety net for a single run's drive. Not the mechanism that decides whether a\n * run finished — see the module doc for why that design was rejected — so this is\n * generous rather than tight: it only has to stop a drive that has genuinely\n * wedged on a run the journal already said was over.\n *\n * On the expiry path it is not merely a net: it is what stops a still-producing\n * agent, since nothing polls the cancel recorded before that drive. A caller that\n * expires live agents may want a tighter value there than a finalization replay\n * needs.\n */\nexport const DEFAULT_RUN_BUDGET_MS = 30_000\n\n/**\n * Runs one sweep will touch. A cron invocation is bounded (a Worker's CPU\n * budget, a Lambda timeout), and an unbounded sweep over a backlog of thousands\n * would be killed mid-run rather than finishing 25 and returning; the next tick\n * takes the next batch.\n */\nexport const DEFAULT_MAX_RUNS = 25\n\n/** Journal tail bytes {@link probeRunExit} reads. The sentinel is the last line. */\nexport const DEFAULT_EXIT_PROBE_BYTES = 4096\n\n/**\n * What the out-of-band probe learned about a detached run's agent.\n *\n * THREE ARMS, not a boolean, because \"could not tell\" must not be\n * indistinguishable from \"still working\": both leave the run alone, but only one\n * of them is a condition an operator should see. A two-valued probe would also\n * invite the caller to treat a provider `exec` failure as \"finished\" and drive a\n * live run — the exact defect this module exists to prevent.\n */\nexport type RunExitProbe =\n /** The `{\"__exit\":N}` sentinel is in the journal. The agent is over. */\n | { state: 'finished'; exitCode: number }\n /** No sentinel. The agent is mid-flight (or never started). LEAVE IT ALONE. */\n | { state: 'producing' }\n /** The probe could not answer — no sandbox, `exec` rejected, frame undecodable. */\n | { state: 'unknown'; error?: unknown }\n\n/** What one sweep did to one run. */\nexport type ReapRunOutcome =\n /**\n * The probe saw `{\"__exit\":N}`, the run was driven to a terminal status, and\n * its transcript is saved. The happy path.\n */\n | 'finalized'\n /**\n * Past `detachedRunTtlMs`. Cancelled first, then driven to terminal. The probe\n * is skipped: the outcome is terminal whether the agent finished or not.\n *\n * Reported even when {@link ReapOptions.runBudgetMs} is what ended the drive —\n * on this path that is the mechanism rather than an anomaly, so `'expired'` is\n * the truthful outcome. The run's own `status` distinguishes the two shapes:\n * an agent that had already finished replays to `'completed'`, while one still\n * producing when the budget fired is `'aborted'`.\n */\n | 'expired'\n /**\n * Still producing. `pipeToRunLog` was NEVER entered — nothing appended, no\n * terminal record written, `close()` not called, `detachedSince` untouched.\n */\n | 'producing'\n /** The probe could not answer. Left exactly as untouched as `'producing'`. */\n | 'unknown'\n /**\n * ANOMALY. The drive outran {@link ReapOptions.runBudgetMs} on a run the\n * journal already said was finished. The record IS terminal and the log IS\n * closed (`pipeToRunLog` guarantees both), so this is a diagnostic, not a leak\n * — but a finished run that would not replay in 30s means the journal read, the\n * translation, or the log is misbehaving.\n *\n * FINALIZATION ONLY. An expired run that outran its budget reports `'expired'`:\n * there was no probe on that path and the agent may legitimately still have been\n * producing, so the budget firing is the designed stop, not a misbehaving replay.\n */\n | 'budget-exceeded'\n /**\n * Another host holds the claim, or held it and superseded us mid-drive. Normal:\n * a real viewer attaching mid-sweep is exactly this. Also covers a run that\n * reached terminal in another host's hands between the listing and the claim.\n */\n | 'not-claimed'\n /**\n * The transcript IS saved and the record IS terminal — only\n * {@link ReapOptions.reclaim} threw, so the sandbox is still up.\n *\n * A DISTINCT outcome rather than `'failed'`, because the two need opposite\n * operator responses and `'failed'` cannot express this one: it carries no\n * `status` and no `exitCode`, so \"transcript saved, sandbox NOT reclaimed\"\n * read identically to \"the sweep failed and the run was never finalized\".\n *\n * NOT RETRYABLE BY THE SWEEP. The record is terminal by now, so the run has\n * left `listReclaimable` for good; the sandbox leaks until something else\n * tears it down. This entry, with its `error`, is the only notice of that.\n *\n * OVERWRITES `'budget-exceeded'` when both happened, because the leak is what\n * needs acting on — {@link ReapRunEntry.terminalizedAnyway} is what preserves\n * the budget half of that pair.\n *\n * `sandboxReclaimer` REJECTS on its `'destroy-failed'` arm precisely so this\n * outcome is reachable through the shipped reclaimer and not only through a\n * custom one; see `SandboxReclaimFailedError` in `reclaim.ts`.\n */\n | 'reclaim-failed'\n /** Something threw. Logged, recorded here, and the sweep continued. */\n | 'failed'\n\n/** One run's line in the sweep summary. */\nexport interface ReapRunEntry {\n runId: string\n outcome: ReapRunOutcome\n /** The run's status after the sweep, when the run was driven. */\n status?: RunStatus\n /** The agent's exit code, when the probe read one. */\n exitCode?: number\n /**\n * THE BUDGET ANOMALY MARKER, and the only field whose mere PRESENCE carries a\n * fact: it is set if and only if the drive outran\n * {@link ReapOptions.runBudgetMs} on the finalization path — the condition\n * `'budget-exceeded'` names. Its value is whether the record nonetheless\n * reached a terminal status, practically always `true` since `pipeToRunLog` is\n * total; it is reported rather than assumed so an operator does not have to\n * infer it.\n *\n * SURVIVES A FAILED RECLAIM. `reclaim` runs after the outcome is classified\n * and overwrites it with `'reclaim-failed'`, which is the more urgent fact (a\n * leaked sandbox nothing will retry) and so wins the single `outcome` slot.\n * This field is therefore what keeps the budget anomaly on the entry: an\n * operator seeing `'reclaim-failed'` WITH `terminalizedAnyway` present is\n * looking at a run that blew its budget and then leaked, and needs both halves.\n */\n terminalizedAnyway?: boolean\n error?: unknown\n}\n\nexport interface ReapResult {\n /** Runs in this batch — i.e. after the {@link ReapOptions.maxRuns} cap. */\n considered: number\n /** Runs {@link ReapOptions.hasFinished} was actually called for. */\n probed: number\n outcomes: Record<ReapRunOutcome, number>\n runs: Array<ReapRunEntry>\n}\n\nexport interface ReapOptions<TOffset extends string = string> {\n runs: RunStore\n locks: LockStore\n /**\n * Per-run event log factory, same shape `RunDeps.durability` takes.\n *\n * Generic in the offset type, defaulted to `string` so an existing call site\n * needs no change — see {@link SandboxRunDriverOptions.durability} for why\n * hardcoding the default locked out branded-cursor backends.\n */\n durability: (runId: string) => StreamDurability<TOffset>\n /**\n * The out-of-band \"did the agent reach its sentinel?\" probe. INJECTED, because\n * neither the delivery log nor this package can answer it — see the module doc.\n * {@link probeRunExit} is the implementation an application wires in once it has\n * resolved the run's `SandboxHandle`.\n */\n hasFinished: (record: RunRecord) => Promise<RunExitProbe>\n /** Produce the run's remaining events. Called only once the claim is held. */\n drive: (input: {\n runId: string\n threadId: string\n signal: AbortSignal\n }) => AsyncIterable<StreamChunk>\n /** Sweep clock, passed rather than read so a sweep is reproducible. */\n now: number\n /** Detached-run TTL; `detachedSince <= now - ttl` expires, INCLUSIVELY. */\n detachedRunTtlMs: number\n /** Safety net per drive. Defaults to {@link DEFAULT_RUN_BUDGET_MS}. */\n runBudgetMs?: number\n /** Batch cap. Defaults to {@link DEFAULT_MAX_RUNS}. */\n maxRuns?: number\n /** Quiescence window; defaults to `DEFAULT_FENCE_QUIET_MS`. */\n fenceQuietMs?: number\n /**\n * Tear the run's sandbox down. Called ONLY after the run reached a terminal\n * status, and with the ORIGINALLY LISTED record — see {@link reapDetachedRuns}.\n * `sandboxReclaimer` in `reclaim.ts` is the ready-made implementation.\n */\n reclaim?: (record: RunRecord) => Promise<void>\n logger?: InternalLogger\n}\n\nasync function* singleValue(value: string): AsyncIterable<string> {\n yield value\n}\n\n/** Decode the base64 frame `journalExitProbeCommand` emits. */\nasync function decodeFrame(stdout: string): Promise<string> {\n const decoder = new TextDecoder()\n let text = ''\n for await (const bytes of decodeBase64Stream(singleValue(stdout))) {\n text += decoder.decode(bytes, { stream: true })\n }\n return text + decoder.decode()\n}\n\n/**\n * Read the END of a run's journal and answer whether the agent reached its\n * `{\"__exit\":N}` sentinel. Read-only: no append, no record write, no `close()`.\n *\n * This is the whole reason the reaper is safe. It is the ONLY way to learn that a\n * detached run is over without driving it, because the delivery log stops growing\n * the moment the viewer leaves while the journal does not.\n *\n * ANY failure answers `'unknown'`, never `'finished'`: the caller drives a run it\n * is told finished, so a provider `exec` that rejected, a sandbox that is gone, or\n * a frame the provider truncated must never be read as \"the agent exited\".\n *\n * An EMPTY tail answers `'producing'` — the fail-safe direction. A journal that\n * does not exist yet is indistinguishable here from one with no sentinel, and both\n * mean \"do not touch this run\".\n */\nexport async function probeRunExit(input: {\n handle: SandboxHandle\n runId: string\n /** Journal directory; defaults to `DEFAULT_JOURNAL_DIR`, as `journalPaths` does. */\n dir?: string\n /** Tail bytes to read. Defaults to {@link DEFAULT_EXIT_PROBE_BYTES}. */\n maxBytes?: number\n}): Promise<RunExitProbe> {\n try {\n const paths = journalPaths(input.runId, input.dir)\n const result = await input.handle.process.exec(\n journalExitProbeCommand(\n paths,\n input.maxBytes ?? DEFAULT_EXIT_PROBE_BYTES,\n ),\n )\n // `paths` supplies the per-run sentinel nonce: without it a mid-flight\n // agent that printed any JSON object carrying `__exit` would read as\n // `'finished'` here, and the caller would drive and reclaim a LIVE run.\n const exitCode = parseJournalExit(await decodeFrame(result.stdout), paths)\n return exitCode === null\n ? { state: 'producing' }\n : { state: 'finished', exitCode }\n } catch (error) {\n return { state: 'unknown', error }\n }\n}\n\n/** Every outcome key present at zero, so a consumer can read any of them. */\nfunction emptyOutcomes(): Record<ReapRunOutcome, number> {\n return {\n finalized: 0,\n expired: 0,\n producing: 0,\n unknown: 0,\n 'budget-exceeded': 0,\n 'not-claimed': 0,\n 'reclaim-failed': 0,\n failed: 0,\n }\n}\n\n/**\n * Report through a consumer-supplied logger without letting it break the sweep.\n * Mirrors `run.ts`'s `safeLog`: this module's totality must not be defeated by a\n * sink that cannot serialize a thrown value.\n */\nfunction safeLog(\n logger: InternalLogger | undefined,\n level: 'errors' | 'sandbox',\n message: string,\n context: Record<string, unknown>,\n): void {\n try {\n if (level === 'errors') logger?.errors(message, context)\n else logger?.sandbox(message, context)\n } catch {\n // Intentionally empty: there is no second channel to report on.\n }\n}\n\n/** Resolved-once settings shared by every run in one sweep. */\ninterface ReapContext<TOffset extends string = string> {\n options: ReapOptions<TOffset>\n runBudgetMs: number\n fenceQuietMs: number\n /** Inclusive expiry cutoff: `detachedSince <= cutoff` is expired. */\n cutoff: number\n}\n\n/** Whether a thrown value means \"we do not own this run\", which is normal. */\nfunction isClaimRefusal(error: unknown): boolean {\n return (\n error instanceof RunClaimNotAcquiredError ||\n error instanceof RunClaimLostError\n )\n}\n\n/**\n * Sweep ONE run. Never rejects: the caller folds the returned entry into the\n * summary and moves on.\n *\n * The ORDER of the steps below is the contract, not an implementation detail:\n *\n * 1. **Classify expiry first**, because an expired run needs no probe — its\n * outcome is terminal whether or not the agent finished, so a probe would only\n * add a provider round-trip and a way to fail.\n * 2. **Otherwise probe BEFORE touching anything.** `'producing'` and `'unknown'`\n * return here, having made no claim, no append, no record write, and no\n * `close()`. Driving past this point is the whole defect described in the\n * module doc.\n * 3. Claim, so two hosts never drive one run.\n * 4. **Re-derive expiry from a record read INSIDE the lock**, and only then\n * record the cancel. The listed record is stale by the time the claim is\n * held, and the cancel is sticky.\n * 5. Quiesce, so a predecessor still writing is observed rather than raced.\n * 6. **Arm the run budget**, so it bounds the drive rather than the queue the\n * two steps above stood in.\n * 7. Pipe with BOTH authoritative seams fenced, mirroring `driver.ts`.\n * 8. Reclaim, and ONLY once the record actually reached terminal.\n */\nasync function reapOne<TOffset extends string>(\n record: RunRecord,\n ctx: ReapContext<TOffset>,\n counters: { probed: number },\n): Promise<ReapRunEntry> {\n const { runs, locks, logger } = ctx.options\n const { runId, threadId } = record\n\n try {\n // INCLUSIVE, exactly as `RunStore.listReclaimable` documents its own cutoff:\n // a run detached at precisely `now - ttlMs` IS expired. The two must agree,\n // or a run would be listed as reclaimable and then classified as fresh on\n // every single sweep, forever.\n const expired =\n record.detachedSince !== undefined && record.detachedSince <= ctx.cutoff\n\n let exitCode: number | undefined\n if (!expired) {\n counters.probed += 1\n const probe = await ctx.options.hasFinished(record)\n if (probe.state !== 'finished') {\n // THE LEAVE-ALONE PATH. Deliberately returns before `withRunClaim`, so\n // not even `driverEpoch` moves — and above all `detachedSince` is left\n // exactly as it was, since it is both this run's TTL evidence and the\n // field the next sweep selects on.\n safeLog(logger, 'sandbox', `reap: leaving run ${runId} alone`, {\n runId,\n state: probe.state,\n ...(probe.state === 'unknown' && probe.error !== undefined\n ? { error: probe.error }\n : {}),\n })\n return {\n runId,\n outcome: probe.state,\n ...(probe.state === 'unknown' && probe.error !== undefined\n ? { error: probe.error }\n : {}),\n }\n }\n exitCode = probe.exitCode\n }\n\n // Armed INSIDE the claim, below. Read after it for the outcome, so it is\n // hoisted here rather than declared in the callback.\n let budget: AbortSignal | undefined\n const final = await withRunClaim(\n {\n runs,\n locks,\n runId,\n fenceQuietMs: ctx.fenceQuietMs,\n ...(logger === undefined ? {} : { logger }),\n },\n async (claim) => {\n if (expired) {\n // RE-DERIVED FROM A RECORD READ INSIDE THE LOCK, never from the listed\n // one. `stream-to-response.ts`'s `startRunDriver` CLEARS\n // `detachedSince` when a real viewer attaches — deliberately stopping\n // the TTL clock — and it takes this same per-run lock, so an\n // expiry decided at listing time is stale by the time the claim is\n // held. Cancelling on the stale value poisoned a now-live run:\n // nothing in the tree ever clears `cancelRequested`, so on that\n // viewer's next ORDINARY disconnect `middleware.ts`'s\n // `wasCancelRequested` read skips the detach branch and destroys the\n // sandbox of a healthy, actively-viewed run.\n const current = await runs.get(runId)\n if (current === null) {\n throw new RunClaimNotAcquiredError(runId, 'unknown')\n }\n if (\n current.detachedSince === undefined ||\n current.detachedSince > ctx.cutoff\n ) {\n // The viewer came back. `'not-claimed'` already documents \"a real\n // viewer attaching mid-sweep is exactly this\", and refusing here\n // leaves the run as untouched as the leave-alone path does: no\n // cancel, no append, no terminal record, no `close()`.\n throw new RunClaimNotAcquiredError(runId, 'superseded')\n }\n // BEFORE the drive, never after — and never before the claim.\n // `withSandbox`'s `onAbort` resolves the out-of-band cancel band from\n // the record, so recording the intent first is what makes the teardown\n // an explicit cancel that DESTROYS the sandbox rather than a second\n // detach that re-arms `detachedSince` and leaves the run to be swept\n // again forever. Recorded after the drive it is pure bookkeeping on a\n // run that already tore down the wrong way. Recorded before the CLAIM\n // it is an unfenced, sticky write on a record this host does not own,\n // derived from a value the lock exists to make current.\n await requestRunCancel(runs, runId)\n }\n // Before the first append, never after: `pipeToRunLog` snapshots to align.\n await awaitLogQuiescence(\n ctx.options.durability(runId),\n ctx.fenceQuietMs,\n )\n // A safety net, not a mechanism (see the module doc). `AbortSignal.any`\n // is this package's idiom for linking one — see\n // `testkit/takeover-conformance.ts`.\n //\n // ARMED HERE, not before `withRunClaim`. Both the lock wait and the\n // quiescence wait consume a timer started earlier: quiescence always\n // sleeps at least one `fenceQuietMs` and may sleep six, and lock\n // acquisition waits behind whoever holds it, unbounded. The effective\n // budget was silently `runBudgetMs − fenceQuietMs − lockWait`, and once\n // it went negative the claim was acquired with the timer already fired:\n // `pipeToRunLog` hit its entry `signal.aborted` check before pulling one\n // chunk, so a FINISHED agent's transcript was recorded `'aborted'` and\n // its log closed — and a terminal record leaves `listReclaimable`\n // forever, so that transcript was then unreplayable while `reclaim`\n // destroyed the sandbox holding the only copy. The budget bounds the\n // DRIVE, not the queue.\n budget = AbortSignal.timeout(ctx.runBudgetMs)\n // `claim.signal` is in the composed signal because losing the lease MUST\n // stop the drive: a successor that took the run over is appending to the\n // same log, and this drive continuing would double every chunk.\n const signal = AbortSignal.any([claim.signal, budget])\n return pipeToRunLog(ctx.options.drive({ runId, threadId, signal }), {\n // BOTH seams, over the SAME claim, as `driver.ts` explains: fencing the\n // log alone just moves the harm to \"a dead host marks the successor's\n // live run failed\".\n runs: fenceRunStore(runs, claim, {\n ...(logger === undefined ? {} : { logger }),\n }),\n durability: (id) =>\n fenceDurability(ctx.options.durability(id), claim, { runs }),\n runId,\n threadId,\n signal,\n ...(logger === undefined ? {} : { logger }),\n })\n },\n )\n\n const terminal = isTerminalRunStatus(final.status)\n let outcome: ReapRunOutcome\n // `&& !expired` is the whole subtlety. The budget is an ANOMALY only on the\n // finalization path, where the probe already said the agent hit its sentinel\n // and a replay that will not finish in 30s means the journal read, the\n // translation, or the log is misbehaving. On the EXPIRY path there was no\n // probe and the agent may well be mid-sentence: `requestRunCancel` writes a\n // record field whose only reader is `withSandbox`'s `onAbort` (which runs\n // after something else has already aborted), so the budget is the sole thing\n // that ends the drive of a still-producing expired run. That is the designed\n // path, not a misbehaving one, and reporting it as the anomaly made\n // `'expired'` unreachable for exactly the runs the TTL exists to expire.\n // `budget` is armed inside the claim, so reaching here means it was armed;\n // `?? false` keeps the read total rather than asserting that.\n if ((budget?.aborted ?? false) && !expired) {\n outcome = 'budget-exceeded'\n } else if (!terminal) {\n // The terminal write was SUPPRESSED and `finish`'s re-read answered with a\n // live record, which `fenceRunStore` only does when this host lost the claim\n // to another one. That is the same fact as a refused claim, reported the\n // same way rather than as a success that wrote nothing.\n outcome = 'not-claimed'\n } else {\n outcome = expired ? 'expired' : 'finalized'\n }\n\n // CAPTURED BEFORE THE RECLAIM BLOCK, which may overwrite `outcome` with\n // `'reclaim-failed'`. Conditioning the `terminalizedAnyway` spread on the\n // post-reclaim `outcome` dropped the budget diagnostic from exactly the\n // entries that need it most: a run that blew its budget AND then failed to\n // reclaim reported neither fact but the leak, and an operator cannot\n // diagnose a leak on a run whose replay was already misbehaving without\n // knowing that it was.\n const budgetAnomaly = outcome === 'budget-exceeded'\n\n let reclaimError: unknown\n if (terminal && ctx.options.reclaim !== undefined) {\n try {\n // `record`, NOT `final`. When the terminal `update` fails, `finish` returns\n // a LOCALLY REBUILT record that carries only `runId`/`threadId`/`startedAt`\n // plus the terminal patch — no `sandboxKey` — so `reclaimSandbox` would see\n // `undefined`, answer `'no-sandbox-key'`, and the sandbox would leak\n // silently on exactly the path where something already went wrong.\n await ctx.options.reclaim(record)\n } catch (error) {\n // CAUGHT HERE rather than in the outer catch, which would report a bare\n // `'failed'` with no `status` and no `exitCode`. `reclaimSandbox`\n // deliberately does not guard `instances.get` (its contract is that the\n // CALLER records the failure) and neither does `sandboxReclaimer`, so a\n // throwing instance store landed there. By this point the record is\n // terminal and the log closed, so the run is out of `listReclaimable`\n // forever and no later sweep will retry: the sandbox leaks, and an\n // operator reading `'failed'` cannot tell \"transcript saved, sandbox NOT\n // reclaimed\" from \"the sweep failed and the run was never finalized\".\n reclaimError = error\n outcome = 'reclaim-failed'\n safeLog(logger, 'errors', `reap: reclaiming run ${runId} failed`, {\n runId,\n status: final.status,\n error,\n })\n }\n }\n\n return {\n runId,\n outcome,\n status: final.status,\n ...(exitCode === undefined ? {} : { exitCode }),\n ...(budgetAnomaly ? { terminalizedAnyway: terminal } : {}),\n ...(reclaimError === undefined ? {} : { error: reclaimError }),\n }\n } catch (error) {\n if (isClaimRefusal(error)) {\n safeLog(logger, 'sandbox', `reap: not driving run ${runId}`, {\n runId,\n error,\n })\n return { runId, outcome: 'not-claimed', error }\n }\n // Folded into the summary rather than rethrown: one bad run must not abandon\n // the rest of the batch, and there is no caller to receive a rejection.\n safeLog(logger, 'errors', `reap: sweeping run ${runId} failed`, {\n runId,\n error,\n })\n return { runId, outcome: 'failed', error }\n }\n}\n\n/**\n * Sweep the detached runs a `RunStore` surfaces, saving each finished run's\n * transcript and reclaiming its sandbox.\n *\n * A plain async function with no timer and no daemon: call it from a cron, a\n * queue consumer, a Durable Object `alarm()`, or a `waitUntil`. It NEVER rejects\n * — every failure is logged and counted in the returned {@link ReapResult}.\n *\n * ONE `listReclaimable({ now, ttlMs: 0 })` call, deliberately: `ttlMs: 0` is\n * every detached run, which is the candidate set for FINALIZATION (a run that hit\n * its sentinel one second after the viewer left has an unsaved transcript and\n * must not wait out the TTL), and expiry is then classified in-process against\n * the same inclusive cutoff. Listing twice with two TTLs would cost a second\n * store round-trip to compute a subset.\n *\n * `listReclaimable` is OPTIONAL on `RunStore`. A backend without it cannot be\n * reaped, which answers `{ considered: 0 }` plus one log line rather than\n * throwing — the same graceful degrade every other optional-method call site in\n * the repo does (`store.findActiveRun?.(threadId)`).\n */\nexport async function reapDetachedRuns<TOffset extends string = string>(\n options: ReapOptions<TOffset>,\n): Promise<ReapResult> {\n const logger = options.logger\n const outcomes = emptyOutcomes()\n const entries: Array<ReapRunEntry> = []\n const empty = (): ReapResult => ({\n considered: 0,\n probed: 0,\n outcomes,\n runs: entries,\n })\n\n const list = options.runs.listReclaimable?.bind(options.runs)\n if (list === undefined) {\n safeLog(\n logger,\n 'sandbox',\n 'reap: the run store does not implement listReclaimable; nothing to sweep',\n {},\n )\n return empty()\n }\n\n let candidates: Array<RunRecord>\n try {\n candidates = await list({ now: options.now, ttlMs: 0 })\n } catch (error) {\n safeLog(logger, 'errors', 'reap: listing reclaimable runs failed', {\n error,\n })\n return empty()\n }\n\n // Capped so one invocation cannot outlive its platform's budget and be killed\n // mid-drive. `slice` and not a `break`, so `considered` reports the batch the\n // sweep actually took responsibility for.\n const maxRuns = Math.max(0, Math.trunc(options.maxRuns ?? DEFAULT_MAX_RUNS))\n const batch = candidates.slice(0, maxRuns)\n\n const ctx: ReapContext<TOffset> = {\n options,\n runBudgetMs: options.runBudgetMs ?? DEFAULT_RUN_BUDGET_MS,\n fenceQuietMs: options.fenceQuietMs ?? DEFAULT_FENCE_QUIET_MS,\n cutoff: options.now - options.detachedRunTtlMs,\n }\n const counters = { probed: 0 }\n\n // Sequential on purpose: each run costs a lock, a provider round-trip, and a\n // full replay, and a cron invocation's budget is the scarce resource. Fanning\n // out would multiply peak load against the provider for no throughput a\n // subsequent tick cannot supply.\n for (const record of batch) {\n const entry = await reapOne(record, ctx, counters)\n outcomes[entry.outcome] += 1\n entries.push(entry)\n }\n\n return {\n considered: batch.length,\n probed: counters.probed,\n outcomes,\n runs: entries,\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAyGA,IAAa,wBAAwB;;;;;;;AAQrC,IAAa,mBAAmB;;AAGhC,IAAa,2BAA2B;AAuKxC,gBAAgB,YAAY,OAAsC;CAChE,MAAM;AACR;;AAGA,eAAe,YAAY,QAAiC;CAC1D,MAAM,UAAU,IAAI,YAAY;CAChC,IAAI,OAAO;CACX,WAAW,MAAM,SAAS,mBAAmB,YAAY,MAAM,CAAC,GAC9D,QAAQ,QAAQ,OAAO,OAAO,EAAE,QAAQ,KAAK,CAAC;CAEhD,OAAO,OAAO,QAAQ,OAAO;AAC/B;;;;;;;;;;;;;;;;;AAkBA,eAAsB,aAAa,OAOT;CACxB,IAAI;EACF,MAAM,QAAQ,aAAa,MAAM,OAAO,MAAM,GAAG;EAUjD,MAAM,WAAW,iBAAiB,MAAM,aAAY,MAT/B,MAAM,OAAO,QAAQ,KACxC,wBACE,OACA,MAAM,YAAA,IACR,CACF,EAAA,CAI2D,MAAM,GAAG,KAAK;EACzE,OAAO,aAAa,OAChB,EAAE,OAAO,YAAY,IACrB;GAAE,OAAO;GAAY;EAAS;CACpC,SAAS,OAAO;EACd,OAAO;GAAE,OAAO;GAAW;EAAM;CACnC;AACF;;AAGA,SAAS,gBAAgD;CACvD,OAAO;EACL,WAAW;EACX,SAAS;EACT,WAAW;EACX,SAAS;EACT,mBAAmB;EACnB,eAAe;EACf,kBAAkB;EAClB,QAAQ;CACV;AACF;;;;;;AAOA,SAAS,QACP,QACA,OACA,SACA,SACM;CACN,IAAI;EACF,IAAI,UAAU,UAAU,QAAQ,OAAO,SAAS,OAAO;OAClD,QAAQ,QAAQ,SAAS,OAAO;CACvC,QAAQ,CAER;AACF;;AAYA,SAAS,eAAe,OAAyB;CAC/C,OACE,iBAAiB,4BACjB,iBAAiB;AAErB;;;;;;;;;;;;;;;;;;;;;;;;AAyBA,eAAe,QACb,QACA,KACA,UACuB;CACvB,MAAM,EAAE,MAAM,OAAO,WAAW,IAAI;CACpC,MAAM,EAAE,OAAO,aAAa;CAE5B,IAAI;EAKF,MAAM,UACJ,OAAO,kBAAkB,KAAA,KAAa,OAAO,iBAAiB,IAAI;EAEpE,IAAI;EACJ,IAAI,CAAC,SAAS;GACZ,SAAS,UAAU;GACnB,MAAM,QAAQ,MAAM,IAAI,QAAQ,YAAY,MAAM;GAClD,IAAI,MAAM,UAAU,YAAY;IAK9B,QAAQ,QAAQ,WAAW,qBAAqB,MAAM,SAAS;KAC7D;KACA,OAAO,MAAM;KACb,GAAI,MAAM,UAAU,aAAa,MAAM,UAAU,KAAA,IAC7C,EAAE,OAAO,MAAM,MAAM,IACrB,CAAC;IACP,CAAC;IACD,OAAO;KACL;KACA,SAAS,MAAM;KACf,GAAI,MAAM,UAAU,aAAa,MAAM,UAAU,KAAA,IAC7C,EAAE,OAAO,MAAM,MAAM,IACrB,CAAC;IACP;GACF;GACA,WAAW,MAAM;EACnB;EAIA,IAAI;EACJ,MAAM,QAAQ,MAAM,aAClB;GACE;GACA;GACA;GACA,cAAc,IAAI;GAClB,GAAI,WAAW,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO;EAC3C,GACA,OAAO,UAAU;GACf,IAAI,SAAS;IAWX,MAAM,UAAU,MAAM,KAAK,IAAI,KAAK;IACpC,IAAI,YAAY,MACd,MAAM,IAAI,yBAAyB,OAAO,SAAS;IAErD,IACE,QAAQ,kBAAkB,KAAA,KAC1B,QAAQ,gBAAgB,IAAI,QAM5B,MAAM,IAAI,yBAAyB,OAAO,YAAY;IAWxD,MAAM,iBAAiB,MAAM,KAAK;GACpC;GAEA,MAAM,mBACJ,IAAI,QAAQ,WAAW,KAAK,GAC5B,IAAI,YACN;GAiBA,SAAS,YAAY,QAAQ,IAAI,WAAW;GAI5C,MAAM,SAAS,YAAY,IAAI,CAAC,MAAM,QAAQ,MAAM,CAAC;GACrD,OAAO,aAAa,IAAI,QAAQ,MAAM;IAAE;IAAO;IAAU;GAAO,CAAC,GAAG;IAIlE,MAAM,cAAc,MAAM,OAAO,EAC/B,GAAI,WAAW,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,EAC3C,CAAC;IACD,aAAa,OACX,gBAAgB,IAAI,QAAQ,WAAW,EAAE,GAAG,OAAO,EAAE,KAAK,CAAC;IAC7D;IACA;IACA;IACA,GAAI,WAAW,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO;GAC3C,CAAC;EACH,CACF;EAEA,MAAM,WAAW,oBAAoB,MAAM,MAAM;EACjD,IAAI;EAaJ,KAAK,QAAQ,WAAW,UAAU,CAAC,SACjC,UAAU;OACL,IAAI,CAAC,UAKV,UAAU;OAEV,UAAU,UAAU,YAAY;EAUlC,MAAM,gBAAgB,YAAY;EAElC,IAAI;EACJ,IAAI,YAAY,IAAI,QAAQ,YAAY,KAAA,GACtC,IAAI;GAMF,MAAM,IAAI,QAAQ,QAAQ,MAAM;EAClC,SAAS,OAAO;GAUd,eAAe;GACf,UAAU;GACV,QAAQ,QAAQ,UAAU,wBAAwB,MAAM,UAAU;IAChE;IACA,QAAQ,MAAM;IACd;GACF,CAAC;EACH;EAGF,OAAO;GACL;GACA;GACA,QAAQ,MAAM;GACd,GAAI,aAAa,KAAA,IAAY,CAAC,IAAI,EAAE,SAAS;GAC7C,GAAI,gBAAgB,EAAE,oBAAoB,SAAS,IAAI,CAAC;GACxD,GAAI,iBAAiB,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,aAAa;EAC9D;CACF,SAAS,OAAO;EACd,IAAI,eAAe,KAAK,GAAG;GACzB,QAAQ,QAAQ,WAAW,yBAAyB,SAAS;IAC3D;IACA;GACF,CAAC;GACD,OAAO;IAAE;IAAO,SAAS;IAAe;GAAM;EAChD;EAGA,QAAQ,QAAQ,UAAU,sBAAsB,MAAM,UAAU;GAC9D;GACA;EACF,CAAC;EACD,OAAO;GAAE;GAAO,SAAS;GAAU;EAAM;CAC3C;AACF;;;;;;;;;;;;;;;;;;;;;AAsBA,eAAsB,iBACpB,SACqB;CACrB,MAAM,SAAS,QAAQ;CACvB,MAAM,WAAW,cAAc;CAC/B,MAAM,UAA+B,CAAC;CACtC,MAAM,eAA2B;EAC/B,YAAY;EACZ,QAAQ;EACR;EACA,MAAM;CACR;CAEA,MAAM,OAAO,QAAQ,KAAK,iBAAiB,KAAK,QAAQ,IAAI;CAC5D,IAAI,SAAS,KAAA,GAAW;EACtB,QACE,QACA,WACA,4EACA,CAAC,CACH;EACA,OAAO,MAAM;CACf;CAEA,IAAI;CACJ,IAAI;EACF,aAAa,MAAM,KAAK;GAAE,KAAK,QAAQ;GAAK,OAAO;EAAE,CAAC;CACxD,SAAS,OAAO;EACd,QAAQ,QAAQ,UAAU,yCAAyC,EACjE,MACF,CAAC;EACD,OAAO,MAAM;CACf;CAKA,MAAM,UAAU,KAAK,IAAI,GAAG,KAAK,MAAM,QAAQ,WAAA,EAA2B,CAAC;CAC3E,MAAM,QAAQ,WAAW,MAAM,GAAG,OAAO;CAEzC,MAAM,MAA4B;EAChC;EACA,aAAa,QAAQ,eAAA;EACrB,cAAc,QAAQ,gBAAA;EACtB,QAAQ,QAAQ,MAAM,QAAQ;CAChC;CACA,MAAM,WAAW,EAAE,QAAQ,EAAE;CAM7B,KAAK,MAAM,UAAU,OAAO;EAC1B,MAAM,QAAQ,MAAM,QAAQ,QAAQ,KAAK,QAAQ;EACjD,SAAS,MAAM,YAAY;EAC3B,QAAQ,KAAK,KAAK;CACpB;CAEA,OAAO;EACL,YAAY,MAAM;EAClB,QAAQ,SAAS;EACjB;EACA,MAAM;CACR;AACF"}
|
|
1
|
+
{"version":3,"file":"reap.js","names":[],"sources":["../../src/reap.ts"],"sourcesContent":["/**\n * The sweep `RunStore.listReclaimable` was always missing a consumer for: take a\n * detached run whose viewer never came back, save its transcript, terminalize its\n * record, and tear its sandbox down.\n *\n * THE ONE RULE THAT SHAPES EVERYTHING HERE: **never drive a run to find out\n * whether it finished.**\n *\n * The obvious design — hand the run to `pipeToRunLog` under a short\n * `runBudgetMs` and see whether it terminalizes — was measured and is broken.\n * `pipeToRunLog` is total by construction: it ALWAYS writes a terminal status and\n * ALWAYS calls `durability.close()`. Against a run that has not finished, all\n * three producer shapes are destructive:\n *\n * | producer's reaction to the budget signal | stored status | `close()` |\n * | ---------------------------------------- | ------------- | --------- |\n * | ignores it and keeps producing | `aborted` | called |\n * | returns on abort (the realistic `drive`) | `aborted` | called |\n * | throws an AbortError | `failed` | called |\n *\n * The middle row USED to read `completed`, which was the fatal one: a signal-aware\n * producer exits its loop NORMALLY, and `pipeToRunLog` only checked its signal\n * per chunk, so a healthy mid-flight run was recorded as `'completed'` with a\n * `finishedAt` — a false transcript. That gap is fixed (`run.ts` re-checks the\n * signal after the loop), so the status is now honest on all three rows. The rule\n * above is UNCHANGED, because the status was never the whole harm: every row\n * writes a terminal record and closes a log that commit `5a1f821c9` deliberately\n * leaves OPEN for takeover (ending every attached client's stream), and a terminal\n * record drops out of `listReclaimable` forever, so TTL expiry can never reclaim\n * that run's sandbox. A cost leak with no recovery path. There is therefore no\n * \"still running\" outcome in {@link ReapRunOutcome}: it is unreachable by\n * construction, not merely unlikely.\n *\n * So sentinel-reached is detected OUT OF BAND, through the in-sandbox journal\n * ({@link probeRunExit}), and `pipeToRunLog` is entered only for a run already\n * KNOWN to have finished, or for one whose TTL has expired (terminal either way).\n * On the FINALIZATION path `runBudgetMs` therefore degrades from a load-bearing\n * mechanism into a safety net whose expiry is a genuine anomaly — see\n * `'budget-exceeded'`. On the EXPIRY path it stays load-bearing: nothing polls the\n * cancel this module records, so the budget is what ends the drive of an expired\n * run whose agent is still producing, and its expiry there is the designed path.\n *\n * WHY THE PROBE IS INJECTED (`ReapOptions.hasFinished`) rather than resolved\n * here, exactly like `ReapOptions.reclaim`:\n *\n * - It cannot read `durability.snapshot()`. After a detach nothing appends to the\n * delivery log — the host that would have appended is the host that left — so\n * the log is frozen at the last delivered chunk while the JOURNAL keeps\n * growing. The log can only ever say \"no news\".\n * - It cannot resolve a `SandboxHandle` either. `SandboxInstanceStore` is\n * `get`/`upsert`/`delete` with no `list` (see `reclaim.ts` for why that is\n * deliberate), and only the application maps a `sandboxKey` to a live handle.\n *\n * NEVER REJECTS. This runs from a cron, an `alarm()`, or a `waitUntil` with\n * nobody to catch it, so every per-run failure is logged and folded into\n * {@link ReapResult} rather than escaping.\n *\n * NEVER CLEARS `detachedSince`. That field is what the reaper SELECTS on, and\n * `packages/ai/src/stream-to-response.ts`'s `startRunDriver` clears it because a\n * real viewer stopping the TTL clock is the opposite job. Its comment there names\n * borrowing that path \"the single most likely bug in this phase\"; clearing the\n * marker would reset the TTL on every sweep and a detached run would never\n * expire.\n */\nimport { isTerminalRunStatus, requestRunCancel } from '@tanstack/ai'\nimport {\n DEFAULT_FENCE_QUIET_MS,\n RunClaimLostError,\n // Thrown, not merely caught: the expiry re-derivation under the lock refuses\n // its own claim when the run's viewer has come back.\n RunClaimNotAcquiredError,\n awaitLogQuiescence,\n fenceDurability,\n fenceRunStore,\n withRunClaim,\n} from './claim'\nimport { pipeToRunLog } from './run'\nimport {\n journalExitProbeCommand,\n journalPaths,\n parseJournalExit,\n} from './journal'\nimport { decodeBase64Stream } from './journal-bytes'\nimport type { SandboxHandle } from './contracts'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { LockStore } from '@tanstack/ai/locks'\nimport type {\n RunRecord,\n RunStatus,\n RunStore,\n StreamChunk,\n StreamDurability,\n} from '@tanstack/ai'\n\n/**\n * Safety net for a single run's drive. Not the mechanism that decides whether a\n * run finished — see the module doc for why that design was rejected — so this is\n * generous rather than tight: it only has to stop a drive that has genuinely\n * wedged on a run the journal already said was over.\n *\n * On the expiry path it is not merely a net: it is what stops a still-producing\n * agent, since nothing polls the cancel recorded before that drive. A caller that\n * expires live agents may want a tighter value there than a finalization replay\n * needs.\n */\nexport const DEFAULT_RUN_BUDGET_MS = 30_000\n\n/**\n * Runs one sweep will touch. A cron invocation is bounded (a Worker's CPU\n * budget, a Lambda timeout), and an unbounded sweep over a backlog of thousands\n * would be killed mid-run rather than finishing 25 and returning; the next tick\n * takes the next batch.\n */\nexport const DEFAULT_MAX_RUNS = 25\n\n/** Journal tail bytes {@link probeRunExit} reads. The sentinel is the last line. */\nexport const DEFAULT_EXIT_PROBE_BYTES = 4096\n\n/**\n * What the out-of-band probe learned about a detached run's agent.\n *\n * THREE ARMS, not a boolean, because \"could not tell\" must not be\n * indistinguishable from \"still working\": both leave the run alone, but only one\n * of them is a condition an operator should see. A two-valued probe would also\n * invite the caller to treat a provider `exec` failure as \"finished\" and drive a\n * live run — the exact defect this module exists to prevent.\n */\nexport type RunExitProbe =\n /** The `{\"__exit\":N}` sentinel is in the journal. The agent is over. */\n | { state: 'finished'; exitCode: number }\n /** No sentinel. The agent is mid-flight (or never started). LEAVE IT ALONE. */\n | { state: 'producing' }\n /** The probe could not answer — no sandbox, `exec` rejected, frame undecodable. */\n | { state: 'unknown'; error?: unknown }\n\n/** What one sweep did to one run. */\nexport type ReapRunOutcome =\n /**\n * The probe saw `{\"__exit\":N}`, the run was driven to a terminal status, and\n * its transcript is saved. The happy path.\n */\n | 'finalized'\n /**\n * Past `detachedRunTtlMs`. Cancelled first, then driven to terminal. The probe\n * is skipped: the outcome is terminal whether the agent finished or not.\n *\n * Reported even when {@link ReapOptions.runBudgetMs} is what ended the drive —\n * on this path that is the mechanism rather than an anomaly, so `'expired'` is\n * the truthful outcome. The run's own `status` distinguishes the two shapes:\n * an agent that had already finished replays to `'completed'`, while one still\n * producing when the budget fired is `'aborted'`.\n */\n | 'expired'\n /**\n * Still producing. `pipeToRunLog` was NEVER entered — nothing appended, no\n * terminal record written, `close()` not called, `detachedSince` untouched.\n */\n | 'producing'\n /** The probe could not answer. Left exactly as untouched as `'producing'`. */\n | 'unknown'\n /**\n * ANOMALY. The drive outran {@link ReapOptions.runBudgetMs} on a run the\n * journal already said was finished. The record IS terminal and the log IS\n * closed (`pipeToRunLog` guarantees both), so this is a diagnostic, not a leak\n * — but a finished run that would not replay in 30s means the journal read, the\n * translation, or the log is misbehaving.\n *\n * FINALIZATION ONLY. An expired run that outran its budget reports `'expired'`:\n * there was no probe on that path and the agent may legitimately still have been\n * producing, so the budget firing is the designed stop, not a misbehaving replay.\n */\n | 'budget-exceeded'\n /**\n * Another host holds the claim, or held it and superseded us mid-drive. Normal:\n * a real viewer attaching mid-sweep is exactly this. Also covers a run that\n * reached terminal in another host's hands between the listing and the claim.\n */\n | 'not-claimed'\n /**\n * The transcript IS saved and the record IS terminal — only\n * {@link ReapOptions.reclaim} threw, so the sandbox is still up.\n *\n * A DISTINCT outcome rather than `'failed'`, because the two need opposite\n * operator responses and `'failed'` cannot express this one: it carries no\n * `status` and no `exitCode`, so \"transcript saved, sandbox NOT reclaimed\"\n * read identically to \"the sweep failed and the run was never finalized\".\n *\n * NOT RETRYABLE BY THE SWEEP. The record is terminal by now, so the run has\n * left `listReclaimable` for good; the sandbox leaks until something else\n * tears it down. This entry, with its `error`, is the only notice of that.\n *\n * OVERWRITES `'budget-exceeded'` when both happened, because the leak is what\n * needs acting on — {@link ReapRunEntry.terminalizedAnyway} is what preserves\n * the budget half of that pair.\n *\n * `sandboxReclaimer` REJECTS on its `'destroy-failed'` arm precisely so this\n * outcome is reachable through the shipped reclaimer and not only through a\n * custom one; see `SandboxReclaimFailedError` in `reclaim.ts`.\n */\n | 'reclaim-failed'\n /** Something threw. Logged, recorded here, and the sweep continued. */\n | 'failed'\n\n/** One run's line in the sweep summary. */\nexport interface ReapRunEntry {\n runId: string\n outcome: ReapRunOutcome\n /** The run's status after the sweep, when the run was driven. */\n status?: RunStatus\n /** The agent's exit code, when the probe read one. */\n exitCode?: number\n /**\n * THE BUDGET ANOMALY MARKER, and the only field whose mere PRESENCE carries a\n * fact: it is set if and only if the drive outran\n * {@link ReapOptions.runBudgetMs} on the finalization path — the condition\n * `'budget-exceeded'` names. Its value is whether the record nonetheless\n * reached a terminal status, practically always `true` since `pipeToRunLog` is\n * total; it is reported rather than assumed so an operator does not have to\n * infer it.\n *\n * SURVIVES A FAILED RECLAIM. `reclaim` runs after the outcome is classified\n * and overwrites it with `'reclaim-failed'`, which is the more urgent fact (a\n * leaked sandbox nothing will retry) and so wins the single `outcome` slot.\n * This field is therefore what keeps the budget anomaly on the entry: an\n * operator seeing `'reclaim-failed'` WITH `terminalizedAnyway` present is\n * looking at a run that blew its budget and then leaked, and needs both halves.\n */\n terminalizedAnyway?: boolean\n error?: unknown\n}\n\nexport interface ReapResult {\n /** Runs in this batch — i.e. after the {@link ReapOptions.maxRuns} cap. */\n considered: number\n /** Runs {@link ReapOptions.hasFinished} was actually called for. */\n probed: number\n outcomes: Record<ReapRunOutcome, number>\n runs: Array<ReapRunEntry>\n}\n\nexport interface ReapOptions<TOffset extends string = string> {\n runs: RunStore\n locks: LockStore\n /**\n * Per-run event log factory, same shape `RunDeps.durability` takes.\n *\n * Generic in the offset type, defaulted to `string` so an existing call site\n * needs no change — see {@link SandboxRunDriverOptions.durability} for why\n * hardcoding the default locked out branded-cursor backends.\n */\n durability: (runId: string) => StreamDurability<TOffset>\n /**\n * The out-of-band \"did the agent reach its sentinel?\" probe. INJECTED, because\n * neither the delivery log nor this package can answer it — see the module doc.\n * {@link probeRunExit} is the implementation an application wires in once it has\n * resolved the run's `SandboxHandle`.\n */\n hasFinished: (record: RunRecord) => Promise<RunExitProbe>\n /** Produce the run's remaining events. Called only once the claim is held. */\n drive: (input: {\n runId: string\n threadId: string\n signal: AbortSignal\n }) => AsyncIterable<StreamChunk>\n /** Sweep clock, passed rather than read so a sweep is reproducible. */\n now: number\n /** Detached-run TTL; `detachedSince <= now - ttl` expires, INCLUSIVELY. */\n detachedRunTtlMs: number\n /** Safety net per drive. Defaults to {@link DEFAULT_RUN_BUDGET_MS}. */\n runBudgetMs?: number\n /** Batch cap. Defaults to {@link DEFAULT_MAX_RUNS}. */\n maxRuns?: number\n /** Quiescence window; defaults to `DEFAULT_FENCE_QUIET_MS`. */\n fenceQuietMs?: number\n /**\n * Tear the run's sandbox down. Called ONLY after the run reached a terminal\n * status, and with the ORIGINALLY LISTED record — see {@link reapDetachedRuns}.\n * `sandboxReclaimer` in `reclaim.ts` is the ready-made implementation.\n */\n reclaim?: (record: RunRecord) => Promise<void>\n logger?: InternalLogger\n}\n\nasync function* singleValue(value: string): AsyncIterable<string> {\n yield value\n}\n\n/** Decode the base64 frame `journalExitProbeCommand` emits. */\nasync function decodeFrame(stdout: string): Promise<string> {\n const decoder = new TextDecoder()\n let text = ''\n for await (const bytes of decodeBase64Stream(singleValue(stdout))) {\n text += decoder.decode(bytes, { stream: true })\n }\n return text + decoder.decode()\n}\n\n/**\n * Read the END of a run's journal and answer whether the agent reached its\n * `{\"__exit\":N}` sentinel. Read-only: no append, no record write, no `close()`.\n *\n * This is the whole reason the reaper is safe. It is the ONLY way to learn that a\n * detached run is over without driving it, because the delivery log stops growing\n * the moment the viewer leaves while the journal does not.\n *\n * ANY failure answers `'unknown'`, never `'finished'`: the caller drives a run it\n * is told finished, so a provider `exec` that rejected, a sandbox that is gone, or\n * a frame the provider truncated must never be read as \"the agent exited\".\n *\n * An EMPTY tail answers `'producing'` — the fail-safe direction. A journal that\n * does not exist yet is indistinguishable here from one with no sentinel, and both\n * mean \"do not touch this run\".\n */\nexport async function probeRunExit(input: {\n handle: SandboxHandle\n runId: string\n /** Journal directory; defaults to `DEFAULT_JOURNAL_DIR`, as `journalPaths` does. */\n dir?: string\n /** Tail bytes to read. Defaults to {@link DEFAULT_EXIT_PROBE_BYTES}. */\n maxBytes?: number\n}): Promise<RunExitProbe> {\n try {\n const paths = journalPaths(input.runId, input.dir)\n const result = await input.handle.process.exec(\n journalExitProbeCommand(\n paths,\n input.maxBytes ?? DEFAULT_EXIT_PROBE_BYTES,\n ),\n )\n // `paths` supplies the per-run sentinel nonce: without it a mid-flight\n // agent that printed any JSON object carrying `__exit` would read as\n // `'finished'` here, and the caller would drive and reclaim a LIVE run.\n const exitCode = parseJournalExit(await decodeFrame(result.stdout), paths)\n return exitCode === null\n ? { state: 'producing' }\n : { state: 'finished', exitCode }\n } catch (error) {\n return { state: 'unknown', error }\n }\n}\n\n/** Every outcome key present at zero, so a consumer can read any of them. */\nfunction emptyOutcomes(): Record<ReapRunOutcome, number> {\n return {\n finalized: 0,\n expired: 0,\n producing: 0,\n unknown: 0,\n 'budget-exceeded': 0,\n 'not-claimed': 0,\n 'reclaim-failed': 0,\n failed: 0,\n }\n}\n\n/**\n * Report through a consumer-supplied logger without letting it break the sweep.\n * Mirrors `run.ts`'s `safeLog`: this module's totality must not be defeated by a\n * sink that cannot serialize a thrown value.\n */\nfunction safeLog(\n logger: InternalLogger | undefined,\n level: 'errors' | 'sandbox',\n message: string,\n context: Record<string, unknown>,\n): void {\n try {\n if (level === 'errors') logger?.errors(message, context)\n else logger?.sandbox(message, context)\n } catch {\n // Intentionally empty: there is no second channel to report on.\n }\n}\n\n/** Resolved-once settings shared by every run in one sweep. */\ninterface ReapContext<TOffset extends string = string> {\n options: ReapOptions<TOffset>\n runBudgetMs: number\n fenceQuietMs: number\n /** Inclusive expiry cutoff: `detachedSince <= cutoff` is expired. */\n cutoff: number\n}\n\n/** Whether a thrown value means \"we do not own this run\", which is normal. */\nfunction isClaimRefusal(error: unknown): boolean {\n return (\n error instanceof RunClaimNotAcquiredError ||\n error instanceof RunClaimLostError\n )\n}\n\n/**\n * Sweep ONE run. Never rejects: the caller folds the returned entry into the\n * summary and moves on.\n *\n * The ORDER of the steps below is the contract, not an implementation detail:\n *\n * 1. **Classify expiry first**, because an expired run needs no probe — its\n * outcome is terminal whether or not the agent finished, so a probe would only\n * add a provider round-trip and a way to fail.\n * 2. **Otherwise probe BEFORE touching anything.** `'producing'` and `'unknown'`\n * return here, having made no claim, no append, no record write, and no\n * `close()`. Driving past this point is the whole defect described in the\n * module doc.\n * 3. Claim, so two hosts never drive one run.\n * 4. **Re-derive expiry from a record read INSIDE the lock**, and only then\n * record the cancel. The listed record is stale by the time the claim is\n * held, and the cancel is sticky.\n * 5. Quiesce, so a predecessor still writing is observed rather than raced.\n * 6. **Arm the run budget**, so it bounds the drive rather than the queue the\n * two steps above stood in.\n * 7. Pipe with BOTH authoritative seams fenced, mirroring `driver.ts`.\n * 8. Reclaim, and ONLY once the record actually reached terminal.\n */\nasync function reapOne<TOffset extends string>(\n record: RunRecord,\n ctx: ReapContext<TOffset>,\n counters: { probed: number },\n): Promise<ReapRunEntry> {\n const { runs, locks, logger } = ctx.options\n const { runId, threadId } = record\n\n try {\n // INCLUSIVE, exactly as `RunStore.listReclaimable` documents its own cutoff:\n // a run detached at precisely `now - ttlMs` IS expired. The two must agree,\n // or a run would be listed as reclaimable and then classified as fresh on\n // every single sweep, forever.\n const expired =\n record.detachedSince !== undefined && record.detachedSince <= ctx.cutoff\n\n let exitCode: number | undefined\n if (!expired) {\n counters.probed += 1\n const probe = await ctx.options.hasFinished(record)\n if (probe.state !== 'finished') {\n // THE LEAVE-ALONE PATH. Deliberately returns before `withRunClaim`, so\n // not even `driverEpoch` moves — and above all `detachedSince` is left\n // exactly as it was, since it is both this run's TTL evidence and the\n // field the next sweep selects on.\n safeLog(logger, 'sandbox', `reap: leaving run ${runId} alone`, {\n runId,\n state: probe.state,\n ...(probe.state === 'unknown' && probe.error !== undefined\n ? { error: probe.error }\n : {}),\n })\n return {\n runId,\n outcome: probe.state,\n ...(probe.state === 'unknown' && probe.error !== undefined\n ? { error: probe.error }\n : {}),\n }\n }\n exitCode = probe.exitCode\n }\n\n // Armed INSIDE the claim, below. Read after it for the outcome, so it is\n // hoisted here rather than declared in the callback.\n let budget: AbortSignal | undefined\n const final = await withRunClaim(\n {\n runs,\n locks,\n runId,\n fenceQuietMs: ctx.fenceQuietMs,\n ...(logger === undefined ? {} : { logger }),\n },\n async (claim) => {\n if (expired) {\n // RE-DERIVED FROM A RECORD READ INSIDE THE LOCK, never from the listed\n // one. `stream-to-response.ts`'s `startRunDriver` CLEARS\n // `detachedSince` when a real viewer attaches — deliberately stopping\n // the TTL clock — and it takes this same per-run lock, so an\n // expiry decided at listing time is stale by the time the claim is\n // held. Cancelling on the stale value poisoned a now-live run:\n // nothing in the tree ever clears `cancelRequested`, so on that\n // viewer's next ORDINARY disconnect `middleware.ts`'s\n // `wasCancelRequested` read skips the detach branch and destroys the\n // sandbox of a healthy, actively-viewed run.\n const current = await runs.get(runId)\n if (current === null) {\n throw new RunClaimNotAcquiredError(runId, 'unknown')\n }\n if (\n current.detachedSince === undefined ||\n current.detachedSince > ctx.cutoff\n ) {\n // The viewer came back. `'not-claimed'` already documents \"a real\n // viewer attaching mid-sweep is exactly this\", and refusing here\n // leaves the run as untouched as the leave-alone path does: no\n // cancel, no append, no terminal record, no `close()`.\n throw new RunClaimNotAcquiredError(runId, 'superseded')\n }\n // BEFORE the drive, never after — and never before the claim.\n // `withSandbox`'s `onAbort` resolves the out-of-band cancel band from\n // the record, so recording the intent first is what makes the teardown\n // an explicit cancel that DESTROYS the sandbox rather than a second\n // detach that re-arms `detachedSince` and leaves the run to be swept\n // again forever. Recorded after the drive it is pure bookkeeping on a\n // run that already tore down the wrong way. Recorded before the CLAIM\n // it is an unfenced, sticky write on a record this host does not own,\n // derived from a value the lock exists to make current.\n await requestRunCancel(runs, runId)\n }\n // Before the first append, never after: `pipeToRunLog` snapshots to align.\n await awaitLogQuiescence(\n ctx.options.durability(runId),\n ctx.fenceQuietMs,\n )\n // A safety net, not a mechanism (see the module doc). `AbortSignal.any`\n // is this package's idiom for linking one — see\n // `testkit/takeover-conformance.ts`.\n //\n // ARMED HERE, not before `withRunClaim`. Both the lock wait and the\n // quiescence wait consume a timer started earlier: quiescence always\n // sleeps at least one `fenceQuietMs` and may sleep six, and lock\n // acquisition waits behind whoever holds it, unbounded. The effective\n // budget was silently `runBudgetMs − fenceQuietMs − lockWait`, and once\n // it went negative the claim was acquired with the timer already fired:\n // `pipeToRunLog` hit its entry `signal.aborted` check before pulling one\n // chunk, so a FINISHED agent's transcript was recorded `'aborted'` and\n // its log closed — and a terminal record leaves `listReclaimable`\n // forever, so that transcript was then unreplayable while `reclaim`\n // destroyed the sandbox holding the only copy. The budget bounds the\n // DRIVE, not the queue.\n budget = AbortSignal.timeout(ctx.runBudgetMs)\n // `claim.signal` is in the composed signal because losing the lease MUST\n // stop the drive: a successor that took the run over is appending to the\n // same log, and this drive continuing would double every chunk.\n const signal = AbortSignal.any([claim.signal, budget])\n return pipeToRunLog(ctx.options.drive({ runId, threadId, signal }), {\n // BOTH seams, over the SAME claim, as `driver.ts` explains: fencing the\n // log alone just moves the harm to \"a dead host marks the successor's\n // live run failed\".\n runs: fenceRunStore(runs, claim, {\n ...(logger === undefined ? {} : { logger }),\n }),\n durability: (id) =>\n fenceDurability(ctx.options.durability(id), claim, { runs }),\n runId,\n threadId,\n signal,\n ...(logger === undefined ? {} : { logger }),\n })\n },\n )\n\n const terminal = isTerminalRunStatus(final.status)\n let outcome: ReapRunOutcome\n // `&& !expired` is the whole subtlety. The budget is an ANOMALY only on the\n // finalization path, where the probe already said the agent hit its sentinel\n // and a replay that will not finish in 30s means the journal read, the\n // translation, or the log is misbehaving. On the EXPIRY path there was no\n // probe and the agent may well be mid-sentence: `requestRunCancel` writes a\n // record field whose only reader is `withSandbox`'s `onAbort` (which runs\n // after something else has already aborted), so the budget is the sole thing\n // that ends the drive of a still-producing expired run. That is the designed\n // path, not a misbehaving one, and reporting it as the anomaly made\n // `'expired'` unreachable for exactly the runs the TTL exists to expire.\n // `budget` is armed inside the claim, so reaching here means it was armed;\n // `?? false` keeps the read total rather than asserting that.\n if ((budget?.aborted ?? false) && !expired) {\n outcome = 'budget-exceeded'\n } else if (!terminal) {\n // The terminal write was SUPPRESSED and `finish`'s re-read answered with a\n // live record, which `fenceRunStore` only does when this host lost the claim\n // to another one. That is the same fact as a refused claim, reported the\n // same way rather than as a success that wrote nothing.\n outcome = 'not-claimed'\n } else {\n outcome = expired ? 'expired' : 'finalized'\n }\n\n // CAPTURED BEFORE THE RECLAIM BLOCK, which may overwrite `outcome` with\n // `'reclaim-failed'`. Conditioning the `terminalizedAnyway` spread on the\n // post-reclaim `outcome` dropped the budget diagnostic from exactly the\n // entries that need it most: a run that blew its budget AND then failed to\n // reclaim reported neither fact but the leak, and an operator cannot\n // diagnose a leak on a run whose replay was already misbehaving without\n // knowing that it was.\n const budgetAnomaly = outcome === 'budget-exceeded'\n\n let reclaimError: unknown\n if (terminal && ctx.options.reclaim !== undefined) {\n try {\n // `record`, NOT `final`. When the terminal `update` fails, `finish` returns\n // a LOCALLY REBUILT record that carries only `runId`/`threadId`/`startedAt`\n // plus the terminal patch — no `sandboxKey` — so `reclaimSandbox` would see\n // `undefined`, answer `'no-sandbox-key'`, and the sandbox would leak\n // silently on exactly the path where something already went wrong.\n await ctx.options.reclaim(record)\n } catch (error) {\n // CAUGHT HERE rather than in the outer catch, which would report a bare\n // `'failed'` with no `status` and no `exitCode`. `reclaimSandbox`\n // deliberately does not guard `instances.get` (its contract is that the\n // CALLER records the failure) and neither does `sandboxReclaimer`, so a\n // throwing instance store landed there. By this point the record is\n // terminal and the log closed, so the run is out of `listReclaimable`\n // forever and no later sweep will retry: the sandbox leaks, and an\n // operator reading `'failed'` cannot tell \"transcript saved, sandbox NOT\n // reclaimed\" from \"the sweep failed and the run was never finalized\".\n reclaimError = error\n outcome = 'reclaim-failed'\n safeLog(logger, 'errors', `reap: reclaiming run ${runId} failed`, {\n runId,\n status: final.status,\n error,\n })\n }\n }\n\n return {\n runId,\n outcome,\n status: final.status,\n ...(exitCode === undefined ? {} : { exitCode }),\n ...(budgetAnomaly ? { terminalizedAnyway: terminal } : {}),\n ...(reclaimError === undefined ? {} : { error: reclaimError }),\n }\n } catch (error) {\n if (isClaimRefusal(error)) {\n safeLog(logger, 'sandbox', `reap: not driving run ${runId}`, {\n runId,\n error,\n })\n return { runId, outcome: 'not-claimed', error }\n }\n // Folded into the summary rather than rethrown: one bad run must not abandon\n // the rest of the batch, and there is no caller to receive a rejection.\n safeLog(logger, 'errors', `reap: sweeping run ${runId} failed`, {\n runId,\n error,\n })\n return { runId, outcome: 'failed', error }\n }\n}\n\n/**\n * Sweep the detached runs a `RunStore` surfaces, saving each finished run's\n * transcript and reclaiming its sandbox.\n *\n * A plain async function with no timer and no daemon: call it from a cron, a\n * queue consumer, a Durable Object `alarm()`, or a `waitUntil`. It NEVER rejects\n * — every failure is logged and counted in the returned {@link ReapResult}.\n *\n * ONE `listReclaimable({ now, ttlMs: 0 })` call, deliberately: `ttlMs: 0` is\n * every detached run, which is the candidate set for FINALIZATION (a run that hit\n * its sentinel one second after the viewer left has an unsaved transcript and\n * must not wait out the TTL), and expiry is then classified in-process against\n * the same inclusive cutoff. Listing twice with two TTLs would cost a second\n * store round-trip to compute a subset.\n *\n * `listReclaimable` is OPTIONAL on `RunStore`. A backend without it cannot be\n * reaped, which answers `{ considered: 0 }` plus one log line rather than\n * throwing — the same graceful degrade every other optional-method call site in\n * the repo does (`store.findActiveRun?.(threadId)`).\n */\nexport async function reapDetachedRuns<TOffset extends string = string>(\n options: ReapOptions<TOffset>,\n): Promise<ReapResult> {\n const logger = options.logger\n const outcomes = emptyOutcomes()\n const entries: Array<ReapRunEntry> = []\n const empty = (): ReapResult => ({\n considered: 0,\n probed: 0,\n outcomes,\n runs: entries,\n })\n\n const list = options.runs.listReclaimable?.bind(options.runs)\n if (list === undefined) {\n safeLog(\n logger,\n 'sandbox',\n 'reap: the run store does not implement listReclaimable; nothing to sweep',\n {},\n )\n return empty()\n }\n\n let candidates: Array<RunRecord>\n try {\n candidates = await list({ now: options.now, ttlMs: 0 })\n } catch (error) {\n safeLog(logger, 'errors', 'reap: listing reclaimable runs failed', {\n error,\n })\n return empty()\n }\n\n // Capped so one invocation cannot outlive its platform's budget and be killed\n // mid-drive. `slice` and not a `break`, so `considered` reports the batch the\n // sweep actually took responsibility for.\n const maxRuns = Math.max(0, Math.trunc(options.maxRuns ?? DEFAULT_MAX_RUNS))\n const batch = candidates.slice(0, maxRuns)\n\n const ctx: ReapContext<TOffset> = {\n options,\n runBudgetMs: options.runBudgetMs ?? DEFAULT_RUN_BUDGET_MS,\n fenceQuietMs: options.fenceQuietMs ?? DEFAULT_FENCE_QUIET_MS,\n cutoff: options.now - options.detachedRunTtlMs,\n }\n const counters = { probed: 0 }\n\n // Sequential on purpose: each run costs a lock, a provider round-trip, and a\n // full replay, and a cron invocation's budget is the scarce resource. Fanning\n // out would multiply peak load against the provider for no throughput a\n // subsequent tick cannot supply.\n for (const record of batch) {\n const entry = await reapOne(record, ctx, counters)\n outcomes[entry.outcome] += 1\n entries.push(entry)\n }\n\n return {\n considered: batch.length,\n probed: counters.probed,\n outcomes,\n runs: entries,\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAyGA,IAAa,wBAAwB;;;;;;;AAQrC,IAAa,mBAAmB;;AAGhC,IAAa,2BAA2B;AAuKxC,gBAAgB,YAAY,OAAsC;CAChE,MAAM;AACR;;AAGA,eAAe,YAAY,QAAiC;CAC1D,MAAM,UAAU,IAAI,YAAY;CAChC,IAAI,OAAO;CACX,WAAW,MAAM,SAAS,mBAAmB,YAAY,MAAM,CAAC,GAC9D,QAAQ,QAAQ,OAAO,OAAO,EAAE,QAAQ,KAAK,CAAC;CAEhD,OAAO,OAAO,QAAQ,OAAO;AAC/B;;;;;;;;;;;;;;;;;AAkBA,eAAsB,aAAa,OAOT;CACxB,IAAI;EACF,MAAM,QAAQ,aAAa,MAAM,OAAO,MAAM,GAAG;EACjD,MAAM,SAAS,MAAM,MAAM,OAAO,QAAQ,KACxC,wBACE,OACA,MAAM,YAAA,IACR,CACF;EAIA,MAAM,WAAW,iBAAiB,MAAM,YAAY,OAAO,MAAM,GAAG,KAAK;EACzE,OAAO,aAAa,OAChB,EAAE,OAAO,YAAY,IACrB;GAAE,OAAO;GAAY;EAAS;CACpC,SAAS,OAAO;EACd,OAAO;GAAE,OAAO;GAAW;EAAM;CACnC;AACF;;AAGA,SAAS,gBAAgD;CACvD,OAAO;EACL,WAAW;EACX,SAAS;EACT,WAAW;EACX,SAAS;EACT,mBAAmB;EACnB,eAAe;EACf,kBAAkB;EAClB,QAAQ;CACV;AACF;;;;;;AAOA,SAAS,QACP,QACA,OACA,SACA,SACM;CACN,IAAI;EACF,IAAI,UAAU,UAAU,QAAQ,OAAO,SAAS,OAAO;OAClD,QAAQ,QAAQ,SAAS,OAAO;CACvC,QAAQ,CAER;AACF;;AAYA,SAAS,eAAe,OAAyB;CAC/C,OACE,iBAAiB,4BACjB,iBAAiB;AAErB;;;;;;;;;;;;;;;;;;;;;;;;AAyBA,eAAe,QACb,QACA,KACA,UACuB;CACvB,MAAM,EAAE,MAAM,OAAO,WAAW,IAAI;CACpC,MAAM,EAAE,OAAO,aAAa;CAE5B,IAAI;EAKF,MAAM,UACJ,OAAO,kBAAkB,KAAA,KAAa,OAAO,iBAAiB,IAAI;EAEpE,IAAI;EACJ,IAAI,CAAC,SAAS;GACZ,SAAS,UAAU;GACnB,MAAM,QAAQ,MAAM,IAAI,QAAQ,YAAY,MAAM;GAClD,IAAI,MAAM,UAAU,YAAY;IAK9B,QAAQ,QAAQ,WAAW,qBAAqB,MAAM,SAAS;KAC7D;KACA,OAAO,MAAM;KACb,GAAI,MAAM,UAAU,aAAa,MAAM,UAAU,KAAA,IAC7C,EAAE,OAAO,MAAM,MAAM,IACrB,CAAC;IACP,CAAC;IACD,OAAO;KACL;KACA,SAAS,MAAM;KACf,GAAI,MAAM,UAAU,aAAa,MAAM,UAAU,KAAA,IAC7C,EAAE,OAAO,MAAM,MAAM,IACrB,CAAC;IACP;GACF;GACA,WAAW,MAAM;EACnB;EAIA,IAAI;EACJ,MAAM,QAAQ,MAAM,aAClB;GACE;GACA;GACA;GACA,cAAc,IAAI;GAClB,GAAI,WAAW,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO;EAC3C,GACA,OAAO,UAAU;GACf,IAAI,SAAS;IAWX,MAAM,UAAU,MAAM,KAAK,IAAI,KAAK;IACpC,IAAI,YAAY,MACd,MAAM,IAAI,yBAAyB,OAAO,SAAS;IAErD,IACE,QAAQ,kBAAkB,KAAA,KAC1B,QAAQ,gBAAgB,IAAI,QAM5B,MAAM,IAAI,yBAAyB,OAAO,YAAY;IAWxD,MAAM,iBAAiB,MAAM,KAAK;GACpC;GAEA,MAAM,mBACJ,IAAI,QAAQ,WAAW,KAAK,GAC5B,IAAI,YACN;GAiBA,SAAS,YAAY,QAAQ,IAAI,WAAW;GAI5C,MAAM,SAAS,YAAY,IAAI,CAAC,MAAM,QAAQ,MAAM,CAAC;GACrD,OAAO,aAAa,IAAI,QAAQ,MAAM;IAAE;IAAO;IAAU;GAAO,CAAC,GAAG;IAIlE,MAAM,cAAc,MAAM,OAAO,EAC/B,GAAI,WAAW,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,EAC3C,CAAC;IACD,aAAa,OACX,gBAAgB,IAAI,QAAQ,WAAW,EAAE,GAAG,OAAO,EAAE,KAAK,CAAC;IAC7D;IACA;IACA;IACA,GAAI,WAAW,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO;GAC3C,CAAC;EACH,CACF;EAEA,MAAM,WAAW,oBAAoB,MAAM,MAAM;EACjD,IAAI;EAaJ,KAAK,QAAQ,WAAW,UAAU,CAAC,SACjC,UAAU;OACL,IAAI,CAAC,UAKV,UAAU;OAEV,UAAU,UAAU,YAAY;EAUlC,MAAM,gBAAgB,YAAY;EAElC,IAAI;EACJ,IAAI,YAAY,IAAI,QAAQ,YAAY,KAAA,GACtC,IAAI;GAMF,MAAM,IAAI,QAAQ,QAAQ,MAAM;EAClC,SAAS,OAAO;GAUd,eAAe;GACf,UAAU;GACV,QAAQ,QAAQ,UAAU,wBAAwB,MAAM,UAAU;IAChE;IACA,QAAQ,MAAM;IACd;GACF,CAAC;EACH;EAGF,OAAO;GACL;GACA;GACA,QAAQ,MAAM;GACd,GAAI,aAAa,KAAA,IAAY,CAAC,IAAI,EAAE,SAAS;GAC7C,GAAI,gBAAgB,EAAE,oBAAoB,SAAS,IAAI,CAAC;GACxD,GAAI,iBAAiB,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,aAAa;EAC9D;CACF,SAAS,OAAO;EACd,IAAI,eAAe,KAAK,GAAG;GACzB,QAAQ,QAAQ,WAAW,yBAAyB,SAAS;IAC3D;IACA;GACF,CAAC;GACD,OAAO;IAAE;IAAO,SAAS;IAAe;GAAM;EAChD;EAGA,QAAQ,QAAQ,UAAU,sBAAsB,MAAM,UAAU;GAC9D;GACA;EACF,CAAC;EACD,OAAO;GAAE;GAAO,SAAS;GAAU;EAAM;CAC3C;AACF;;;;;;;;;;;;;;;;;;;;;AAsBA,eAAsB,iBACpB,SACqB;CACrB,MAAM,SAAS,QAAQ;CACvB,MAAM,WAAW,cAAc;CAC/B,MAAM,UAA+B,CAAC;CACtC,MAAM,eAA2B;EAC/B,YAAY;EACZ,QAAQ;EACR;EACA,MAAM;CACR;CAEA,MAAM,OAAO,QAAQ,KAAK,iBAAiB,KAAK,QAAQ,IAAI;CAC5D,IAAI,SAAS,KAAA,GAAW;EACtB,QACE,QACA,WACA,4EACA,CAAC,CACH;EACA,OAAO,MAAM;CACf;CAEA,IAAI;CACJ,IAAI;EACF,aAAa,MAAM,KAAK;GAAE,KAAK,QAAQ;GAAK,OAAO;EAAE,CAAC;CACxD,SAAS,OAAO;EACd,QAAQ,QAAQ,UAAU,yCAAyC,EACjE,MACF,CAAC;EACD,OAAO,MAAM;CACf;CAKA,MAAM,UAAU,KAAK,IAAI,GAAG,KAAK,MAAM,QAAQ,WAAA,EAA2B,CAAC;CAC3E,MAAM,QAAQ,WAAW,MAAM,GAAG,OAAO;CAEzC,MAAM,MAA4B;EAChC;EACA,aAAa,QAAQ,eAAA;EACrB,cAAc,QAAQ,gBAAA;EACtB,QAAQ,QAAQ,MAAM,QAAQ;CAChC;CACA,MAAM,WAAW,EAAE,QAAQ,EAAE;CAM7B,KAAK,MAAM,UAAU,OAAO;EAC1B,MAAM,QAAQ,MAAM,QAAQ,QAAQ,KAAK,QAAQ;EACjD,SAAS,MAAM,YAAY;EAC3B,QAAQ,KAAK,KAAK;CACpB;CAEA,OAAO;EACL,YAAY,MAAM;EAClB,QAAQ,SAAS;EACjB;EACA,MAAM;CACR;AACF"}
|
package/dist/esm/sandbox.d.ts
CHANGED
|
@@ -62,6 +62,8 @@ export interface SandboxEnsureContext {
|
|
|
62
62
|
orgId?: string;
|
|
63
63
|
};
|
|
64
64
|
signal?: AbortSignal;
|
|
65
|
+
/** Harness adapter name (`grok-build`, `claude-code`, `codex`, `opencode`). Optional. */
|
|
66
|
+
adapterName?: string;
|
|
65
67
|
}
|
|
66
68
|
export interface SandboxDefinition {
|
|
67
69
|
readonly id: string;
|
package/dist/esm/sandbox.js
CHANGED
|
@@ -29,9 +29,23 @@ function parseMaxAgeMs(value) {
|
|
|
29
29
|
* enough that a slow provider API still completes, short enough that a wedged
|
|
30
30
|
* one cannot pin the process forever.
|
|
31
31
|
*/
|
|
32
|
-
var DESTROY_TIMEOUT_MS =
|
|
32
|
+
var DESTROY_TIMEOUT_MS = 6e4;
|
|
33
33
|
var fallbackStore = new InMemorySandboxInstanceStore();
|
|
34
34
|
var fallbackLocks = new InMemoryLockStore();
|
|
35
|
+
/**
|
|
36
|
+
* Put workspace secrets onto a live handle. Resume and snapshot restore skip
|
|
37
|
+
* bootstrap, so this is the only path that re-injects them after reconnect.
|
|
38
|
+
* Create injects secrets via `provider.create({ env })`, but resume/restore
|
|
39
|
+
* return a handle whose process env is empty unless we set it here. sbx in
|
|
40
|
+
* particular has no Docker Env on resume, so this is the only way secrets
|
|
41
|
+
* come back for that provider.
|
|
42
|
+
*/
|
|
43
|
+
async function applyWorkspaceSecrets(handle, workspace) {
|
|
44
|
+
if (workspace?.secrets === void 0) return;
|
|
45
|
+
const resolved = resolveAllSecrets(workspace.secrets);
|
|
46
|
+
if (Object.keys(resolved).length === 0) return;
|
|
47
|
+
await handle.env.set(resolved);
|
|
48
|
+
}
|
|
35
49
|
function defineSandbox(config) {
|
|
36
50
|
const keyInputFor = (ctx) => ({
|
|
37
51
|
threadId: config.lifecycle?.reuse === "none" ? `${ctx.threadId}:${ctx.runId}` : ctx.threadId,
|
|
@@ -56,6 +70,7 @@ function defineSandbox(config) {
|
|
|
56
70
|
signal: ctx.signal
|
|
57
71
|
});
|
|
58
72
|
if (resumed) {
|
|
73
|
+
await applyWorkspaceSecrets(resumed, config.workspace);
|
|
59
74
|
await store.upsert({
|
|
60
75
|
...existing,
|
|
61
76
|
latestRunId: ctx.runId,
|
|
@@ -71,6 +86,7 @@ function defineSandbox(config) {
|
|
|
71
86
|
env: config.workspace?.secrets !== void 0 ? resolveAllSecrets(config.workspace.secrets) : void 0,
|
|
72
87
|
signal: ctx.signal
|
|
73
88
|
});
|
|
89
|
+
await applyWorkspaceSecrets(restored, config.workspace);
|
|
74
90
|
await store.upsert({
|
|
75
91
|
...existing,
|
|
76
92
|
providerSandboxId: restored.id,
|
|
@@ -86,7 +102,8 @@ function defineSandbox(config) {
|
|
|
86
102
|
workspace: config.workspace,
|
|
87
103
|
policy: config.policy,
|
|
88
104
|
env: config.workspace?.secrets !== void 0 ? resolveAllSecrets(config.workspace.secrets) : void 0,
|
|
89
|
-
signal: ctx.signal
|
|
105
|
+
signal: ctx.signal,
|
|
106
|
+
adapterName: ctx.adapterName
|
|
90
107
|
});
|
|
91
108
|
if (config.workspace) try {
|
|
92
109
|
await bootstrapWorkspace(created, config.workspace, { signal: ctx.signal });
|
package/dist/esm/sandbox.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"sandbox.js","names":[],"sources":["../../src/sandbox.ts"],"sourcesContent":["/**\n * `defineSandbox()` returns a LAZY controller — it never creates a sandbox at\n * definition time. `withSandbox()` (and advanced users) call `ensure()` to\n * resume-or-create, following: provider.resume → provider.restoreSnapshot →\n * create + bootstrap. The controller folds provider/workspace/policy/lifecycle\n * into a stable instance key and coordinates through the (optional) lock +\n * sandbox stores.\n */\nimport { bootstrapWorkspace } from './bootstrap'\nimport { resolveAllSecrets } from './secrets'\nimport { computeSandboxKey } from './key'\nimport { InMemoryLockStore } from '@tanstack/ai/locks'\nimport type { LockStore } from '@tanstack/ai/locks'\nimport type { SandboxFileHookEvent } from '@tanstack/ai'\nimport { InMemorySandboxInstanceStore } from './instance-store'\nimport type { SandboxInstanceStore } from './instance-store'\nimport type { SandboxHandle, SandboxProvider } from './contracts'\nimport type { SandboxKeyInput } from './key'\nimport type { SandboxPolicy } from './policy'\nimport type { WorkspaceDefinition } from './workspace'\n\n/**\n * Sandbox-scoped hooks declared on `defineSandbox`. File hooks fire for every\n * create/change/delete during a chat run; lifecycle hooks fire server-side.\n */\nexport interface SandboxHooks {\n onFile?: (e: SandboxFileHookEvent) => void | Promise<void>\n onFileCreate?: (e: SandboxFileHookEvent) => void | Promise<void>\n onFileChange?: (e: SandboxFileHookEvent) => void | Promise<void>\n onFileDelete?: (e: SandboxFileHookEvent) => void | Promise<void>\n onReady?: (handle: SandboxHandle) => void | Promise<void>\n onError?: (err: unknown) => void | Promise<void>\n onDestroy?: () => void | Promise<void>\n}\n\nexport type ReuseStrategy = 'thread' | 'none'\nexport type SnapshotStrategy = 'after-setup' | 'after-run' | 'none'\n\nexport interface SandboxLifecycle {\n /** `'thread'` resumes one sandbox per thread; `'none'` is fresh per run. */\n reuse?: ReuseStrategy\n /** When to snapshot (provider-permitting). */\n snapshot?: SnapshotStrategy\n /** Hint for how long a provider should keep the sandbox warm between runs. */\n keepAlive?: string\n /** Destroy the sandbox after the run completes. */\n destroyOnComplete?: boolean\n /**\n * Maximum age of a sandbox record before it is discarded and re-created\n * instead of resumed. Accepts `'<n>h'` (hours) or `'<n>m'` (minutes),\n * e.g. `'2h'` or `'30m'`.\n */\n snapshotMaxAge?: string\n}\n\nexport interface SandboxConfig {\n id: string\n provider: SandboxProvider\n workspace?: WorkspaceDefinition\n policy?: SandboxPolicy\n lifecycle?: SandboxLifecycle\n /** Sandbox-scoped file/lifecycle hooks. */\n hooks?: SandboxHooks\n /** Watch the workspace for file events (default true). `false` disables the\n * watcher; `{ diff: true }` also emits a per-file `sandbox.file.diff` event. */\n fileEvents?: boolean | { diff?: boolean }\n}\n\n/** Context passed to `ensure()` by `withSandbox` (or advanced callers). */\nexport interface SandboxEnsureContext {\n threadId: string\n runId: string\n /** Persistence seam; falls back to an in-memory store when absent. */\n store?: SandboxInstanceStore\n /** Lock seam; falls back to an in-memory lock when absent. */\n locks?: LockStore\n tenant?: { userId?: string; orgId?: string }\n signal?: AbortSignal\n}\n\nexport interface SandboxDefinition {\n readonly id: string\n readonly provider: SandboxProvider\n readonly workspace?: WorkspaceDefinition\n readonly policy?: SandboxPolicy\n readonly lifecycle?: SandboxLifecycle\n /** Sandbox-scoped file/lifecycle hooks. */\n readonly hooks?: SandboxHooks\n /** Watch the workspace for file events (default true). `false` disables the\n * watcher; `{ diff: true }` also emits a per-file `sandbox.file.diff` event. */\n readonly fileEvents?: boolean | { diff?: boolean }\n /** Compound instance key for a given run context. */\n key: (ctx: SandboxEnsureContext) => string\n /** Resume-or-create the sandbox for this thread/run. */\n ensure: (ctx: SandboxEnsureContext) => Promise<SandboxHandle>\n /** Tear down the sandbox recorded for this key. */\n destroy: (ctx: SandboxEnsureContext) => Promise<void>\n}\n\n/**\n * Parse a human-readable duration string into milliseconds.\n * Supports `'<n>h'` (hours) and `'<n>m'` (minutes).\n * Returns `undefined` when the input is undefined or the format is unrecognised.\n */\nfunction parseMaxAgeMs(value: string | undefined): number | undefined {\n if (value === undefined) return undefined\n const hourMatch = /^(\\d+)h$/.exec(value)\n if (hourMatch) return Number(hourMatch[1]) * 60 * 60 * 1000\n const minuteMatch = /^(\\d+)m$/.exec(value)\n if (minuteMatch) return Number(minuteMatch[1]) * 60 * 1000\n return undefined\n}\n\n/**\n * Bound for the unfenced teardown `destroy` call (see `destroy` below). Long\n * enough that a slow provider API still completes, short enough that a wedged\n * one cannot pin the process forever.\n */\nconst DESTROY_TIMEOUT_MS = 60 * 1000\n\n// Process-lifetime fallbacks shared across all definitions so concurrent\n// ensures for the same key serialize even without an injected store/lock.\nconst fallbackStore = new InMemorySandboxInstanceStore()\nconst fallbackLocks = new InMemoryLockStore()\n\nexport function defineSandbox(config: SandboxConfig): SandboxDefinition {\n const keyInputFor = (ctx: SandboxEnsureContext): SandboxKeyInput => ({\n threadId:\n config.lifecycle?.reuse === 'none'\n ? `${ctx.threadId}:${ctx.runId}`\n : ctx.threadId,\n sandboxId: config.id,\n providerName: config.provider.name,\n workspace: config.workspace,\n tenant: ctx.tenant,\n })\n\n const ensure = async (ctx: SandboxEnsureContext): Promise<SandboxHandle> => {\n const store = ctx.store ?? fallbackStore\n const locks = ctx.locks ?? fallbackLocks\n const key = computeSandboxKey(keyInputFor(ctx))\n const caps = config.provider.capabilities()\n\n return locks.withLock(`sandbox:${key}`, async () => {\n const effectiveSnapshot: SnapshotStrategy =\n config.lifecycle?.snapshot ?? (caps.snapshots ? 'after-setup' : 'none')\n const maxAgeMs = parseMaxAgeMs(config.lifecycle?.snapshotMaxAge)\n\n const existing = await store.get(key)\n if (existing) {\n // Check whether the record has exceeded snapshotMaxAge; if so,\n // discard and fall through to a fresh create.\n const tooOld =\n maxAgeMs !== undefined && Date.now() - existing.updatedAt > maxAgeMs\n\n if (!tooOld) {\n // 1) Try to reconnect to the still-running sandbox.\n const resumed = await config.provider.resume({\n id: existing.providerSandboxId,\n signal: ctx.signal,\n })\n if (resumed) {\n await store.upsert({\n ...existing,\n latestRunId: ctx.runId,\n updatedAt: Date.now(),\n })\n return resumed\n }\n // 2) Else restore from the latest snapshot, if supported.\n if (\n existing.latestSnapshotId &&\n caps.snapshots &&\n config.provider.restoreSnapshot\n ) {\n const restored = await config.provider.restoreSnapshot({\n snapshotId: existing.latestSnapshotId,\n workspace: config.workspace,\n policy: config.policy,\n env:\n config.workspace?.secrets !== undefined\n ? resolveAllSecrets(config.workspace.secrets)\n : undefined,\n signal: ctx.signal,\n })\n await store.upsert({\n ...existing,\n providerSandboxId: restored.id,\n latestRunId: ctx.runId,\n updatedAt: Date.now(),\n })\n return restored\n }\n }\n // 3) Else fall through and re-create under the same identity\n // (capability-aware degradation for ephemeral-disk providers, or\n // snapshotMaxAge TTL exceeded).\n }\n\n const created = await config.provider.create({\n // Deterministic id so consumers can reconstruct the provider sandbox\n // address from run context (not just from the store record).\n id: key,\n workspace: config.workspace,\n policy: config.policy,\n env:\n config.workspace?.secrets !== undefined\n ? resolveAllSecrets(config.workspace.secrets)\n : undefined,\n signal: ctx.signal,\n })\n\n if (config.workspace) {\n try {\n await bootstrapWorkspace(created, config.workspace, {\n signal: ctx.signal,\n })\n } catch (error) {\n // Bootstrap failed after the sandbox was created but before it was\n // recorded — destroy the orphan so a failed/retried run doesn't leak\n // a (billed) sandbox, then surface the original error.\n await created.destroy().catch(() => {})\n throw error\n }\n }\n\n let latestSnapshotId: string | undefined\n if (\n effectiveSnapshot === 'after-setup' &&\n caps.snapshots &&\n created.snapshot\n ) {\n latestSnapshotId = (await created.snapshot('after-setup')).id\n }\n\n await store.upsert({\n key,\n provider: config.provider.name,\n providerSandboxId: created.id,\n latestSnapshotId,\n threadId: ctx.threadId,\n latestRunId: ctx.runId,\n updatedAt: Date.now(),\n })\n return created\n })\n }\n\n const destroy = async (ctx: SandboxEnsureContext): Promise<void> => {\n const store = ctx.store ?? fallbackStore\n const key = computeSandboxKey(keyInputFor(ctx))\n const existing = await store.get(key)\n if (!existing) return\n /*\n * TEARDOWN IS DELIBERATELY NOT FENCED BY `ctx.signal`.\n *\n * `destroy` runs on every teardown path INCLUDING the one caused by that\n * very signal aborting, so forwarding it hands the provider a signal that is\n * already aborted: a provider that honors it does nothing and returns\n * successfully, and `store.delete` below then removes the only pointer to a\n * live, billed sandbox. `SandboxInstanceStore` has no `list` (see the note\n * at the top of `reclaim.ts`), so that sandbox is unreachable from then on.\n *\n * Same reasoning as `close()` never being fenced by the run claim (see\n * `fenceDurability` in `claim.ts`): cleanup must outlive whatever cancelled\n * the work. A fresh controller with its own bounded timeout keeps the call\n * from hanging forever without letting the caller's abort cancel it.\n */\n const teardown = new AbortController()\n const timer = setTimeout(() => teardown.abort(), DESTROY_TIMEOUT_MS)\n try {\n await config.provider.destroy({\n id: existing.providerSandboxId,\n signal: teardown.signal,\n })\n } finally {\n clearTimeout(timer)\n }\n await store.delete(key)\n }\n\n return {\n id: config.id,\n provider: config.provider,\n workspace: config.workspace,\n policy: config.policy,\n lifecycle: config.lifecycle,\n hooks: config.hooks,\n fileEvents: config.fileEvents,\n key: (ctx) => computeSandboxKey(keyInputFor(ctx)),\n ensure,\n destroy,\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;AAwGA,SAAS,cAAc,OAA+C;CACpE,IAAI,UAAU,KAAA,GAAW,OAAO,KAAA;CAChC,MAAM,YAAY,WAAW,KAAK,KAAK;CACvC,IAAI,WAAW,OAAO,OAAO,UAAU,EAAE,IAAI,KAAK,KAAK;CACvD,MAAM,cAAc,WAAW,KAAK,KAAK;CACzC,IAAI,aAAa,OAAO,OAAO,YAAY,EAAE,IAAI,KAAK;AAExD;;;;;;AAOA,IAAM,qBAAqB,KAAK;AAIhC,IAAM,gBAAgB,IAAI,6BAA6B;AACvD,IAAM,gBAAgB,IAAI,kBAAkB;AAE5C,SAAgB,cAAc,QAA0C;CACtE,MAAM,eAAe,SAAgD;EACnE,UACE,OAAO,WAAW,UAAU,SACxB,GAAG,IAAI,SAAS,GAAG,IAAI,UACvB,IAAI;EACV,WAAW,OAAO;EAClB,cAAc,OAAO,SAAS;EAC9B,WAAW,OAAO;EAClB,QAAQ,IAAI;CACd;CAEA,MAAM,SAAS,OAAO,QAAsD;EAC1E,MAAM,QAAQ,IAAI,SAAS;EAC3B,MAAM,QAAQ,IAAI,SAAS;EAC3B,MAAM,MAAM,kBAAkB,YAAY,GAAG,CAAC;EAC9C,MAAM,OAAO,OAAO,SAAS,aAAa;EAE1C,OAAO,MAAM,SAAS,WAAW,OAAO,YAAY;GAClD,MAAM,oBACJ,OAAO,WAAW,aAAa,KAAK,YAAY,gBAAgB;GAClE,MAAM,WAAW,cAAc,OAAO,WAAW,cAAc;GAE/D,MAAM,WAAW,MAAM,MAAM,IAAI,GAAG;GACpC,IAAI;QAME,EAFF,aAAa,KAAA,KAAa,KAAK,IAAI,IAAI,SAAS,YAAY,WAEjD;KAEX,MAAM,UAAU,MAAM,OAAO,SAAS,OAAO;MAC3C,IAAI,SAAS;MACb,QAAQ,IAAI;KACd,CAAC;KACD,IAAI,SAAS;MACX,MAAM,MAAM,OAAO;OACjB,GAAG;OACH,aAAa,IAAI;OACjB,WAAW,KAAK,IAAI;MACtB,CAAC;MACD,OAAO;KACT;KAEA,IACE,SAAS,oBACT,KAAK,aACL,OAAO,SAAS,iBAChB;MACA,MAAM,WAAW,MAAM,OAAO,SAAS,gBAAgB;OACrD,YAAY,SAAS;OACrB,WAAW,OAAO;OAClB,QAAQ,OAAO;OACf,KACE,OAAO,WAAW,YAAY,KAAA,IAC1B,kBAAkB,OAAO,UAAU,OAAO,IAC1C,KAAA;OACN,QAAQ,IAAI;MACd,CAAC;MACD,MAAM,MAAM,OAAO;OACjB,GAAG;OACH,mBAAmB,SAAS;OAC5B,aAAa,IAAI;OACjB,WAAW,KAAK,IAAI;MACtB,CAAC;MACD,OAAO;KACT;IACF;;GAMF,MAAM,UAAU,MAAM,OAAO,SAAS,OAAO;IAG3C,IAAI;IACJ,WAAW,OAAO;IAClB,QAAQ,OAAO;IACf,KACE,OAAO,WAAW,YAAY,KAAA,IAC1B,kBAAkB,OAAO,UAAU,OAAO,IAC1C,KAAA;IACN,QAAQ,IAAI;GACd,CAAC;GAED,IAAI,OAAO,WACT,IAAI;IACF,MAAM,mBAAmB,SAAS,OAAO,WAAW,EAClD,QAAQ,IAAI,OACd,CAAC;GACH,SAAS,OAAO;IAId,MAAM,QAAQ,QAAQ,CAAC,CAAC,YAAY,CAAC,CAAC;IACtC,MAAM;GACR;GAGF,IAAI;GACJ,IACE,sBAAsB,iBACtB,KAAK,aACL,QAAQ,UAER,oBAAoB,MAAM,QAAQ,SAAS,aAAa,EAAA,CAAG;GAG7D,MAAM,MAAM,OAAO;IACjB;IACA,UAAU,OAAO,SAAS;IAC1B,mBAAmB,QAAQ;IAC3B;IACA,UAAU,IAAI;IACd,aAAa,IAAI;IACjB,WAAW,KAAK,IAAI;GACtB,CAAC;GACD,OAAO;EACT,CAAC;CACH;CAEA,MAAM,UAAU,OAAO,QAA6C;EAClE,MAAM,QAAQ,IAAI,SAAS;EAC3B,MAAM,MAAM,kBAAkB,YAAY,GAAG,CAAC;EAC9C,MAAM,WAAW,MAAM,MAAM,IAAI,GAAG;EACpC,IAAI,CAAC,UAAU;EAgBf,MAAM,WAAW,IAAI,gBAAgB;EACrC,MAAM,QAAQ,iBAAiB,SAAS,MAAM,GAAG,kBAAkB;EACnE,IAAI;GACF,MAAM,OAAO,SAAS,QAAQ;IAC5B,IAAI,SAAS;IACb,QAAQ,SAAS;GACnB,CAAC;EACH,UAAU;GACR,aAAa,KAAK;EACpB;EACA,MAAM,MAAM,OAAO,GAAG;CACxB;CAEA,OAAO;EACL,IAAI,OAAO;EACX,UAAU,OAAO;EACjB,WAAW,OAAO;EAClB,QAAQ,OAAO;EACf,WAAW,OAAO;EAClB,OAAO,OAAO;EACd,YAAY,OAAO;EACnB,MAAM,QAAQ,kBAAkB,YAAY,GAAG,CAAC;EAChD;EACA;CACF;AACF"}
|
|
1
|
+
{"version":3,"file":"sandbox.js","names":[],"sources":["../../src/sandbox.ts"],"sourcesContent":["/**\n * `defineSandbox()` returns a LAZY controller — it never creates a sandbox at\n * definition time. `withSandbox()` (and advanced users) call `ensure()` to\n * resume-or-create, following: provider.resume → provider.restoreSnapshot →\n * create + bootstrap. The controller folds provider/workspace/policy/lifecycle\n * into a stable instance key and coordinates through the (optional) lock +\n * sandbox stores.\n */\nimport { bootstrapWorkspace } from './bootstrap'\nimport { resolveAllSecrets } from './secrets'\nimport { computeSandboxKey } from './key'\nimport { InMemoryLockStore } from '@tanstack/ai/locks'\nimport type { LockStore } from '@tanstack/ai/locks'\nimport type { SandboxFileHookEvent } from '@tanstack/ai'\nimport { InMemorySandboxInstanceStore } from './instance-store'\nimport type { SandboxInstanceStore } from './instance-store'\nimport type { SandboxHandle, SandboxProvider } from './contracts'\nimport type { SandboxKeyInput } from './key'\nimport type { SandboxPolicy } from './policy'\nimport type { WorkspaceDefinition } from './workspace'\n\n/**\n * Sandbox-scoped hooks declared on `defineSandbox`. File hooks fire for every\n * create/change/delete during a chat run; lifecycle hooks fire server-side.\n */\nexport interface SandboxHooks {\n onFile?: (e: SandboxFileHookEvent) => void | Promise<void>\n onFileCreate?: (e: SandboxFileHookEvent) => void | Promise<void>\n onFileChange?: (e: SandboxFileHookEvent) => void | Promise<void>\n onFileDelete?: (e: SandboxFileHookEvent) => void | Promise<void>\n onReady?: (handle: SandboxHandle) => void | Promise<void>\n onError?: (err: unknown) => void | Promise<void>\n onDestroy?: () => void | Promise<void>\n}\n\nexport type ReuseStrategy = 'thread' | 'none'\nexport type SnapshotStrategy = 'after-setup' | 'after-run' | 'none'\n\nexport interface SandboxLifecycle {\n /** `'thread'` resumes one sandbox per thread; `'none'` is fresh per run. */\n reuse?: ReuseStrategy\n /** When to snapshot (provider-permitting). */\n snapshot?: SnapshotStrategy\n /** Hint for how long a provider should keep the sandbox warm between runs. */\n keepAlive?: string\n /** Destroy the sandbox after the run completes. */\n destroyOnComplete?: boolean\n /**\n * Maximum age of a sandbox record before it is discarded and re-created\n * instead of resumed. Accepts `'<n>h'` (hours) or `'<n>m'` (minutes),\n * e.g. `'2h'` or `'30m'`.\n */\n snapshotMaxAge?: string\n}\n\nexport interface SandboxConfig {\n id: string\n provider: SandboxProvider\n workspace?: WorkspaceDefinition\n policy?: SandboxPolicy\n lifecycle?: SandboxLifecycle\n /** Sandbox-scoped file/lifecycle hooks. */\n hooks?: SandboxHooks\n /** Watch the workspace for file events (default true). `false` disables the\n * watcher; `{ diff: true }` also emits a per-file `sandbox.file.diff` event. */\n fileEvents?: boolean | { diff?: boolean }\n}\n\n/** Context passed to `ensure()` by `withSandbox` (or advanced callers). */\nexport interface SandboxEnsureContext {\n threadId: string\n runId: string\n /** Persistence seam; falls back to an in-memory store when absent. */\n store?: SandboxInstanceStore\n /** Lock seam; falls back to an in-memory lock when absent. */\n locks?: LockStore\n tenant?: { userId?: string; orgId?: string }\n signal?: AbortSignal\n /** Harness adapter name (`grok-build`, `claude-code`, `codex`, `opencode`). Optional. */\n adapterName?: string\n}\n\nexport interface SandboxDefinition {\n readonly id: string\n readonly provider: SandboxProvider\n readonly workspace?: WorkspaceDefinition\n readonly policy?: SandboxPolicy\n readonly lifecycle?: SandboxLifecycle\n /** Sandbox-scoped file/lifecycle hooks. */\n readonly hooks?: SandboxHooks\n /** Watch the workspace for file events (default true). `false` disables the\n * watcher; `{ diff: true }` also emits a per-file `sandbox.file.diff` event. */\n readonly fileEvents?: boolean | { diff?: boolean }\n /** Compound instance key for a given run context. */\n key: (ctx: SandboxEnsureContext) => string\n /** Resume-or-create the sandbox for this thread/run. */\n ensure: (ctx: SandboxEnsureContext) => Promise<SandboxHandle>\n /** Tear down the sandbox recorded for this key. */\n destroy: (ctx: SandboxEnsureContext) => Promise<void>\n}\n\n/**\n * Parse a human-readable duration string into milliseconds.\n * Supports `'<n>h'` (hours) and `'<n>m'` (minutes).\n * Returns `undefined` when the input is undefined or the format is unrecognised.\n */\nfunction parseMaxAgeMs(value: string | undefined): number | undefined {\n if (value === undefined) return undefined\n const hourMatch = /^(\\d+)h$/.exec(value)\n if (hourMatch) return Number(hourMatch[1]) * 60 * 60 * 1000\n const minuteMatch = /^(\\d+)m$/.exec(value)\n if (minuteMatch) return Number(minuteMatch[1]) * 60 * 1000\n return undefined\n}\n\n/**\n * Bound for the unfenced teardown `destroy` call (see `destroy` below). Long\n * enough that a slow provider API still completes, short enough that a wedged\n * one cannot pin the process forever.\n */\nconst DESTROY_TIMEOUT_MS = 60 * 1000\n\n// Process-lifetime fallbacks shared across all definitions so concurrent\n// ensures for the same key serialize even without an injected store/lock.\nconst fallbackStore = new InMemorySandboxInstanceStore()\nconst fallbackLocks = new InMemoryLockStore()\n\n/**\n * Put workspace secrets onto a live handle. Resume and snapshot restore skip\n * bootstrap, so this is the only path that re-injects them after reconnect.\n * Create injects secrets via `provider.create({ env })`, but resume/restore\n * return a handle whose process env is empty unless we set it here. sbx in\n * particular has no Docker Env on resume, so this is the only way secrets\n * come back for that provider.\n */\nasync function applyWorkspaceSecrets(\n handle: SandboxHandle,\n workspace: WorkspaceDefinition | undefined,\n): Promise<void> {\n if (workspace?.secrets === undefined) return\n const resolved = resolveAllSecrets(workspace.secrets)\n if (Object.keys(resolved).length === 0) return\n await handle.env.set(resolved)\n}\n\nexport function defineSandbox(config: SandboxConfig): SandboxDefinition {\n const keyInputFor = (ctx: SandboxEnsureContext): SandboxKeyInput => ({\n threadId:\n config.lifecycle?.reuse === 'none'\n ? `${ctx.threadId}:${ctx.runId}`\n : ctx.threadId,\n sandboxId: config.id,\n providerName: config.provider.name,\n workspace: config.workspace,\n tenant: ctx.tenant,\n })\n\n const ensure = async (ctx: SandboxEnsureContext): Promise<SandboxHandle> => {\n const store = ctx.store ?? fallbackStore\n const locks = ctx.locks ?? fallbackLocks\n const key = computeSandboxKey(keyInputFor(ctx))\n const caps = config.provider.capabilities()\n\n return locks.withLock(`sandbox:${key}`, async () => {\n const effectiveSnapshot: SnapshotStrategy =\n config.lifecycle?.snapshot ?? (caps.snapshots ? 'after-setup' : 'none')\n const maxAgeMs = parseMaxAgeMs(config.lifecycle?.snapshotMaxAge)\n\n const existing = await store.get(key)\n if (existing) {\n // Check whether the record has exceeded snapshotMaxAge; if so,\n // discard and fall through to a fresh create.\n const tooOld =\n maxAgeMs !== undefined && Date.now() - existing.updatedAt > maxAgeMs\n\n if (!tooOld) {\n // 1) Try to reconnect to the still-running sandbox.\n const resumed = await config.provider.resume({\n id: existing.providerSandboxId,\n signal: ctx.signal,\n })\n if (resumed) {\n await applyWorkspaceSecrets(resumed, config.workspace)\n await store.upsert({\n ...existing,\n latestRunId: ctx.runId,\n updatedAt: Date.now(),\n })\n return resumed\n }\n // 2) Else restore from the latest snapshot, if supported.\n if (\n existing.latestSnapshotId &&\n caps.snapshots &&\n config.provider.restoreSnapshot\n ) {\n const restored = await config.provider.restoreSnapshot({\n snapshotId: existing.latestSnapshotId,\n workspace: config.workspace,\n policy: config.policy,\n env:\n config.workspace?.secrets !== undefined\n ? resolveAllSecrets(config.workspace.secrets)\n : undefined,\n signal: ctx.signal,\n })\n await applyWorkspaceSecrets(restored, config.workspace)\n await store.upsert({\n ...existing,\n providerSandboxId: restored.id,\n latestRunId: ctx.runId,\n updatedAt: Date.now(),\n })\n return restored\n }\n }\n // 3) Else fall through and re-create under the same identity\n // (capability-aware degradation for ephemeral-disk providers, or\n // snapshotMaxAge TTL exceeded).\n }\n\n const created = await config.provider.create({\n // Deterministic id so consumers can reconstruct the provider sandbox\n // address from run context (not just from the store record).\n id: key,\n workspace: config.workspace,\n policy: config.policy,\n env:\n config.workspace?.secrets !== undefined\n ? resolveAllSecrets(config.workspace.secrets)\n : undefined,\n signal: ctx.signal,\n adapterName: ctx.adapterName,\n })\n\n if (config.workspace) {\n try {\n await bootstrapWorkspace(created, config.workspace, {\n signal: ctx.signal,\n })\n } catch (error) {\n // Bootstrap failed after the sandbox was created but before it was\n // recorded — destroy the orphan so a failed/retried run doesn't leak\n // a (billed) sandbox, then surface the original error.\n await created.destroy().catch(() => {})\n throw error\n }\n }\n\n let latestSnapshotId: string | undefined\n if (\n effectiveSnapshot === 'after-setup' &&\n caps.snapshots &&\n created.snapshot\n ) {\n latestSnapshotId = (await created.snapshot('after-setup')).id\n }\n\n await store.upsert({\n key,\n provider: config.provider.name,\n providerSandboxId: created.id,\n latestSnapshotId,\n threadId: ctx.threadId,\n latestRunId: ctx.runId,\n updatedAt: Date.now(),\n })\n return created\n })\n }\n\n const destroy = async (ctx: SandboxEnsureContext): Promise<void> => {\n const store = ctx.store ?? fallbackStore\n const key = computeSandboxKey(keyInputFor(ctx))\n const existing = await store.get(key)\n if (!existing) return\n /*\n * TEARDOWN IS DELIBERATELY NOT FENCED BY `ctx.signal`.\n *\n * `destroy` runs on every teardown path INCLUDING the one caused by that\n * very signal aborting, so forwarding it hands the provider a signal that is\n * already aborted: a provider that honors it does nothing and returns\n * successfully, and `store.delete` below then removes the only pointer to a\n * live, billed sandbox. `SandboxInstanceStore` has no `list` (see the note\n * at the top of `reclaim.ts`), so that sandbox is unreachable from then on.\n *\n * Same reasoning as `close()` never being fenced by the run claim (see\n * `fenceDurability` in `claim.ts`): cleanup must outlive whatever cancelled\n * the work. A fresh controller with its own bounded timeout keeps the call\n * from hanging forever without letting the caller's abort cancel it.\n */\n const teardown = new AbortController()\n const timer = setTimeout(() => teardown.abort(), DESTROY_TIMEOUT_MS)\n try {\n await config.provider.destroy({\n id: existing.providerSandboxId,\n signal: teardown.signal,\n })\n } finally {\n clearTimeout(timer)\n }\n await store.delete(key)\n }\n\n return {\n id: config.id,\n provider: config.provider,\n workspace: config.workspace,\n policy: config.policy,\n lifecycle: config.lifecycle,\n hooks: config.hooks,\n fileEvents: config.fileEvents,\n key: (ctx) => computeSandboxKey(keyInputFor(ctx)),\n ensure,\n destroy,\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;AA0GA,SAAS,cAAc,OAA+C;CACpE,IAAI,UAAU,KAAA,GAAW,OAAO,KAAA;CAChC,MAAM,YAAY,WAAW,KAAK,KAAK;CACvC,IAAI,WAAW,OAAO,OAAO,UAAU,EAAE,IAAI,KAAK,KAAK;CACvD,MAAM,cAAc,WAAW,KAAK,KAAK;CACzC,IAAI,aAAa,OAAO,OAAO,YAAY,EAAE,IAAI,KAAK;AAExD;;;;;;AAOA,IAAM,qBAAqB;AAI3B,IAAM,gBAAgB,IAAI,6BAA6B;AACvD,IAAM,gBAAgB,IAAI,kBAAkB;;;;;;;;;AAU5C,eAAe,sBACb,QACA,WACe;CACf,IAAI,WAAW,YAAY,KAAA,GAAW;CACtC,MAAM,WAAW,kBAAkB,UAAU,OAAO;CACpD,IAAI,OAAO,KAAK,QAAQ,CAAC,CAAC,WAAW,GAAG;CACxC,MAAM,OAAO,IAAI,IAAI,QAAQ;AAC/B;AAEA,SAAgB,cAAc,QAA0C;CACtE,MAAM,eAAe,SAAgD;EACnE,UACE,OAAO,WAAW,UAAU,SACxB,GAAG,IAAI,SAAS,GAAG,IAAI,UACvB,IAAI;EACV,WAAW,OAAO;EAClB,cAAc,OAAO,SAAS;EAC9B,WAAW,OAAO;EAClB,QAAQ,IAAI;CACd;CAEA,MAAM,SAAS,OAAO,QAAsD;EAC1E,MAAM,QAAQ,IAAI,SAAS;EAC3B,MAAM,QAAQ,IAAI,SAAS;EAC3B,MAAM,MAAM,kBAAkB,YAAY,GAAG,CAAC;EAC9C,MAAM,OAAO,OAAO,SAAS,aAAa;EAE1C,OAAO,MAAM,SAAS,WAAW,OAAO,YAAY;GAClD,MAAM,oBACJ,OAAO,WAAW,aAAa,KAAK,YAAY,gBAAgB;GAClE,MAAM,WAAW,cAAc,OAAO,WAAW,cAAc;GAE/D,MAAM,WAAW,MAAM,MAAM,IAAI,GAAG;GACpC,IAAI,UAME;QAAA,EAFF,aAAa,KAAA,KAAa,KAAK,IAAI,IAAI,SAAS,YAAY,WAEjD;KAEX,MAAM,UAAU,MAAM,OAAO,SAAS,OAAO;MAC3C,IAAI,SAAS;MACb,QAAQ,IAAI;KACd,CAAC;KACD,IAAI,SAAS;MACX,MAAM,sBAAsB,SAAS,OAAO,SAAS;MACrD,MAAM,MAAM,OAAO;OACjB,GAAG;OACH,aAAa,IAAI;OACjB,WAAW,KAAK,IAAI;MACtB,CAAC;MACD,OAAO;KACT;KAEA,IACE,SAAS,oBACT,KAAK,aACL,OAAO,SAAS,iBAChB;MACA,MAAM,WAAW,MAAM,OAAO,SAAS,gBAAgB;OACrD,YAAY,SAAS;OACrB,WAAW,OAAO;OAClB,QAAQ,OAAO;OACf,KACE,OAAO,WAAW,YAAY,KAAA,IAC1B,kBAAkB,OAAO,UAAU,OAAO,IAC1C,KAAA;OACN,QAAQ,IAAI;MACd,CAAC;MACD,MAAM,sBAAsB,UAAU,OAAO,SAAS;MACtD,MAAM,MAAM,OAAO;OACjB,GAAG;OACH,mBAAmB,SAAS;OAC5B,aAAa,IAAI;OACjB,WAAW,KAAK,IAAI;MACtB,CAAC;MACD,OAAO;KACT;IACF;;GAMF,MAAM,UAAU,MAAM,OAAO,SAAS,OAAO;IAG3C,IAAI;IACJ,WAAW,OAAO;IAClB,QAAQ,OAAO;IACf,KACE,OAAO,WAAW,YAAY,KAAA,IAC1B,kBAAkB,OAAO,UAAU,OAAO,IAC1C,KAAA;IACN,QAAQ,IAAI;IACZ,aAAa,IAAI;GACnB,CAAC;GAED,IAAI,OAAO,WACT,IAAI;IACF,MAAM,mBAAmB,SAAS,OAAO,WAAW,EAClD,QAAQ,IAAI,OACd,CAAC;GACH,SAAS,OAAO;IAId,MAAM,QAAQ,QAAQ,CAAC,CAAC,YAAY,CAAC,CAAC;IACtC,MAAM;GACR;GAGF,IAAI;GACJ,IACE,sBAAsB,iBACtB,KAAK,aACL,QAAQ,UAER,oBAAoB,MAAM,QAAQ,SAAS,aAAa,EAAA,CAAG;GAG7D,MAAM,MAAM,OAAO;IACjB;IACA,UAAU,OAAO,SAAS;IAC1B,mBAAmB,QAAQ;IAC3B;IACA,UAAU,IAAI;IACd,aAAa,IAAI;IACjB,WAAW,KAAK,IAAI;GACtB,CAAC;GACD,OAAO;EACT,CAAC;CACH;CAEA,MAAM,UAAU,OAAO,QAA6C;EAClE,MAAM,QAAQ,IAAI,SAAS;EAC3B,MAAM,MAAM,kBAAkB,YAAY,GAAG,CAAC;EAC9C,MAAM,WAAW,MAAM,MAAM,IAAI,GAAG;EACpC,IAAI,CAAC,UAAU;EAgBf,MAAM,WAAW,IAAI,gBAAgB;EACrC,MAAM,QAAQ,iBAAiB,SAAS,MAAM,GAAG,kBAAkB;EACnE,IAAI;GACF,MAAM,OAAO,SAAS,QAAQ;IAC5B,IAAI,SAAS;IACb,QAAQ,SAAS;GACnB,CAAC;EACH,UAAU;GACR,aAAa,KAAK;EACpB;EACA,MAAM,MAAM,OAAO,GAAG;CACxB;CAEA,OAAO;EACL,IAAI,OAAO;EACX,UAAU,OAAO;EACjB,WAAW,OAAO;EAClB,QAAQ,OAAO;EACf,WAAW,OAAO;EAClB,OAAO,OAAO;EACd,YAAY,OAAO;EACnB,MAAM,QAAQ,kBAAkB,YAAY,GAAG,CAAC;EAChD;EACA;CACF;AACF"}
|
package/dist/esm/shell.js
CHANGED
|
@@ -18,7 +18,7 @@ function parseExports(output) {
|
|
|
18
18
|
return env;
|
|
19
19
|
}
|
|
20
20
|
/** Default {@link BootstrapShellOptions.commandTimeoutMs} — 30 minutes. */
|
|
21
|
-
var DEFAULT_COMMAND_TIMEOUT_MS =
|
|
21
|
+
var DEFAULT_COMMAND_TIMEOUT_MS = 18e5;
|
|
22
22
|
/** Race marker for the per-command deadline. A symbol cannot collide with a
|
|
23
23
|
* literal stdout line (a line of text `'timeout'` would). */
|
|
24
24
|
var TIMED_OUT = Symbol("bootstrap-shell-timeout");
|
|
@@ -170,9 +170,10 @@ function createExecBootstrapShell(handle, opts = {}) {
|
|
|
170
170
|
cmdOut.push(line);
|
|
171
171
|
} else if (phase === "await-cwd") {
|
|
172
172
|
if (line === `${sentinel}_CWD`) phase = "cwd";
|
|
173
|
-
} else if (phase === "cwd")
|
|
174
|
-
|
|
175
|
-
|
|
173
|
+
} else if (phase === "cwd") {
|
|
174
|
+
if (line === `${sentinel}_ENV`) phase = "env";
|
|
175
|
+
else cwdLines.push(line);
|
|
176
|
+
} else envLines.push(line);
|
|
176
177
|
const newCwd = cwdLines.map((l) => l.trim()).filter(Boolean).pop();
|
|
177
178
|
if (newCwd) cwd = newCwd;
|
|
178
179
|
const newEnv = parseExports(envLines.join("\n"));
|