@tanstack/ai-sandbox 0.2.3 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/agents-file.js +53 -34
- package/dist/esm/agents-file.js.map +1 -1
- package/dist/esm/align.d.ts +121 -0
- package/dist/esm/align.js +197 -0
- package/dist/esm/align.js.map +1 -0
- package/dist/esm/approvals.js +63 -29
- package/dist/esm/approvals.js.map +1 -1
- package/dist/esm/attach-preflight.d.ts +85 -0
- package/dist/esm/attach-preflight.js +189 -0
- package/dist/esm/attach-preflight.js.map +1 -0
- package/dist/esm/bootstrap.js +103 -117
- package/dist/esm/bootstrap.js.map +1 -1
- package/dist/esm/bridge-events.js +96 -71
- package/dist/esm/bridge-events.js.map +1 -1
- package/dist/esm/capabilities.d.ts +0 -5
- package/dist/esm/capabilities.js +32 -28
- package/dist/esm/capabilities.js.map +1 -1
- package/dist/esm/chunk-identity.d.ts +52 -0
- package/dist/esm/chunk-identity.js +102 -0
- package/dist/esm/chunk-identity.js.map +1 -0
- package/dist/esm/claim.d.ts +187 -0
- package/dist/esm/claim.js +349 -0
- package/dist/esm/claim.js.map +1 -0
- package/dist/esm/contracts.d.ts +13 -0
- package/dist/esm/driver.d.ts +83 -0
- package/dist/esm/driver.js +138 -0
- package/dist/esm/driver.js.map +1 -0
- package/dist/esm/durability.d.ts +263 -0
- package/dist/esm/durability.js +230 -0
- package/dist/esm/durability.js.map +1 -0
- package/dist/esm/errors.js +28 -24
- package/dist/esm/errors.js.map +1 -1
- package/dist/esm/file-diff.js +151 -135
- package/dist/esm/file-diff.js.map +1 -1
- package/dist/esm/git-exec.js +51 -62
- package/dist/esm/git-exec.js.map +1 -1
- package/dist/esm/harness-cwd.js +24 -19
- package/dist/esm/harness-cwd.js.map +1 -1
- package/dist/esm/index.d.ts +30 -8
- package/dist/esm/index.js +23 -91
- package/dist/esm/instance-store.d.ts +88 -0
- package/dist/esm/instance-store.js +67 -0
- package/dist/esm/instance-store.js.map +1 -0
- package/dist/esm/journal-bytes.d.ts +67 -0
- package/dist/esm/journal-bytes.js +110 -0
- package/dist/esm/journal-bytes.js.map +1 -0
- package/dist/esm/journal-reader.d.ts +66 -0
- package/dist/esm/journal-reader.js +228 -0
- package/dist/esm/journal-reader.js.map +1 -0
- package/dist/esm/journal-sweep.d.ts +113 -0
- package/dist/esm/journal-sweep.js +309 -0
- package/dist/esm/journal-sweep.js.map +1 -0
- package/dist/esm/journal.d.ts +542 -0
- package/dist/esm/journal.js +679 -0
- package/dist/esm/journal.js.map +1 -0
- package/dist/esm/key.js +36 -33
- package/dist/esm/key.js.map +1 -1
- package/dist/esm/middleware.d.ts +50 -2
- package/dist/esm/middleware.js +335 -208
- package/dist/esm/middleware.js.map +1 -1
- package/dist/esm/ngrok.js +75 -49
- package/dist/esm/ngrok.js.map +1 -1
- package/dist/esm/policy.js +43 -34
- package/dist/esm/policy.js.map +1 -1
- package/dist/esm/projection.js +16 -8
- package/dist/esm/projection.js.map +1 -1
- package/dist/esm/reap.d.ts +238 -0
- package/dist/esm/reap.js +355 -0
- package/dist/esm/reap.js.map +1 -0
- package/dist/esm/reclaim.d.ts +84 -0
- package/dist/esm/reclaim.js +106 -0
- package/dist/esm/reclaim.js.map +1 -0
- package/dist/esm/remote-tools.js +73 -62
- package/dist/esm/remote-tools.js.map +1 -1
- package/dist/esm/run.d.ts +93 -25
- package/dist/esm/run.js +274 -79
- package/dist/esm/run.js.map +1 -1
- package/dist/esm/runner.d.ts +119 -2
- package/dist/esm/runner.js +270 -51
- package/dist/esm/runner.js.map +1 -1
- package/dist/esm/sandbox.d.ts +3 -2
- package/dist/esm/sandbox.js +139 -123
- package/dist/esm/sandbox.js.map +1 -1
- package/dist/esm/secrets.js +39 -47
- package/dist/esm/secrets.js.map +1 -1
- package/dist/esm/setup-plan.js +22 -14
- package/dist/esm/setup-plan.js.map +1 -1
- package/dist/esm/shell.d.ts +8 -0
- package/dist/esm/shell.js +197 -158
- package/dist/esm/shell.js.map +1 -1
- package/dist/esm/testkit/conformance.d.ts +16 -0
- package/dist/esm/testkit/conformance.js +97 -0
- package/dist/esm/testkit/conformance.js.map +1 -0
- package/dist/esm/testkit/durable-run-fields-conformance.d.ts +4 -0
- package/dist/esm/testkit/durable-run-fields-conformance.js +95 -0
- package/dist/esm/testkit/durable-run-fields-conformance.js.map +1 -0
- package/dist/esm/testkit/journal-conformance.d.ts +51 -0
- package/dist/esm/testkit/journal-conformance.js +378 -0
- package/dist/esm/testkit/journal-conformance.js.map +1 -0
- package/dist/esm/testkit/reaper-conformance.d.ts +37 -0
- package/dist/esm/testkit/reaper-conformance.js +847 -0
- package/dist/esm/testkit/reaper-conformance.js.map +1 -0
- package/dist/esm/testkit/shell-spawn.d.ts +2 -0
- package/dist/esm/testkit/shell-spawn.js +60 -0
- package/dist/esm/testkit/shell-spawn.js.map +1 -0
- package/dist/esm/testkit/takeover-conformance.d.ts +24 -0
- package/dist/esm/testkit/takeover-conformance.js +685 -0
- package/dist/esm/testkit/takeover-conformance.js.map +1 -0
- package/dist/esm/tool-bridge.js +227 -180
- package/dist/esm/tool-bridge.js.map +1 -1
- package/dist/esm/tool-history.d.ts +62 -0
- package/dist/esm/tool-history.js +171 -0
- package/dist/esm/tool-history.js.map +1 -0
- package/dist/esm/watch.js +310 -236
- package/dist/esm/watch.js.map +1 -1
- package/dist/esm/workspace.d.ts +1 -1
- package/dist/esm/workspace.js +49 -28
- package/dist/esm/workspace.js.map +1 -1
- package/package.json +16 -6
- package/skills/ai-sandbox/SKILL.md +658 -20
- package/src/align.ts +297 -0
- package/src/attach-preflight.ts +292 -0
- package/src/capabilities.ts +4 -13
- package/src/chunk-identity.ts +154 -0
- package/src/claim.ts +479 -0
- package/src/contracts.ts +13 -0
- package/src/driver.ts +205 -0
- package/src/durability.ts +380 -0
- package/src/index.ts +212 -27
- package/src/instance-store.ts +122 -0
- package/src/journal-bytes.ts +136 -0
- package/src/journal-reader.ts +359 -0
- package/src/journal-sweep.ts +406 -0
- package/src/journal.ts +875 -0
- package/src/middleware.ts +470 -30
- package/src/reap.ts +723 -0
- package/src/reclaim.ts +191 -0
- package/src/run.ts +365 -75
- package/src/runner.ts +347 -3
- package/src/sandbox.ts +38 -8
- package/src/shell.ts +106 -38
- package/src/testkit/conformance.ts +117 -0
- package/src/testkit/durable-run-fields-conformance.ts +147 -0
- package/src/testkit/journal-conformance.ts +676 -0
- package/src/testkit/reaper-conformance.ts +1201 -0
- package/src/testkit/shell-spawn.ts +67 -0
- package/src/testkit/takeover-conformance.ts +1040 -0
- package/src/tool-history.ts +245 -0
- package/src/workspace.ts +1 -1
- package/dist/esm/index.js.map +0 -1
- package/dist/esm/run-log.d.ts +0 -81
- package/dist/esm/run-log.js +0 -107
- package/dist/esm/run-log.js.map +0 -1
- package/dist/esm/store.d.ts +0 -53
- package/dist/esm/store.js +0 -34
- package/dist/esm/store.js.map +0 -1
- package/src/run-log.ts +0 -224
- package/src/store.ts +0 -83
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"reap.js","names":[],"sources":["../../src/reap.ts"],"sourcesContent":["/**\n * The sweep `RunStore.listReclaimable` was always missing a consumer for: take a\n * detached run whose viewer never came back, save its transcript, terminalize its\n * record, and tear its sandbox down.\n *\n * THE ONE RULE THAT SHAPES EVERYTHING HERE: **never drive a run to find out\n * whether it finished.**\n *\n * The obvious design — hand the run to `pipeToRunLog` under a short\n * `runBudgetMs` and see whether it terminalizes — was measured and is broken.\n * `pipeToRunLog` is total by construction: it ALWAYS writes a terminal status and\n * ALWAYS calls `durability.close()`. Against a run that has not finished, all\n * three producer shapes are destructive:\n *\n * | producer's reaction to the budget signal | stored status | `close()` |\n * | ---------------------------------------- | ------------- | --------- |\n * | ignores it and keeps producing | `aborted` | called |\n * | returns on abort (the realistic `drive`) | `aborted` | called |\n * | throws an AbortError | `failed` | called |\n *\n * The middle row USED to read `completed`, which was the fatal one: a signal-aware\n * producer exits its loop NORMALLY, and `pipeToRunLog` only checked its signal\n * per chunk, so a healthy mid-flight run was recorded as `'completed'` with a\n * `finishedAt` — a false transcript. That gap is fixed (`run.ts` re-checks the\n * signal after the loop), so the status is now honest on all three rows. The rule\n * above is UNCHANGED, because the status was never the whole harm: every row\n * writes a terminal record and closes a log that commit `5a1f821c9` deliberately\n * leaves OPEN for takeover (ending every attached client's stream), and a terminal\n * record drops out of `listReclaimable` forever, so TTL expiry can never reclaim\n * that run's sandbox. A cost leak with no recovery path. There is therefore no\n * \"still running\" outcome in {@link ReapRunOutcome}: it is unreachable by\n * construction, not merely unlikely.\n *\n * So sentinel-reached is detected OUT OF BAND, through the in-sandbox journal\n * ({@link probeRunExit}), and `pipeToRunLog` is entered only for a run already\n * KNOWN to have finished, or for one whose TTL has expired (terminal either way).\n * On the FINALIZATION path `runBudgetMs` therefore degrades from a load-bearing\n * mechanism into a safety net whose expiry is a genuine anomaly — see\n * `'budget-exceeded'`. On the EXPIRY path it stays load-bearing: nothing polls the\n * cancel this module records, so the budget is what ends the drive of an expired\n * run whose agent is still producing, and its expiry there is the designed path.\n *\n * WHY THE PROBE IS INJECTED (`ReapOptions.hasFinished`) rather than resolved\n * here, exactly like `ReapOptions.reclaim`:\n *\n * - It cannot read `durability.snapshot()`. After a detach nothing appends to the\n * delivery log — the host that would have appended is the host that left — so\n * the log is frozen at the last delivered chunk while the JOURNAL keeps\n * growing. The log can only ever say \"no news\".\n * - It cannot resolve a `SandboxHandle` either. `SandboxInstanceStore` is\n * `get`/`upsert`/`delete` with no `list` (see `reclaim.ts` for why that is\n * deliberate), and only the application maps a `sandboxKey` to a live handle.\n *\n * NEVER REJECTS. This runs from a cron, an `alarm()`, or a `waitUntil` with\n * nobody to catch it, so every per-run failure is logged and folded into\n * {@link ReapResult} rather than escaping.\n *\n * NEVER CLEARS `detachedSince`. That field is what the reaper SELECTS on, and\n * `packages/ai/src/stream-to-response.ts`'s `startRunDriver` clears it because a\n * real viewer stopping the TTL clock is the opposite job. Its comment there names\n * borrowing that path \"the single most likely bug in this phase\"; clearing the\n * marker would reset the TTL on every sweep and a detached run would never\n * expire.\n */\nimport { isTerminalRunStatus, requestRunCancel } from '@tanstack/ai'\nimport {\n DEFAULT_FENCE_QUIET_MS,\n RunClaimLostError,\n // Thrown, not merely caught: the expiry re-derivation under the lock refuses\n // its own claim when the run's viewer has come back.\n RunClaimNotAcquiredError,\n awaitLogQuiescence,\n fenceDurability,\n fenceRunStore,\n withRunClaim,\n} from './claim'\nimport { pipeToRunLog } from './run'\nimport {\n journalExitProbeCommand,\n journalPaths,\n parseJournalExit,\n} from './journal'\nimport { decodeBase64Stream } from './journal-bytes'\nimport type { SandboxHandle } from './contracts'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { LockStore } from '@tanstack/ai/locks'\nimport type {\n RunRecord,\n RunStatus,\n RunStore,\n StreamChunk,\n StreamDurability,\n} from '@tanstack/ai'\n\n/**\n * Safety net for a single run's drive. Not the mechanism that decides whether a\n * run finished — see the module doc for why that design was rejected — so this is\n * generous rather than tight: it only has to stop a drive that has genuinely\n * wedged on a run the journal already said was over.\n *\n * On the expiry path it is not merely a net: it is what stops a still-producing\n * agent, since nothing polls the cancel recorded before that drive. A caller that\n * expires live agents may want a tighter value there than a finalization replay\n * needs.\n */\nexport const DEFAULT_RUN_BUDGET_MS = 30_000\n\n/**\n * Runs one sweep will touch. A cron invocation is bounded (a Worker's CPU\n * budget, a Lambda timeout), and an unbounded sweep over a backlog of thousands\n * would be killed mid-run rather than finishing 25 and returning; the next tick\n * takes the next batch.\n */\nexport const DEFAULT_MAX_RUNS = 25\n\n/** Journal tail bytes {@link probeRunExit} reads. The sentinel is the last line. */\nexport const DEFAULT_EXIT_PROBE_BYTES = 4096\n\n/**\n * What the out-of-band probe learned about a detached run's agent.\n *\n * THREE ARMS, not a boolean, because \"could not tell\" must not be\n * indistinguishable from \"still working\": both leave the run alone, but only one\n * of them is a condition an operator should see. A two-valued probe would also\n * invite the caller to treat a provider `exec` failure as \"finished\" and drive a\n * live run — the exact defect this module exists to prevent.\n */\nexport type RunExitProbe =\n /** The `{\"__exit\":N}` sentinel is in the journal. The agent is over. */\n | { state: 'finished'; exitCode: number }\n /** No sentinel. The agent is mid-flight (or never started). LEAVE IT ALONE. */\n | { state: 'producing' }\n /** The probe could not answer — no sandbox, `exec` rejected, frame undecodable. */\n | { state: 'unknown'; error?: unknown }\n\n/** What one sweep did to one run. */\nexport type ReapRunOutcome =\n /**\n * The probe saw `{\"__exit\":N}`, the run was driven to a terminal status, and\n * its transcript is saved. The happy path.\n */\n | 'finalized'\n /**\n * Past `detachedRunTtlMs`. Cancelled first, then driven to terminal. The probe\n * is skipped: the outcome is terminal whether the agent finished or not.\n *\n * Reported even when {@link ReapOptions.runBudgetMs} is what ended the drive —\n * on this path that is the mechanism rather than an anomaly, so `'expired'` is\n * the truthful outcome. The run's own `status` distinguishes the two shapes:\n * an agent that had already finished replays to `'completed'`, while one still\n * producing when the budget fired is `'aborted'`.\n */\n | 'expired'\n /**\n * Still producing. `pipeToRunLog` was NEVER entered — nothing appended, no\n * terminal record written, `close()` not called, `detachedSince` untouched.\n */\n | 'producing'\n /** The probe could not answer. Left exactly as untouched as `'producing'`. */\n | 'unknown'\n /**\n * ANOMALY. The drive outran {@link ReapOptions.runBudgetMs} on a run the\n * journal already said was finished. The record IS terminal and the log IS\n * closed (`pipeToRunLog` guarantees both), so this is a diagnostic, not a leak\n * — but a finished run that would not replay in 30s means the journal read, the\n * translation, or the log is misbehaving.\n *\n * FINALIZATION ONLY. An expired run that outran its budget reports `'expired'`:\n * there was no probe on that path and the agent may legitimately still have been\n * producing, so the budget firing is the designed stop, not a misbehaving replay.\n */\n | 'budget-exceeded'\n /**\n * Another host holds the claim, or held it and superseded us mid-drive. Normal:\n * a real viewer attaching mid-sweep is exactly this. Also covers a run that\n * reached terminal in another host's hands between the listing and the claim.\n */\n | 'not-claimed'\n /**\n * The transcript IS saved and the record IS terminal — only\n * {@link ReapOptions.reclaim} threw, so the sandbox is still up.\n *\n * A DISTINCT outcome rather than `'failed'`, because the two need opposite\n * operator responses and `'failed'` cannot express this one: it carries no\n * `status` and no `exitCode`, so \"transcript saved, sandbox NOT reclaimed\"\n * read identically to \"the sweep failed and the run was never finalized\".\n *\n * NOT RETRYABLE BY THE SWEEP. The record is terminal by now, so the run has\n * left `listReclaimable` for good; the sandbox leaks until something else\n * tears it down. This entry, with its `error`, is the only notice of that.\n *\n * OVERWRITES `'budget-exceeded'` when both happened, because the leak is what\n * needs acting on — {@link ReapRunEntry.terminalizedAnyway} is what preserves\n * the budget half of that pair.\n *\n * `sandboxReclaimer` REJECTS on its `'destroy-failed'` arm precisely so this\n * outcome is reachable through the shipped reclaimer and not only through a\n * custom one; see `SandboxReclaimFailedError` in `reclaim.ts`.\n */\n | 'reclaim-failed'\n /** Something threw. Logged, recorded here, and the sweep continued. */\n | 'failed'\n\n/** One run's line in the sweep summary. */\nexport interface ReapRunEntry {\n runId: string\n outcome: ReapRunOutcome\n /** The run's status after the sweep, when the run was driven. */\n status?: RunStatus\n /** The agent's exit code, when the probe read one. */\n exitCode?: number\n /**\n * THE BUDGET ANOMALY MARKER, and the only field whose mere PRESENCE carries a\n * fact: it is set if and only if the drive outran\n * {@link ReapOptions.runBudgetMs} on the finalization path — the condition\n * `'budget-exceeded'` names. Its value is whether the record nonetheless\n * reached a terminal status, practically always `true` since `pipeToRunLog` is\n * total; it is reported rather than assumed so an operator does not have to\n * infer it.\n *\n * SURVIVES A FAILED RECLAIM. `reclaim` runs after the outcome is classified\n * and overwrites it with `'reclaim-failed'`, which is the more urgent fact (a\n * leaked sandbox nothing will retry) and so wins the single `outcome` slot.\n * This field is therefore what keeps the budget anomaly on the entry: an\n * operator seeing `'reclaim-failed'` WITH `terminalizedAnyway` present is\n * looking at a run that blew its budget and then leaked, and needs both halves.\n */\n terminalizedAnyway?: boolean\n error?: unknown\n}\n\nexport interface ReapResult {\n /** Runs in this batch — i.e. after the {@link ReapOptions.maxRuns} cap. */\n considered: number\n /** Runs {@link ReapOptions.hasFinished} was actually called for. */\n probed: number\n outcomes: Record<ReapRunOutcome, number>\n runs: Array<ReapRunEntry>\n}\n\nexport interface ReapOptions<TOffset extends string = string> {\n runs: RunStore\n locks: LockStore\n /**\n * Per-run event log factory, same shape `RunDeps.durability` takes.\n *\n * Generic in the offset type, defaulted to `string` so an existing call site\n * needs no change — see {@link SandboxRunDriverOptions.durability} for why\n * hardcoding the default locked out branded-cursor backends.\n */\n durability: (runId: string) => StreamDurability<TOffset>\n /**\n * The out-of-band \"did the agent reach its sentinel?\" probe. INJECTED, because\n * neither the delivery log nor this package can answer it — see the module doc.\n * {@link probeRunExit} is the implementation an application wires in once it has\n * resolved the run's `SandboxHandle`.\n */\n hasFinished: (record: RunRecord) => Promise<RunExitProbe>\n /** Produce the run's remaining events. Called only once the claim is held. */\n drive: (input: {\n runId: string\n threadId: string\n signal: AbortSignal\n }) => AsyncIterable<StreamChunk>\n /** Sweep clock, passed rather than read so a sweep is reproducible. */\n now: number\n /** Detached-run TTL; `detachedSince <= now - ttl` expires, INCLUSIVELY. */\n detachedRunTtlMs: number\n /** Safety net per drive. Defaults to {@link DEFAULT_RUN_BUDGET_MS}. */\n runBudgetMs?: number\n /** Batch cap. Defaults to {@link DEFAULT_MAX_RUNS}. */\n maxRuns?: number\n /** Quiescence window; defaults to `DEFAULT_FENCE_QUIET_MS`. */\n fenceQuietMs?: number\n /**\n * Tear the run's sandbox down. Called ONLY after the run reached a terminal\n * status, and with the ORIGINALLY LISTED record — see {@link reapDetachedRuns}.\n * `sandboxReclaimer` in `reclaim.ts` is the ready-made implementation.\n */\n reclaim?: (record: RunRecord) => Promise<void>\n logger?: InternalLogger\n}\n\nasync function* singleValue(value: string): AsyncIterable<string> {\n yield value\n}\n\n/** Decode the base64 frame `journalExitProbeCommand` emits. */\nasync function decodeFrame(stdout: string): Promise<string> {\n const decoder = new TextDecoder()\n let text = ''\n for await (const bytes of decodeBase64Stream(singleValue(stdout))) {\n text += decoder.decode(bytes, { stream: true })\n }\n return text + decoder.decode()\n}\n\n/**\n * Read the END of a run's journal and answer whether the agent reached its\n * `{\"__exit\":N}` sentinel. Read-only: no append, no record write, no `close()`.\n *\n * This is the whole reason the reaper is safe. It is the ONLY way to learn that a\n * detached run is over without driving it, because the delivery log stops growing\n * the moment the viewer leaves while the journal does not.\n *\n * ANY failure answers `'unknown'`, never `'finished'`: the caller drives a run it\n * is told finished, so a provider `exec` that rejected, a sandbox that is gone, or\n * a frame the provider truncated must never be read as \"the agent exited\".\n *\n * An EMPTY tail answers `'producing'` — the fail-safe direction. A journal that\n * does not exist yet is indistinguishable here from one with no sentinel, and both\n * mean \"do not touch this run\".\n */\nexport async function probeRunExit(input: {\n handle: SandboxHandle\n runId: string\n /** Journal directory; defaults to `DEFAULT_JOURNAL_DIR`, as `journalPaths` does. */\n dir?: string\n /** Tail bytes to read. Defaults to {@link DEFAULT_EXIT_PROBE_BYTES}. */\n maxBytes?: number\n}): Promise<RunExitProbe> {\n try {\n const paths = journalPaths(input.runId, input.dir)\n const result = await input.handle.process.exec(\n journalExitProbeCommand(\n paths,\n input.maxBytes ?? DEFAULT_EXIT_PROBE_BYTES,\n ),\n )\n // `paths` supplies the per-run sentinel nonce: without it a mid-flight\n // agent that printed any JSON object carrying `__exit` would read as\n // `'finished'` here, and the caller would drive and reclaim a LIVE run.\n const exitCode = parseJournalExit(await decodeFrame(result.stdout), paths)\n return exitCode === null\n ? { state: 'producing' }\n : { state: 'finished', exitCode }\n } catch (error) {\n return { state: 'unknown', error }\n }\n}\n\n/** Every outcome key present at zero, so a consumer can read any of them. */\nfunction emptyOutcomes(): Record<ReapRunOutcome, number> {\n return {\n finalized: 0,\n expired: 0,\n producing: 0,\n unknown: 0,\n 'budget-exceeded': 0,\n 'not-claimed': 0,\n 'reclaim-failed': 0,\n failed: 0,\n }\n}\n\n/**\n * Report through a consumer-supplied logger without letting it break the sweep.\n * Mirrors `run.ts`'s `safeLog`: this module's totality must not be defeated by a\n * sink that cannot serialize a thrown value.\n */\nfunction safeLog(\n logger: InternalLogger | undefined,\n level: 'errors' | 'sandbox',\n message: string,\n context: Record<string, unknown>,\n): void {\n try {\n if (level === 'errors') logger?.errors(message, context)\n else logger?.sandbox(message, context)\n } catch {\n // Intentionally empty: there is no second channel to report on.\n }\n}\n\n/** Resolved-once settings shared by every run in one sweep. */\ninterface ReapContext<TOffset extends string = string> {\n options: ReapOptions<TOffset>\n runBudgetMs: number\n fenceQuietMs: number\n /** Inclusive expiry cutoff: `detachedSince <= cutoff` is expired. */\n cutoff: number\n}\n\n/** Whether a thrown value means \"we do not own this run\", which is normal. */\nfunction isClaimRefusal(error: unknown): boolean {\n return (\n error instanceof RunClaimNotAcquiredError ||\n error instanceof RunClaimLostError\n )\n}\n\n/**\n * Sweep ONE run. Never rejects: the caller folds the returned entry into the\n * summary and moves on.\n *\n * The ORDER of the steps below is the contract, not an implementation detail:\n *\n * 1. **Classify expiry first**, because an expired run needs no probe — its\n * outcome is terminal whether or not the agent finished, so a probe would only\n * add a provider round-trip and a way to fail.\n * 2. **Otherwise probe BEFORE touching anything.** `'producing'` and `'unknown'`\n * return here, having made no claim, no append, no record write, and no\n * `close()`. Driving past this point is the whole defect described in the\n * module doc.\n * 3. Claim, so two hosts never drive one run.\n * 4. **Re-derive expiry from a record read INSIDE the lock**, and only then\n * record the cancel. The listed record is stale by the time the claim is\n * held, and the cancel is sticky.\n * 5. Quiesce, so a predecessor still writing is observed rather than raced.\n * 6. **Arm the run budget**, so it bounds the drive rather than the queue the\n * two steps above stood in.\n * 7. Pipe with BOTH authoritative seams fenced, mirroring `driver.ts`.\n * 8. Reclaim, and ONLY once the record actually reached terminal.\n */\nasync function reapOne<TOffset extends string>(\n record: RunRecord,\n ctx: ReapContext<TOffset>,\n counters: { probed: number },\n): Promise<ReapRunEntry> {\n const { runs, locks, logger } = ctx.options\n const { runId, threadId } = record\n\n try {\n // INCLUSIVE, exactly as `RunStore.listReclaimable` documents its own cutoff:\n // a run detached at precisely `now - ttlMs` IS expired. The two must agree,\n // or a run would be listed as reclaimable and then classified as fresh on\n // every single sweep, forever.\n const expired =\n record.detachedSince !== undefined && record.detachedSince <= ctx.cutoff\n\n let exitCode: number | undefined\n if (!expired) {\n counters.probed += 1\n const probe = await ctx.options.hasFinished(record)\n if (probe.state !== 'finished') {\n // THE LEAVE-ALONE PATH. Deliberately returns before `withRunClaim`, so\n // not even `driverEpoch` moves — and above all `detachedSince` is left\n // exactly as it was, since it is both this run's TTL evidence and the\n // field the next sweep selects on.\n safeLog(logger, 'sandbox', `reap: leaving run ${runId} alone`, {\n runId,\n state: probe.state,\n ...(probe.state === 'unknown' && probe.error !== undefined\n ? { error: probe.error }\n : {}),\n })\n return {\n runId,\n outcome: probe.state,\n ...(probe.state === 'unknown' && probe.error !== undefined\n ? { error: probe.error }\n : {}),\n }\n }\n exitCode = probe.exitCode\n }\n\n // Armed INSIDE the claim, below. Read after it for the outcome, so it is\n // hoisted here rather than declared in the callback.\n let budget: AbortSignal | undefined\n const final = await withRunClaim(\n {\n runs,\n locks,\n runId,\n fenceQuietMs: ctx.fenceQuietMs,\n ...(logger === undefined ? {} : { logger }),\n },\n async (claim) => {\n if (expired) {\n // RE-DERIVED FROM A RECORD READ INSIDE THE LOCK, never from the listed\n // one. `stream-to-response.ts`'s `startRunDriver` CLEARS\n // `detachedSince` when a real viewer attaches — deliberately stopping\n // the TTL clock — and it takes this same per-run lock, so an\n // expiry decided at listing time is stale by the time the claim is\n // held. Cancelling on the stale value poisoned a now-live run:\n // nothing in the tree ever clears `cancelRequested`, so on that\n // viewer's next ORDINARY disconnect `middleware.ts`'s\n // `wasCancelRequested` read skips the detach branch and destroys the\n // sandbox of a healthy, actively-viewed run.\n const current = await runs.get(runId)\n if (current === null) {\n throw new RunClaimNotAcquiredError(runId, 'unknown')\n }\n if (\n current.detachedSince === undefined ||\n current.detachedSince > ctx.cutoff\n ) {\n // The viewer came back. `'not-claimed'` already documents \"a real\n // viewer attaching mid-sweep is exactly this\", and refusing here\n // leaves the run as untouched as the leave-alone path does: no\n // cancel, no append, no terminal record, no `close()`.\n throw new RunClaimNotAcquiredError(runId, 'superseded')\n }\n // BEFORE the drive, never after — and never before the claim.\n // `withSandbox`'s `onAbort` resolves the out-of-band cancel band from\n // the record, so recording the intent first is what makes the teardown\n // an explicit cancel that DESTROYS the sandbox rather than a second\n // detach that re-arms `detachedSince` and leaves the run to be swept\n // again forever. Recorded after the drive it is pure bookkeeping on a\n // run that already tore down the wrong way. Recorded before the CLAIM\n // it is an unfenced, sticky write on a record this host does not own,\n // derived from a value the lock exists to make current.\n await requestRunCancel(runs, runId)\n }\n // Before the first append, never after: `pipeToRunLog` snapshots to align.\n await awaitLogQuiescence(\n ctx.options.durability(runId),\n ctx.fenceQuietMs,\n )\n // A safety net, not a mechanism (see the module doc). `AbortSignal.any`\n // is this package's idiom for linking one — see\n // `testkit/takeover-conformance.ts`.\n //\n // ARMED HERE, not before `withRunClaim`. Both the lock wait and the\n // quiescence wait consume a timer started earlier: quiescence always\n // sleeps at least one `fenceQuietMs` and may sleep six, and lock\n // acquisition waits behind whoever holds it, unbounded. The effective\n // budget was silently `runBudgetMs − fenceQuietMs − lockWait`, and once\n // it went negative the claim was acquired with the timer already fired:\n // `pipeToRunLog` hit its entry `signal.aborted` check before pulling one\n // chunk, so a FINISHED agent's transcript was recorded `'aborted'` and\n // its log closed — and a terminal record leaves `listReclaimable`\n // forever, so that transcript was then unreplayable while `reclaim`\n // destroyed the sandbox holding the only copy. The budget bounds the\n // DRIVE, not the queue.\n budget = AbortSignal.timeout(ctx.runBudgetMs)\n // `claim.signal` is in the composed signal because losing the lease MUST\n // stop the drive: a successor that took the run over is appending to the\n // same log, and this drive continuing would double every chunk.\n const signal = AbortSignal.any([claim.signal, budget])\n return pipeToRunLog(ctx.options.drive({ runId, threadId, signal }), {\n // BOTH seams, over the SAME claim, as `driver.ts` explains: fencing the\n // log alone just moves the harm to \"a dead host marks the successor's\n // live run failed\".\n runs: fenceRunStore(runs, claim, {\n ...(logger === undefined ? {} : { logger }),\n }),\n durability: (id) =>\n fenceDurability(ctx.options.durability(id), claim, { runs }),\n runId,\n threadId,\n signal,\n ...(logger === undefined ? {} : { logger }),\n })\n },\n )\n\n const terminal = isTerminalRunStatus(final.status)\n let outcome: ReapRunOutcome\n // `&& !expired` is the whole subtlety. The budget is an ANOMALY only on the\n // finalization path, where the probe already said the agent hit its sentinel\n // and a replay that will not finish in 30s means the journal read, the\n // translation, or the log is misbehaving. On the EXPIRY path there was no\n // probe and the agent may well be mid-sentence: `requestRunCancel` writes a\n // record field whose only reader is `withSandbox`'s `onAbort` (which runs\n // after something else has already aborted), so the budget is the sole thing\n // that ends the drive of a still-producing expired run. That is the designed\n // path, not a misbehaving one, and reporting it as the anomaly made\n // `'expired'` unreachable for exactly the runs the TTL exists to expire.\n // `budget` is armed inside the claim, so reaching here means it was armed;\n // `?? false` keeps the read total rather than asserting that.\n if ((budget?.aborted ?? false) && !expired) {\n outcome = 'budget-exceeded'\n } else if (!terminal) {\n // The terminal write was SUPPRESSED and `finish`'s re-read answered with a\n // live record, which `fenceRunStore` only does when this host lost the claim\n // to another one. That is the same fact as a refused claim, reported the\n // same way rather than as a success that wrote nothing.\n outcome = 'not-claimed'\n } else {\n outcome = expired ? 'expired' : 'finalized'\n }\n\n // CAPTURED BEFORE THE RECLAIM BLOCK, which may overwrite `outcome` with\n // `'reclaim-failed'`. Conditioning the `terminalizedAnyway` spread on the\n // post-reclaim `outcome` dropped the budget diagnostic from exactly the\n // entries that need it most: a run that blew its budget AND then failed to\n // reclaim reported neither fact but the leak, and an operator cannot\n // diagnose a leak on a run whose replay was already misbehaving without\n // knowing that it was.\n const budgetAnomaly = outcome === 'budget-exceeded'\n\n let reclaimError: unknown\n if (terminal && ctx.options.reclaim !== undefined) {\n try {\n // `record`, NOT `final`. When the terminal `update` fails, `finish` returns\n // a LOCALLY REBUILT record that carries only `runId`/`threadId`/`startedAt`\n // plus the terminal patch — no `sandboxKey` — so `reclaimSandbox` would see\n // `undefined`, answer `'no-sandbox-key'`, and the sandbox would leak\n // silently on exactly the path where something already went wrong.\n await ctx.options.reclaim(record)\n } catch (error) {\n // CAUGHT HERE rather than in the outer catch, which would report a bare\n // `'failed'` with no `status` and no `exitCode`. `reclaimSandbox`\n // deliberately does not guard `instances.get` (its contract is that the\n // CALLER records the failure) and neither does `sandboxReclaimer`, so a\n // throwing instance store landed there. By this point the record is\n // terminal and the log closed, so the run is out of `listReclaimable`\n // forever and no later sweep will retry: the sandbox leaks, and an\n // operator reading `'failed'` cannot tell \"transcript saved, sandbox NOT\n // reclaimed\" from \"the sweep failed and the run was never finalized\".\n reclaimError = error\n outcome = 'reclaim-failed'\n safeLog(logger, 'errors', `reap: reclaiming run ${runId} failed`, {\n runId,\n status: final.status,\n error,\n })\n }\n }\n\n return {\n runId,\n outcome,\n status: final.status,\n ...(exitCode === undefined ? {} : { exitCode }),\n ...(budgetAnomaly ? { terminalizedAnyway: terminal } : {}),\n ...(reclaimError === undefined ? {} : { error: reclaimError }),\n }\n } catch (error) {\n if (isClaimRefusal(error)) {\n safeLog(logger, 'sandbox', `reap: not driving run ${runId}`, {\n runId,\n error,\n })\n return { runId, outcome: 'not-claimed', error }\n }\n // Folded into the summary rather than rethrown: one bad run must not abandon\n // the rest of the batch, and there is no caller to receive a rejection.\n safeLog(logger, 'errors', `reap: sweeping run ${runId} failed`, {\n runId,\n error,\n })\n return { runId, outcome: 'failed', error }\n }\n}\n\n/**\n * Sweep the detached runs a `RunStore` surfaces, saving each finished run's\n * transcript and reclaiming its sandbox.\n *\n * A plain async function with no timer and no daemon: call it from a cron, a\n * queue consumer, a Durable Object `alarm()`, or a `waitUntil`. It NEVER rejects\n * — every failure is logged and counted in the returned {@link ReapResult}.\n *\n * ONE `listReclaimable({ now, ttlMs: 0 })` call, deliberately: `ttlMs: 0` is\n * every detached run, which is the candidate set for FINALIZATION (a run that hit\n * its sentinel one second after the viewer left has an unsaved transcript and\n * must not wait out the TTL), and expiry is then classified in-process against\n * the same inclusive cutoff. Listing twice with two TTLs would cost a second\n * store round-trip to compute a subset.\n *\n * `listReclaimable` is OPTIONAL on `RunStore`. A backend without it cannot be\n * reaped, which answers `{ considered: 0 }` plus one log line rather than\n * throwing — the same graceful degrade every other optional-method call site in\n * the repo does (`store.findActiveRun?.(threadId)`).\n */\nexport async function reapDetachedRuns<TOffset extends string = string>(\n options: ReapOptions<TOffset>,\n): Promise<ReapResult> {\n const logger = options.logger\n const outcomes = emptyOutcomes()\n const entries: Array<ReapRunEntry> = []\n const empty = (): ReapResult => ({\n considered: 0,\n probed: 0,\n outcomes,\n runs: entries,\n })\n\n const list = options.runs.listReclaimable?.bind(options.runs)\n if (list === undefined) {\n safeLog(\n logger,\n 'sandbox',\n 'reap: the run store does not implement listReclaimable; nothing to sweep',\n {},\n )\n return empty()\n }\n\n let candidates: Array<RunRecord>\n try {\n candidates = await list({ now: options.now, ttlMs: 0 })\n } catch (error) {\n safeLog(logger, 'errors', 'reap: listing reclaimable runs failed', {\n error,\n })\n return empty()\n }\n\n // Capped so one invocation cannot outlive its platform's budget and be killed\n // mid-drive. `slice` and not a `break`, so `considered` reports the batch the\n // sweep actually took responsibility for.\n const maxRuns = Math.max(0, Math.trunc(options.maxRuns ?? DEFAULT_MAX_RUNS))\n const batch = candidates.slice(0, maxRuns)\n\n const ctx: ReapContext<TOffset> = {\n options,\n runBudgetMs: options.runBudgetMs ?? DEFAULT_RUN_BUDGET_MS,\n fenceQuietMs: options.fenceQuietMs ?? DEFAULT_FENCE_QUIET_MS,\n cutoff: options.now - options.detachedRunTtlMs,\n }\n const counters = { probed: 0 }\n\n // Sequential on purpose: each run costs a lock, a provider round-trip, and a\n // full replay, and a cron invocation's budget is the scarce resource. Fanning\n // out would multiply peak load against the provider for no throughput a\n // subsequent tick cannot supply.\n for (const record of batch) {\n const entry = await reapOne(record, ctx, counters)\n outcomes[entry.outcome] += 1\n entries.push(entry)\n }\n\n return {\n considered: batch.length,\n probed: counters.probed,\n outcomes,\n runs: entries,\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAyGA,IAAa,wBAAwB;;;;;;;AAQrC,IAAa,mBAAmB;;AAGhC,IAAa,2BAA2B;AAuKxC,gBAAgB,YAAY,OAAsC;CAChE,MAAM;AACR;;AAGA,eAAe,YAAY,QAAiC;CAC1D,MAAM,UAAU,IAAI,YAAY;CAChC,IAAI,OAAO;CACX,WAAW,MAAM,SAAS,mBAAmB,YAAY,MAAM,CAAC,GAC9D,QAAQ,QAAQ,OAAO,OAAO,EAAE,QAAQ,KAAK,CAAC;CAEhD,OAAO,OAAO,QAAQ,OAAO;AAC/B;;;;;;;;;;;;;;;;;AAkBA,eAAsB,aAAa,OAOT;CACxB,IAAI;EACF,MAAM,QAAQ,aAAa,MAAM,OAAO,MAAM,GAAG;EAUjD,MAAM,WAAW,iBAAiB,MAAM,aAAY,MAT/B,MAAM,OAAO,QAAQ,KACxC,wBACE,OACA,MAAM,YAAA,IACR,CACF,EAAA,CAI2D,MAAM,GAAG,KAAK;EACzE,OAAO,aAAa,OAChB,EAAE,OAAO,YAAY,IACrB;GAAE,OAAO;GAAY;EAAS;CACpC,SAAS,OAAO;EACd,OAAO;GAAE,OAAO;GAAW;EAAM;CACnC;AACF;;AAGA,SAAS,gBAAgD;CACvD,OAAO;EACL,WAAW;EACX,SAAS;EACT,WAAW;EACX,SAAS;EACT,mBAAmB;EACnB,eAAe;EACf,kBAAkB;EAClB,QAAQ;CACV;AACF;;;;;;AAOA,SAAS,QACP,QACA,OACA,SACA,SACM;CACN,IAAI;EACF,IAAI,UAAU,UAAU,QAAQ,OAAO,SAAS,OAAO;OAClD,QAAQ,QAAQ,SAAS,OAAO;CACvC,QAAQ,CAER;AACF;;AAYA,SAAS,eAAe,OAAyB;CAC/C,OACE,iBAAiB,4BACjB,iBAAiB;AAErB;;;;;;;;;;;;;;;;;;;;;;;;AAyBA,eAAe,QACb,QACA,KACA,UACuB;CACvB,MAAM,EAAE,MAAM,OAAO,WAAW,IAAI;CACpC,MAAM,EAAE,OAAO,aAAa;CAE5B,IAAI;EAKF,MAAM,UACJ,OAAO,kBAAkB,KAAA,KAAa,OAAO,iBAAiB,IAAI;EAEpE,IAAI;EACJ,IAAI,CAAC,SAAS;GACZ,SAAS,UAAU;GACnB,MAAM,QAAQ,MAAM,IAAI,QAAQ,YAAY,MAAM;GAClD,IAAI,MAAM,UAAU,YAAY;IAK9B,QAAQ,QAAQ,WAAW,qBAAqB,MAAM,SAAS;KAC7D;KACA,OAAO,MAAM;KACb,GAAI,MAAM,UAAU,aAAa,MAAM,UAAU,KAAA,IAC7C,EAAE,OAAO,MAAM,MAAM,IACrB,CAAC;IACP,CAAC;IACD,OAAO;KACL;KACA,SAAS,MAAM;KACf,GAAI,MAAM,UAAU,aAAa,MAAM,UAAU,KAAA,IAC7C,EAAE,OAAO,MAAM,MAAM,IACrB,CAAC;IACP;GACF;GACA,WAAW,MAAM;EACnB;EAIA,IAAI;EACJ,MAAM,QAAQ,MAAM,aAClB;GACE;GACA;GACA;GACA,cAAc,IAAI;GAClB,GAAI,WAAW,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO;EAC3C,GACA,OAAO,UAAU;GACf,IAAI,SAAS;IAWX,MAAM,UAAU,MAAM,KAAK,IAAI,KAAK;IACpC,IAAI,YAAY,MACd,MAAM,IAAI,yBAAyB,OAAO,SAAS;IAErD,IACE,QAAQ,kBAAkB,KAAA,KAC1B,QAAQ,gBAAgB,IAAI,QAM5B,MAAM,IAAI,yBAAyB,OAAO,YAAY;IAWxD,MAAM,iBAAiB,MAAM,KAAK;GACpC;GAEA,MAAM,mBACJ,IAAI,QAAQ,WAAW,KAAK,GAC5B,IAAI,YACN;GAiBA,SAAS,YAAY,QAAQ,IAAI,WAAW;GAI5C,MAAM,SAAS,YAAY,IAAI,CAAC,MAAM,QAAQ,MAAM,CAAC;GACrD,OAAO,aAAa,IAAI,QAAQ,MAAM;IAAE;IAAO;IAAU;GAAO,CAAC,GAAG;IAIlE,MAAM,cAAc,MAAM,OAAO,EAC/B,GAAI,WAAW,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,EAC3C,CAAC;IACD,aAAa,OACX,gBAAgB,IAAI,QAAQ,WAAW,EAAE,GAAG,OAAO,EAAE,KAAK,CAAC;IAC7D;IACA;IACA;IACA,GAAI,WAAW,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO;GAC3C,CAAC;EACH,CACF;EAEA,MAAM,WAAW,oBAAoB,MAAM,MAAM;EACjD,IAAI;EAaJ,KAAK,QAAQ,WAAW,UAAU,CAAC,SACjC,UAAU;OACL,IAAI,CAAC,UAKV,UAAU;OAEV,UAAU,UAAU,YAAY;EAUlC,MAAM,gBAAgB,YAAY;EAElC,IAAI;EACJ,IAAI,YAAY,IAAI,QAAQ,YAAY,KAAA,GACtC,IAAI;GAMF,MAAM,IAAI,QAAQ,QAAQ,MAAM;EAClC,SAAS,OAAO;GAUd,eAAe;GACf,UAAU;GACV,QAAQ,QAAQ,UAAU,wBAAwB,MAAM,UAAU;IAChE;IACA,QAAQ,MAAM;IACd;GACF,CAAC;EACH;EAGF,OAAO;GACL;GACA;GACA,QAAQ,MAAM;GACd,GAAI,aAAa,KAAA,IAAY,CAAC,IAAI,EAAE,SAAS;GAC7C,GAAI,gBAAgB,EAAE,oBAAoB,SAAS,IAAI,CAAC;GACxD,GAAI,iBAAiB,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,aAAa;EAC9D;CACF,SAAS,OAAO;EACd,IAAI,eAAe,KAAK,GAAG;GACzB,QAAQ,QAAQ,WAAW,yBAAyB,SAAS;IAC3D;IACA;GACF,CAAC;GACD,OAAO;IAAE;IAAO,SAAS;IAAe;GAAM;EAChD;EAGA,QAAQ,QAAQ,UAAU,sBAAsB,MAAM,UAAU;GAC9D;GACA;EACF,CAAC;EACD,OAAO;GAAE;GAAO,SAAS;GAAU;EAAM;CAC3C;AACF;;;;;;;;;;;;;;;;;;;;;AAsBA,eAAsB,iBACpB,SACqB;CACrB,MAAM,SAAS,QAAQ;CACvB,MAAM,WAAW,cAAc;CAC/B,MAAM,UAA+B,CAAC;CACtC,MAAM,eAA2B;EAC/B,YAAY;EACZ,QAAQ;EACR;EACA,MAAM;CACR;CAEA,MAAM,OAAO,QAAQ,KAAK,iBAAiB,KAAK,QAAQ,IAAI;CAC5D,IAAI,SAAS,KAAA,GAAW;EACtB,QACE,QACA,WACA,4EACA,CAAC,CACH;EACA,OAAO,MAAM;CACf;CAEA,IAAI;CACJ,IAAI;EACF,aAAa,MAAM,KAAK;GAAE,KAAK,QAAQ;GAAK,OAAO;EAAE,CAAC;CACxD,SAAS,OAAO;EACd,QAAQ,QAAQ,UAAU,yCAAyC,EACjE,MACF,CAAC;EACD,OAAO,MAAM;CACf;CAKA,MAAM,UAAU,KAAK,IAAI,GAAG,KAAK,MAAM,QAAQ,WAAA,EAA2B,CAAC;CAC3E,MAAM,QAAQ,WAAW,MAAM,GAAG,OAAO;CAEzC,MAAM,MAA4B;EAChC;EACA,aAAa,QAAQ,eAAA;EACrB,cAAc,QAAQ,gBAAA;EACtB,QAAQ,QAAQ,MAAM,QAAQ;CAChC;CACA,MAAM,WAAW,EAAE,QAAQ,EAAE;CAM7B,KAAK,MAAM,UAAU,OAAO;EAC1B,MAAM,QAAQ,MAAM,QAAQ,QAAQ,KAAK,QAAQ;EACjD,SAAS,MAAM,YAAY;EAC3B,QAAQ,KAAK,KAAK;CACpB;CAEA,OAAO;EACL,YAAY,MAAM;EAClB,QAAQ,SAAS;EACjB;EACA,MAAM;CACR;AACF"}
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
import { InternalLogger } from '@tanstack/ai/adapter-internals';
|
|
2
|
+
import { RunRecord } from '@tanstack/ai';
|
|
3
|
+
import { SandboxProvider } from './contracts.js';
|
|
4
|
+
import { SandboxInstanceStore } from './instance-store.js';
|
|
5
|
+
export interface ReclaimSandboxOptions {
|
|
6
|
+
provider: SandboxProvider;
|
|
7
|
+
instances: SandboxInstanceStore;
|
|
8
|
+
logger?: InternalLogger;
|
|
9
|
+
}
|
|
10
|
+
export type ReclaimOutcome =
|
|
11
|
+
/** The provider was asked to destroy it and the instance record is gone. */
|
|
12
|
+
'destroyed'
|
|
13
|
+
/**
|
|
14
|
+
* The provider's `destroy` THREW. The instance record was still deleted (see
|
|
15
|
+
* the ordering note on {@link reclaimSandbox}), so the sandbox — if it is in
|
|
16
|
+
* fact still running — is now unreachable from here and bills until the
|
|
17
|
+
* provider's own idle reclamation, if any.
|
|
18
|
+
*
|
|
19
|
+
* This is the one outcome that means the cost leak the reaper exists to stop
|
|
20
|
+
* is still leaking, so it is reported distinctly instead of being folded into
|
|
21
|
+
* `'destroyed'`, and {@link sandboxReclaimer} logs it above debug level.
|
|
22
|
+
*/
|
|
23
|
+
| 'destroy-failed'
|
|
24
|
+
/** The run never ran in a sandbox. */
|
|
25
|
+
| 'no-sandbox-key'
|
|
26
|
+
/** No instance record for that key; nothing to do. */
|
|
27
|
+
| 'not-found'
|
|
28
|
+
/** The record belongs to a different provider; refused. */
|
|
29
|
+
| 'provider-mismatch';
|
|
30
|
+
/**
|
|
31
|
+
* Destroy the sandbox a terminal run was bound to.
|
|
32
|
+
*
|
|
33
|
+
* Two orderings are load-bearing:
|
|
34
|
+
*
|
|
35
|
+
* - **The provider check before either `destroy` or `delete`.** A multi-provider
|
|
36
|
+
* application would otherwise hand a Docker container id to Daytona's
|
|
37
|
+
* `destroy`, which at best errors and at worst matches an unrelated sandbox
|
|
38
|
+
* in the other provider's id namespace. Getting this wrong destroys a
|
|
39
|
+
* stranger's workload, so it is the first gate — a mismatch touches NOTHING,
|
|
40
|
+
* including the record, which the right provider still needs.
|
|
41
|
+
* - **`destroy` before `delete`, and `delete` regardless of whether `destroy`
|
|
42
|
+
* succeeded.** The provider sandbox may already be gone (idle-reclaimed, the
|
|
43
|
+
* region wiped, the container pruned). Keeping an instance record that points
|
|
44
|
+
* at nothing guarantees a failed `resume` on the thread's next turn, which is
|
|
45
|
+
* strictly worse than an orphaned provider sandbox — one is a broken user
|
|
46
|
+
* experience, the other is a bounded cost the provider itself will reclaim.
|
|
47
|
+
* The delete is therefore unconditional — but a failed `destroy` returns
|
|
48
|
+
* `'destroy-failed'`, not `'destroyed'`: the record is gone either way, and an
|
|
49
|
+
* operator has to be able to tell "torn down" from "possibly still billing and
|
|
50
|
+
* no longer reachable from here".
|
|
51
|
+
*/
|
|
52
|
+
export declare function reclaimSandbox(record: RunRecord, options: ReclaimSandboxOptions): Promise<ReclaimOutcome>;
|
|
53
|
+
/**
|
|
54
|
+
* Thrown by {@link sandboxReclaimer} when {@link reclaimSandbox} answers
|
|
55
|
+
* `'destroy-failed'`.
|
|
56
|
+
*
|
|
57
|
+
* WHY AN EXCEPTION AND NOT A RETURN VALUE. `ReapOptions.reclaim` is
|
|
58
|
+
* `(record) => Promise<void>`, and the sweep's only channel for "the sandbox was
|
|
59
|
+
* NOT reclaimed" is a rejection — `reapOne` catches one and reports the run
|
|
60
|
+
* `'reclaim-failed'` with its `status` and `exitCode` intact. A reclaimer that
|
|
61
|
+
* logged this arm and returned normally therefore reported `'finalized'`, and
|
|
62
|
+
* `outcomes['reclaim-failed']` read `0` on precisely the leak it watches for.
|
|
63
|
+
*
|
|
64
|
+
* It carries no `cause`: `reclaimSandbox` returns a {@link ReclaimOutcome}, not
|
|
65
|
+
* the provider's rejection, and widening that return to smuggle the error out
|
|
66
|
+
* would change an outcome contract whose ordering and arms are load-bearing. The
|
|
67
|
+
* underlying `destroy` rejection is on `reclaimSandbox`'s own `warn` line, which
|
|
68
|
+
* carries the same `runId` and `sandboxKey` this error does.
|
|
69
|
+
*/
|
|
70
|
+
export declare class SandboxReclaimFailedError extends Error {
|
|
71
|
+
readonly runId: string;
|
|
72
|
+
/** Absent only in the impossible case; see the throw site in `sandboxReclaimer`. */
|
|
73
|
+
readonly sandboxKey: string | undefined;
|
|
74
|
+
constructor(runId: string, sandboxKey: string | undefined);
|
|
75
|
+
}
|
|
76
|
+
/**
|
|
77
|
+
* Adapt {@link reclaimSandbox} to `ReapOptions.reclaim`.
|
|
78
|
+
*
|
|
79
|
+
* REJECTS on `'destroy-failed'` — see {@link SandboxReclaimFailedError} for why
|
|
80
|
+
* that arm must not resolve. Every other outcome resolves: `'destroyed'` did the
|
|
81
|
+
* job, and `'no-sandbox-key'` / `'not-found'` / `'provider-mismatch'` all mean
|
|
82
|
+
* there is nothing for this reclaimer to tear down, which is not a sweep failure.
|
|
83
|
+
*/
|
|
84
|
+
export declare function sandboxReclaimer(options: ReclaimSandboxOptions): (record: RunRecord) => Promise<void>;
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
//#region src/reclaim.ts
|
|
2
|
+
/**
|
|
3
|
+
* Destroy the sandbox a terminal run was bound to.
|
|
4
|
+
*
|
|
5
|
+
* Two orderings are load-bearing:
|
|
6
|
+
*
|
|
7
|
+
* - **The provider check before either `destroy` or `delete`.** A multi-provider
|
|
8
|
+
* application would otherwise hand a Docker container id to Daytona's
|
|
9
|
+
* `destroy`, which at best errors and at worst matches an unrelated sandbox
|
|
10
|
+
* in the other provider's id namespace. Getting this wrong destroys a
|
|
11
|
+
* stranger's workload, so it is the first gate — a mismatch touches NOTHING,
|
|
12
|
+
* including the record, which the right provider still needs.
|
|
13
|
+
* - **`destroy` before `delete`, and `delete` regardless of whether `destroy`
|
|
14
|
+
* succeeded.** The provider sandbox may already be gone (idle-reclaimed, the
|
|
15
|
+
* region wiped, the container pruned). Keeping an instance record that points
|
|
16
|
+
* at nothing guarantees a failed `resume` on the thread's next turn, which is
|
|
17
|
+
* strictly worse than an orphaned provider sandbox — one is a broken user
|
|
18
|
+
* experience, the other is a bounded cost the provider itself will reclaim.
|
|
19
|
+
* The delete is therefore unconditional — but a failed `destroy` returns
|
|
20
|
+
* `'destroy-failed'`, not `'destroyed'`: the record is gone either way, and an
|
|
21
|
+
* operator has to be able to tell "torn down" from "possibly still billing and
|
|
22
|
+
* no longer reachable from here".
|
|
23
|
+
*/
|
|
24
|
+
async function reclaimSandbox(record, options) {
|
|
25
|
+
const key = record.sandboxKey;
|
|
26
|
+
if (key === void 0) return "no-sandbox-key";
|
|
27
|
+
const instance = await options.instances.get(key);
|
|
28
|
+
if (instance === null) return "not-found";
|
|
29
|
+
if (instance.provider !== options.provider.name) {
|
|
30
|
+
options.logger?.warn("reclaim: instance record belongs to a different provider; refusing to destroy", {
|
|
31
|
+
runId: record.runId,
|
|
32
|
+
sandboxKey: key,
|
|
33
|
+
recordProvider: instance.provider,
|
|
34
|
+
reclaimerProvider: options.provider.name
|
|
35
|
+
});
|
|
36
|
+
return "provider-mismatch";
|
|
37
|
+
}
|
|
38
|
+
let destroyFailed = false;
|
|
39
|
+
try {
|
|
40
|
+
await options.provider.destroy({ id: instance.providerSandboxId });
|
|
41
|
+
} catch (error) {
|
|
42
|
+
destroyFailed = true;
|
|
43
|
+
options.logger?.warn("reclaim: provider destroy failed; deleting the record anyway", {
|
|
44
|
+
runId: record.runId,
|
|
45
|
+
sandboxKey: key,
|
|
46
|
+
providerSandboxId: instance.providerSandboxId,
|
|
47
|
+
error
|
|
48
|
+
});
|
|
49
|
+
}
|
|
50
|
+
await options.instances.delete(key);
|
|
51
|
+
return destroyFailed ? "destroy-failed" : "destroyed";
|
|
52
|
+
}
|
|
53
|
+
/**
|
|
54
|
+
* Thrown by {@link sandboxReclaimer} when {@link reclaimSandbox} answers
|
|
55
|
+
* `'destroy-failed'`.
|
|
56
|
+
*
|
|
57
|
+
* WHY AN EXCEPTION AND NOT A RETURN VALUE. `ReapOptions.reclaim` is
|
|
58
|
+
* `(record) => Promise<void>`, and the sweep's only channel for "the sandbox was
|
|
59
|
+
* NOT reclaimed" is a rejection — `reapOne` catches one and reports the run
|
|
60
|
+
* `'reclaim-failed'` with its `status` and `exitCode` intact. A reclaimer that
|
|
61
|
+
* logged this arm and returned normally therefore reported `'finalized'`, and
|
|
62
|
+
* `outcomes['reclaim-failed']` read `0` on precisely the leak it watches for.
|
|
63
|
+
*
|
|
64
|
+
* It carries no `cause`: `reclaimSandbox` returns a {@link ReclaimOutcome}, not
|
|
65
|
+
* the provider's rejection, and widening that return to smuggle the error out
|
|
66
|
+
* would change an outcome contract whose ordering and arms are load-bearing. The
|
|
67
|
+
* underlying `destroy` rejection is on `reclaimSandbox`'s own `warn` line, which
|
|
68
|
+
* carries the same `runId` and `sandboxKey` this error does.
|
|
69
|
+
*/
|
|
70
|
+
var SandboxReclaimFailedError = class extends Error {
|
|
71
|
+
runId;
|
|
72
|
+
/** Absent only in the impossible case; see the throw site in `sandboxReclaimer`. */
|
|
73
|
+
sandboxKey;
|
|
74
|
+
constructor(runId, sandboxKey) {
|
|
75
|
+
super(`Reclaiming the sandbox for run "${runId}" failed: the provider's destroy rejected and the instance record${sandboxKey === void 0 ? "" : ` for "${sandboxKey}"`} was deleted anyway, so the sandbox may still be running and is no longer reachable from the instance store.`);
|
|
76
|
+
this.name = "SandboxReclaimFailedError";
|
|
77
|
+
this.runId = runId;
|
|
78
|
+
this.sandboxKey = sandboxKey;
|
|
79
|
+
}
|
|
80
|
+
};
|
|
81
|
+
/**
|
|
82
|
+
* Adapt {@link reclaimSandbox} to `ReapOptions.reclaim`.
|
|
83
|
+
*
|
|
84
|
+
* REJECTS on `'destroy-failed'` — see {@link SandboxReclaimFailedError} for why
|
|
85
|
+
* that arm must not resolve. Every other outcome resolves: `'destroyed'` did the
|
|
86
|
+
* job, and `'no-sandbox-key'` / `'not-found'` / `'provider-mismatch'` all mean
|
|
87
|
+
* there is nothing for this reclaimer to tear down, which is not a sweep failure.
|
|
88
|
+
*/
|
|
89
|
+
function sandboxReclaimer(options) {
|
|
90
|
+
return async (record) => {
|
|
91
|
+
const outcome = await reclaimSandbox(record, options);
|
|
92
|
+
const meta = {
|
|
93
|
+
runId: record.runId,
|
|
94
|
+
...record.sandboxKey === void 0 ? {} : { sandboxKey: record.sandboxKey }
|
|
95
|
+
};
|
|
96
|
+
if (outcome === "destroy-failed") {
|
|
97
|
+
options.logger?.errors("reclaim: destroy failed; sandbox may still be running", meta);
|
|
98
|
+
throw new SandboxReclaimFailedError(record.runId, record.sandboxKey);
|
|
99
|
+
}
|
|
100
|
+
options.logger?.sandbox(`reclaim: ${outcome}`, meta);
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
//#endregion
|
|
104
|
+
export { SandboxReclaimFailedError, reclaimSandbox, sandboxReclaimer };
|
|
105
|
+
|
|
106
|
+
//# sourceMappingURL=reclaim.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"reclaim.js","names":[],"sources":["../../src/reclaim.ts"],"sourcesContent":["/**\n * Tear down the sandbox behind a terminal run.\n *\n * `RunRecord.sandboxKey` exists so this does not have to re-derive the compound\n * key: `definition.key(ctx)` folds in the thread, the workspace hash, the tenant\n * and the reuse strategy, and the reaper has none of those. Phase 3's detach path\n * records the key at the moment it still knows it.\n *\n * DELIBERATELY NOT AN ENUMERATION. `SandboxInstanceStore` is `get`/`upsert`/\n * `delete` only, and no `list` is added: it would force every backend and the\n * conformance suite to grow an enumeration for one hypothetical caller. The\n * consequence is real and documented rather than hidden — a sandbox whose run\n * record was deleted before a sweep saw it is unreachable from here and leaks\n * until the provider's own idle reclamation takes it.\n */\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { RunRecord } from '@tanstack/ai'\nimport type { SandboxProvider } from './contracts'\nimport type { SandboxInstanceStore } from './instance-store'\n\nexport interface ReclaimSandboxOptions {\n provider: SandboxProvider\n instances: SandboxInstanceStore\n logger?: InternalLogger\n}\n\nexport type ReclaimOutcome =\n /** The provider was asked to destroy it and the instance record is gone. */\n | 'destroyed'\n /**\n * The provider's `destroy` THREW. The instance record was still deleted (see\n * the ordering note on {@link reclaimSandbox}), so the sandbox — if it is in\n * fact still running — is now unreachable from here and bills until the\n * provider's own idle reclamation, if any.\n *\n * This is the one outcome that means the cost leak the reaper exists to stop\n * is still leaking, so it is reported distinctly instead of being folded into\n * `'destroyed'`, and {@link sandboxReclaimer} logs it above debug level.\n */\n | 'destroy-failed'\n /** The run never ran in a sandbox. */\n | 'no-sandbox-key'\n /** No instance record for that key; nothing to do. */\n | 'not-found'\n /** The record belongs to a different provider; refused. */\n | 'provider-mismatch'\n\n/**\n * Destroy the sandbox a terminal run was bound to.\n *\n * Two orderings are load-bearing:\n *\n * - **The provider check before either `destroy` or `delete`.** A multi-provider\n * application would otherwise hand a Docker container id to Daytona's\n * `destroy`, which at best errors and at worst matches an unrelated sandbox\n * in the other provider's id namespace. Getting this wrong destroys a\n * stranger's workload, so it is the first gate — a mismatch touches NOTHING,\n * including the record, which the right provider still needs.\n * - **`destroy` before `delete`, and `delete` regardless of whether `destroy`\n * succeeded.** The provider sandbox may already be gone (idle-reclaimed, the\n * region wiped, the container pruned). Keeping an instance record that points\n * at nothing guarantees a failed `resume` on the thread's next turn, which is\n * strictly worse than an orphaned provider sandbox — one is a broken user\n * experience, the other is a bounded cost the provider itself will reclaim.\n * The delete is therefore unconditional — but a failed `destroy` returns\n * `'destroy-failed'`, not `'destroyed'`: the record is gone either way, and an\n * operator has to be able to tell \"torn down\" from \"possibly still billing and\n * no longer reachable from here\".\n */\nexport async function reclaimSandbox(\n record: RunRecord,\n options: ReclaimSandboxOptions,\n): Promise<ReclaimOutcome> {\n const key = record.sandboxKey\n if (key === undefined) return 'no-sandbox-key'\n\n // NOT guarded: a store failure here means we do not know what to destroy, and\n // the caller (the reaper) records it against the run. Swallowing it would hide\n // a leaking sandbox entirely.\n const instance = await options.instances.get(key)\n if (instance === null) return 'not-found'\n\n if (instance.provider !== options.provider.name) {\n options.logger?.warn(\n 'reclaim: instance record belongs to a different provider; refusing to destroy',\n {\n runId: record.runId,\n sandboxKey: key,\n recordProvider: instance.provider,\n reclaimerProvider: options.provider.name,\n },\n )\n return 'provider-mismatch'\n }\n\n let destroyFailed = false\n try {\n await options.provider.destroy({ id: instance.providerSandboxId })\n } catch (error) {\n destroyFailed = true\n options.logger?.warn(\n 'reclaim: provider destroy failed; deleting the record anyway',\n {\n runId: record.runId,\n sandboxKey: key,\n providerSandboxId: instance.providerSandboxId,\n error,\n },\n )\n }\n // Unconditional, per the ordering note above — but the OUTCOME must not claim\n // success when the destroy threw. Reporting `'destroyed'` here made a leaked,\n // now-unreachable sandbox indistinguishable from a clean teardown.\n await options.instances.delete(key)\n return destroyFailed ? 'destroy-failed' : 'destroyed'\n}\n\n/**\n * Thrown by {@link sandboxReclaimer} when {@link reclaimSandbox} answers\n * `'destroy-failed'`.\n *\n * WHY AN EXCEPTION AND NOT A RETURN VALUE. `ReapOptions.reclaim` is\n * `(record) => Promise<void>`, and the sweep's only channel for \"the sandbox was\n * NOT reclaimed\" is a rejection — `reapOne` catches one and reports the run\n * `'reclaim-failed'` with its `status` and `exitCode` intact. A reclaimer that\n * logged this arm and returned normally therefore reported `'finalized'`, and\n * `outcomes['reclaim-failed']` read `0` on precisely the leak it watches for.\n *\n * It carries no `cause`: `reclaimSandbox` returns a {@link ReclaimOutcome}, not\n * the provider's rejection, and widening that return to smuggle the error out\n * would change an outcome contract whose ordering and arms are load-bearing. The\n * underlying `destroy` rejection is on `reclaimSandbox`'s own `warn` line, which\n * carries the same `runId` and `sandboxKey` this error does.\n */\nexport class SandboxReclaimFailedError extends Error {\n readonly runId: string\n /** Absent only in the impossible case; see the throw site in `sandboxReclaimer`. */\n readonly sandboxKey: string | undefined\n\n constructor(runId: string, sandboxKey: string | undefined) {\n super(\n `Reclaiming the sandbox for run \"${runId}\" failed: the provider's destroy rejected and the instance record${\n sandboxKey === undefined ? '' : ` for \"${sandboxKey}\"`\n } was deleted anyway, so the sandbox may still be running and is no longer reachable from the instance store.`,\n )\n this.name = 'SandboxReclaimFailedError'\n this.runId = runId\n this.sandboxKey = sandboxKey\n }\n}\n\n/**\n * Adapt {@link reclaimSandbox} to `ReapOptions.reclaim`.\n *\n * REJECTS on `'destroy-failed'` — see {@link SandboxReclaimFailedError} for why\n * that arm must not resolve. Every other outcome resolves: `'destroyed'` did the\n * job, and `'no-sandbox-key'` / `'not-found'` / `'provider-mismatch'` all mean\n * there is nothing for this reclaimer to tear down, which is not a sweep failure.\n */\nexport function sandboxReclaimer(\n options: ReclaimSandboxOptions,\n): (record: RunRecord) => Promise<void> {\n return async (record) => {\n const outcome = await reclaimSandbox(record, options)\n const meta = {\n runId: record.runId,\n ...(record.sandboxKey === undefined\n ? {}\n : { sandboxKey: record.sandboxKey }),\n }\n if (outcome === 'destroy-failed') {\n // ABOVE DEBUG DELIBERATELY. Every other outcome is bookkeeping an operator\n // never needs to see; this one says a billed sandbox may still be running\n // with its only lookup row deleted, which nothing downstream will retry.\n options.logger?.errors(\n 'reclaim: destroy failed; sandbox may still be running',\n meta,\n )\n // AND THEN THROWS, so the sweep's `'reclaim-failed'` outcome is reachable\n // through the shipped reclaimer and not only through a custom one. The log\n // line alone is invisible to a `ReapResult` consumer.\n //\n // `record.sandboxKey` is defined on this arm — `reclaimSandbox` answers\n // `'no-sandbox-key'` before it ever reaches `destroy` otherwise — so this\n // is passed through rather than asserted: a non-null assertion is banned\n // here, and inventing a placeholder key would put a fake id in the message.\n throw new SandboxReclaimFailedError(record.runId, record.sandboxKey)\n }\n options.logger?.sandbox(`reclaim: ${outcome}`, meta)\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;AAqEA,eAAsB,eACpB,QACA,SACyB;CACzB,MAAM,MAAM,OAAO;CACnB,IAAI,QAAQ,KAAA,GAAW,OAAO;CAK9B,MAAM,WAAW,MAAM,QAAQ,UAAU,IAAI,GAAG;CAChD,IAAI,aAAa,MAAM,OAAO;CAE9B,IAAI,SAAS,aAAa,QAAQ,SAAS,MAAM;EAC/C,QAAQ,QAAQ,KACd,iFACA;GACE,OAAO,OAAO;GACd,YAAY;GACZ,gBAAgB,SAAS;GACzB,mBAAmB,QAAQ,SAAS;EACtC,CACF;EACA,OAAO;CACT;CAEA,IAAI,gBAAgB;CACpB,IAAI;EACF,MAAM,QAAQ,SAAS,QAAQ,EAAE,IAAI,SAAS,kBAAkB,CAAC;CACnE,SAAS,OAAO;EACd,gBAAgB;EAChB,QAAQ,QAAQ,KACd,gEACA;GACE,OAAO,OAAO;GACd,YAAY;GACZ,mBAAmB,SAAS;GAC5B;EACF,CACF;CACF;CAIA,MAAM,QAAQ,UAAU,OAAO,GAAG;CAClC,OAAO,gBAAgB,mBAAmB;AAC5C;;;;;;;;;;;;;;;;;;AAmBA,IAAa,4BAAb,cAA+C,MAAM;CACnD;;CAEA;CAEA,YAAY,OAAe,YAAgC;EACzD,MACE,mCAAmC,MAAM,mEACvC,eAAe,KAAA,IAAY,KAAK,SAAS,WAAW,GACrD,6GACH;EACA,KAAK,OAAO;EACZ,KAAK,QAAQ;EACb,KAAK,aAAa;CACpB;AACF;;;;;;;;;AAUA,SAAgB,iBACd,SACsC;CACtC,OAAO,OAAO,WAAW;EACvB,MAAM,UAAU,MAAM,eAAe,QAAQ,OAAO;EACpD,MAAM,OAAO;GACX,OAAO,OAAO;GACd,GAAI,OAAO,eAAe,KAAA,IACtB,CAAC,IACD,EAAE,YAAY,OAAO,WAAW;EACtC;EACA,IAAI,YAAY,kBAAkB;GAIhC,QAAQ,QAAQ,OACd,yDACA,IACF;GASA,MAAM,IAAI,0BAA0B,OAAO,OAAO,OAAO,UAAU;EACrE;EACA,QAAQ,QAAQ,QAAQ,YAAY,WAAW,IAAI;CACrD;AACF"}
|
package/dist/esm/remote-tools.js
CHANGED
|
@@ -1,76 +1,87 @@
|
|
|
1
|
+
//#region src/remote-tools.ts
|
|
2
|
+
/** Narrow an unknown body into a {@link ToolExecRequest} (project rule: no `as`). */
|
|
1
3
|
function isToolExecRequest(value) {
|
|
2
|
-
|
|
4
|
+
return value !== null && typeof value === "object" && "name" in value && typeof value.name === "string";
|
|
3
5
|
}
|
|
6
|
+
/**
|
|
7
|
+
* Rebuild `chat()` tool objects (container side) from serialized descriptors.
|
|
8
|
+
* Each stub advertises the descriptor's JSON-schema and delegates `execute` to
|
|
9
|
+
* the executor; the harness adapter bridges them like any other tool. The
|
|
10
|
+
* harness's `abortSignal` is forwarded so a cancelled run cancels the in-flight
|
|
11
|
+
* remote call too.
|
|
12
|
+
*/
|
|
4
13
|
function remoteToolStubs(descriptors, executor) {
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
args,
|
|
12
|
-
options?.abortSignal !== void 0 ? { signal: options.abortSignal } : {}
|
|
13
|
-
)
|
|
14
|
-
}));
|
|
14
|
+
return descriptors.map((descriptor) => ({
|
|
15
|
+
name: descriptor.name,
|
|
16
|
+
description: descriptor.description ?? "",
|
|
17
|
+
inputSchema: descriptor.inputSchema,
|
|
18
|
+
execute: (args, options) => executor.execute(descriptor.name, args, options?.abortSignal !== void 0 ? { signal: options.abortSignal } : {})
|
|
19
|
+
}));
|
|
15
20
|
}
|
|
21
|
+
/**
|
|
22
|
+
* Serialize `chat()` tools to wire descriptors to send into the container.
|
|
23
|
+
* `inputSchema` must already be a plain JSON-schema object (convert Standard
|
|
24
|
+
* Schemas before calling, the same way harness adapters advertise tools).
|
|
25
|
+
*/
|
|
16
26
|
function toolDescriptors(tools) {
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
27
|
+
return tools.map((tool) => ({
|
|
28
|
+
name: tool.name,
|
|
29
|
+
description: tool.description,
|
|
30
|
+
inputSchema: isJsonSchemaObject(tool.inputSchema) ? tool.inputSchema : {
|
|
31
|
+
type: "object",
|
|
32
|
+
properties: {}
|
|
33
|
+
}
|
|
34
|
+
}));
|
|
22
35
|
}
|
|
23
36
|
function isJsonSchemaObject(value) {
|
|
24
|
-
|
|
37
|
+
return value !== null && typeof value === "object" && "type" in value && value.type === "object";
|
|
25
38
|
}
|
|
26
39
|
function isToolExecResponse(value) {
|
|
27
|
-
|
|
40
|
+
return value !== null && typeof value === "object" && "result" in value;
|
|
28
41
|
}
|
|
42
|
+
/**
|
|
43
|
+
* The default {@link RemoteToolExecutor}: POST `{ name, args }` (bearer-gated)
|
|
44
|
+
* to the orchestrator's tool-exec endpoint and return its `result`. A non-2xx
|
|
45
|
+
* or malformed response throws (surfaced to the agent as a failed tool call by
|
|
46
|
+
* the bridge) — never silently swallowed.
|
|
47
|
+
*/
|
|
29
48
|
function httpRemoteToolExecutor(url, token) {
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
);
|
|
52
|
-
}
|
|
53
|
-
return body.result;
|
|
54
|
-
}
|
|
55
|
-
};
|
|
49
|
+
return { async execute(name, args, options) {
|
|
50
|
+
const res = await fetch(url, {
|
|
51
|
+
method: "POST",
|
|
52
|
+
headers: {
|
|
53
|
+
"content-type": "application/json",
|
|
54
|
+
authorization: `Bearer ${token}`
|
|
55
|
+
},
|
|
56
|
+
body: JSON.stringify({
|
|
57
|
+
name,
|
|
58
|
+
args
|
|
59
|
+
}),
|
|
60
|
+
...options?.signal !== void 0 ? { signal: options.signal } : {}
|
|
61
|
+
});
|
|
62
|
+
if (!res.ok) {
|
|
63
|
+
const text = await res.text();
|
|
64
|
+
throw new Error(`remote tool "${name}" failed: ${res.status} ${text.slice(0, 200)}`);
|
|
65
|
+
}
|
|
66
|
+
const body = await res.json();
|
|
67
|
+
if (!isToolExecResponse(body)) throw new Error(`remote tool "${name}": malformed orchestrator response`);
|
|
68
|
+
return body.result;
|
|
69
|
+
} };
|
|
56
70
|
}
|
|
71
|
+
/**
|
|
72
|
+
* Run a host tool by name with the given args, returning its raw result
|
|
73
|
+
* (orchestrator side of {@link httpRemoteToolExecutor}). Throws for an unknown
|
|
74
|
+
* tool or one with no `execute` — the orchestrator surfaces that as a 4xx/5xx.
|
|
75
|
+
*/
|
|
57
76
|
function executeHostTool(tools, name, args, options = {}) {
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
context: options.context,
|
|
65
|
-
abortSignal: options.signal
|
|
66
|
-
})
|
|
67
|
-
);
|
|
77
|
+
const tool = tools.find((candidate) => candidate.name === name);
|
|
78
|
+
if (!tool?.execute) return Promise.reject(/* @__PURE__ */ new Error(`Unknown tool: ${name}`));
|
|
79
|
+
return Promise.resolve(tool.execute(args ?? {}, {
|
|
80
|
+
context: options.context,
|
|
81
|
+
abortSignal: options.signal
|
|
82
|
+
}));
|
|
68
83
|
}
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
remoteToolStubs,
|
|
74
|
-
toolDescriptors
|
|
75
|
-
};
|
|
76
|
-
//# sourceMappingURL=remote-tools.js.map
|
|
84
|
+
//#endregion
|
|
85
|
+
export { executeHostTool, httpRemoteToolExecutor, isToolExecRequest, remoteToolStubs, toolDescriptors };
|
|
86
|
+
|
|
87
|
+
//# sourceMappingURL=remote-tools.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"remote-tools.js","sources":["../../src/remote-tools.ts"],"sourcesContent":["/**\n * Host-tool delegation for the CO-LOCATED (\"combined\") sandbox model.\n *\n * In the co-located model the harness loop AND its MCP tool-bridge run INSIDE\n * the container (the in-container sandbox is just `local-process`, so the\n * existing adapter + `nodeHttpBridgeProvisioner` serve the bridge on the\n * container's own `localhost` with native stdin — nothing new there). The one\n * thing that still must cross the container→orchestrator boundary is the\n * **execution** of `chat()`-provided server tools: their `execute()` closures\n * (DB / secrets / app state) live in the orchestrator, not the container.\n *\n * This module is that narrow seam:\n * - {@link remoteToolStubs} (container side) rebuilds `chat()` tools from\n * serialized {@link ToolDescriptor}s; each stub's `execute` delegates to a\n * {@link RemoteToolExecutor} instead of running locally. The adapter bridges\n * these stubs exactly like real tools.\n * - {@link httpRemoteToolExecutor} (container side) is the default executor: it\n * POSTs `{ name, args }` to the orchestrator's tool-exec endpoint.\n * - {@link executeHostTool} (orchestrator side) runs the REAL tool and returns\n * its raw result — the only host code the container can reach.\n *\n * So the public network surface shrinks from \"the whole MCP protocol\" (served\n * from the orchestrator in the DO-drives-container model) to \"one authenticated\n * tool-exec call\" — the MCP transport itself never leaves the container.\n */\nimport type { AnyTool } from '@tanstack/ai'\nimport type { ToolDescriptor } from './tool-bridge'\n\n/** Per-call options forwarded to a {@link RemoteToolExecutor}. */\nexport interface RemoteToolExecuteOptions {\n /** Cancels the in-flight remote call when the in-container run aborts. */\n signal?: AbortSignal\n}\n\n/** Runs a named host tool with the given args, returning its raw result. */\nexport interface RemoteToolExecutor {\n execute: (\n name: string,\n args: unknown,\n options?: RemoteToolExecuteOptions,\n ) => Promise<unknown>\n}\n\n/** Wire shape of a tool-exec request the container POSTs to the orchestrator. */\nexport interface ToolExecRequest {\n name: string\n args: unknown\n}\n\n/** Narrow an unknown body into a {@link ToolExecRequest} (project rule: no `as`). */\nexport function isToolExecRequest(value: unknown): value is ToolExecRequest {\n return (\n value !== null &&\n typeof value === 'object' &&\n 'name' in value &&\n typeof value.name === 'string'\n )\n}\n\n/**\n * Rebuild `chat()` tool objects (container side) from serialized descriptors.\n * Each stub advertises the descriptor's JSON-schema and delegates `execute` to\n * the executor; the harness adapter bridges them like any other tool. The\n * harness's `abortSignal` is forwarded so a cancelled run cancels the in-flight\n * remote call too.\n */\nexport function remoteToolStubs(\n descriptors: Array<ToolDescriptor>,\n executor: RemoteToolExecutor,\n): Array<AnyTool> {\n return descriptors.map((descriptor) => ({\n name: descriptor.name,\n description: descriptor.description ?? '',\n inputSchema: descriptor.inputSchema,\n execute: (args: unknown, options?: { abortSignal?: AbortSignal }) =>\n executor.execute(\n descriptor.name,\n args,\n options?.abortSignal !== undefined\n ? { signal: options.abortSignal }\n : {},\n ),\n }))\n}\n\n/**\n * Serialize `chat()` tools to wire descriptors to send into the container.\n * `inputSchema` must already be a plain JSON-schema object (convert Standard\n * Schemas before calling, the same way harness adapters advertise tools).\n */\nexport function toolDescriptors(tools: Array<AnyTool>): Array<ToolDescriptor> {\n return tools.map((tool) => ({\n name: tool.name,\n description: tool.description,\n inputSchema: isJsonSchemaObject(tool.inputSchema)\n ? tool.inputSchema\n : { type: 'object', properties: {} },\n }))\n}\n\nfunction isJsonSchemaObject(\n value: unknown,\n): value is { type: 'object'; [key: string]: unknown } {\n return (\n value !== null &&\n typeof value === 'object' &&\n 'type' in value &&\n (value as { type?: unknown }).type === 'object'\n )\n}\n\n/** Wire shape of a tool-exec response from the orchestrator. */\ninterface ToolExecResponse {\n result: unknown\n}\n\nfunction isToolExecResponse(value: unknown): value is ToolExecResponse {\n return value !== null && typeof value === 'object' && 'result' in value\n}\n\n/**\n * The default {@link RemoteToolExecutor}: POST `{ name, args }` (bearer-gated)\n * to the orchestrator's tool-exec endpoint and return its `result`. A non-2xx\n * or malformed response throws (surfaced to the agent as a failed tool call by\n * the bridge) — never silently swallowed.\n */\nexport function httpRemoteToolExecutor(\n url: string,\n token: string,\n): RemoteToolExecutor {\n return {\n async execute(name, args, options) {\n const res = await fetch(url, {\n method: 'POST',\n headers: {\n 'content-type': 'application/json',\n authorization: `Bearer ${token}`,\n },\n body: JSON.stringify({ name, args }),\n ...(options?.signal !== undefined ? { signal: options.signal } : {}),\n })\n if (!res.ok) {\n const text = await res.text()\n throw new Error(\n `remote tool \"${name}\" failed: ${res.status} ${text.slice(0, 200)}`,\n )\n }\n const body: unknown = await res.json()\n if (!isToolExecResponse(body)) {\n throw new Error(\n `remote tool \"${name}\": malformed orchestrator response`,\n )\n }\n return body.result\n },\n }\n}\n\n/**\n * Run a host tool by name with the given args, returning its raw result\n * (orchestrator side of {@link httpRemoteToolExecutor}). Throws for an unknown\n * tool or one with no `execute` — the orchestrator surfaces that as a 4xx/5xx.\n */\nexport function executeHostTool(\n tools: Array<AnyTool>,\n name: string,\n args: unknown,\n options: { context?: unknown; signal?: AbortSignal } = {},\n): Promise<unknown> {\n const tool = tools.find((candidate) => candidate.name === name)\n if (!tool?.execute) {\n return Promise.reject(new Error(`Unknown tool: ${name}`))\n }\n return Promise.resolve(\n tool.execute(args ?? {}, {\n context: options.context,\n abortSignal: options.signal,\n }),\n )\n}\n"],"
|
|
1
|
+
{"version":3,"file":"remote-tools.js","names":[],"sources":["../../src/remote-tools.ts"],"sourcesContent":["/**\n * Host-tool delegation for the CO-LOCATED (\"combined\") sandbox model.\n *\n * In the co-located model the harness loop AND its MCP tool-bridge run INSIDE\n * the container (the in-container sandbox is just `local-process`, so the\n * existing adapter + `nodeHttpBridgeProvisioner` serve the bridge on the\n * container's own `localhost` with native stdin — nothing new there). The one\n * thing that still must cross the container→orchestrator boundary is the\n * **execution** of `chat()`-provided server tools: their `execute()` closures\n * (DB / secrets / app state) live in the orchestrator, not the container.\n *\n * This module is that narrow seam:\n * - {@link remoteToolStubs} (container side) rebuilds `chat()` tools from\n * serialized {@link ToolDescriptor}s; each stub's `execute` delegates to a\n * {@link RemoteToolExecutor} instead of running locally. The adapter bridges\n * these stubs exactly like real tools.\n * - {@link httpRemoteToolExecutor} (container side) is the default executor: it\n * POSTs `{ name, args }` to the orchestrator's tool-exec endpoint.\n * - {@link executeHostTool} (orchestrator side) runs the REAL tool and returns\n * its raw result — the only host code the container can reach.\n *\n * So the public network surface shrinks from \"the whole MCP protocol\" (served\n * from the orchestrator in the DO-drives-container model) to \"one authenticated\n * tool-exec call\" — the MCP transport itself never leaves the container.\n */\nimport type { AnyTool } from '@tanstack/ai'\nimport type { ToolDescriptor } from './tool-bridge'\n\n/** Per-call options forwarded to a {@link RemoteToolExecutor}. */\nexport interface RemoteToolExecuteOptions {\n /** Cancels the in-flight remote call when the in-container run aborts. */\n signal?: AbortSignal\n}\n\n/** Runs a named host tool with the given args, returning its raw result. */\nexport interface RemoteToolExecutor {\n execute: (\n name: string,\n args: unknown,\n options?: RemoteToolExecuteOptions,\n ) => Promise<unknown>\n}\n\n/** Wire shape of a tool-exec request the container POSTs to the orchestrator. */\nexport interface ToolExecRequest {\n name: string\n args: unknown\n}\n\n/** Narrow an unknown body into a {@link ToolExecRequest} (project rule: no `as`). */\nexport function isToolExecRequest(value: unknown): value is ToolExecRequest {\n return (\n value !== null &&\n typeof value === 'object' &&\n 'name' in value &&\n typeof value.name === 'string'\n )\n}\n\n/**\n * Rebuild `chat()` tool objects (container side) from serialized descriptors.\n * Each stub advertises the descriptor's JSON-schema and delegates `execute` to\n * the executor; the harness adapter bridges them like any other tool. The\n * harness's `abortSignal` is forwarded so a cancelled run cancels the in-flight\n * remote call too.\n */\nexport function remoteToolStubs(\n descriptors: Array<ToolDescriptor>,\n executor: RemoteToolExecutor,\n): Array<AnyTool> {\n return descriptors.map((descriptor) => ({\n name: descriptor.name,\n description: descriptor.description ?? '',\n inputSchema: descriptor.inputSchema,\n execute: (args: unknown, options?: { abortSignal?: AbortSignal }) =>\n executor.execute(\n descriptor.name,\n args,\n options?.abortSignal !== undefined\n ? { signal: options.abortSignal }\n : {},\n ),\n }))\n}\n\n/**\n * Serialize `chat()` tools to wire descriptors to send into the container.\n * `inputSchema` must already be a plain JSON-schema object (convert Standard\n * Schemas before calling, the same way harness adapters advertise tools).\n */\nexport function toolDescriptors(tools: Array<AnyTool>): Array<ToolDescriptor> {\n return tools.map((tool) => ({\n name: tool.name,\n description: tool.description,\n inputSchema: isJsonSchemaObject(tool.inputSchema)\n ? tool.inputSchema\n : { type: 'object', properties: {} },\n }))\n}\n\nfunction isJsonSchemaObject(\n value: unknown,\n): value is { type: 'object'; [key: string]: unknown } {\n return (\n value !== null &&\n typeof value === 'object' &&\n 'type' in value &&\n (value as { type?: unknown }).type === 'object'\n )\n}\n\n/** Wire shape of a tool-exec response from the orchestrator. */\ninterface ToolExecResponse {\n result: unknown\n}\n\nfunction isToolExecResponse(value: unknown): value is ToolExecResponse {\n return value !== null && typeof value === 'object' && 'result' in value\n}\n\n/**\n * The default {@link RemoteToolExecutor}: POST `{ name, args }` (bearer-gated)\n * to the orchestrator's tool-exec endpoint and return its `result`. A non-2xx\n * or malformed response throws (surfaced to the agent as a failed tool call by\n * the bridge) — never silently swallowed.\n */\nexport function httpRemoteToolExecutor(\n url: string,\n token: string,\n): RemoteToolExecutor {\n return {\n async execute(name, args, options) {\n const res = await fetch(url, {\n method: 'POST',\n headers: {\n 'content-type': 'application/json',\n authorization: `Bearer ${token}`,\n },\n body: JSON.stringify({ name, args }),\n ...(options?.signal !== undefined ? { signal: options.signal } : {}),\n })\n if (!res.ok) {\n const text = await res.text()\n throw new Error(\n `remote tool \"${name}\" failed: ${res.status} ${text.slice(0, 200)}`,\n )\n }\n const body: unknown = await res.json()\n if (!isToolExecResponse(body)) {\n throw new Error(\n `remote tool \"${name}\": malformed orchestrator response`,\n )\n }\n return body.result\n },\n }\n}\n\n/**\n * Run a host tool by name with the given args, returning its raw result\n * (orchestrator side of {@link httpRemoteToolExecutor}). Throws for an unknown\n * tool or one with no `execute` — the orchestrator surfaces that as a 4xx/5xx.\n */\nexport function executeHostTool(\n tools: Array<AnyTool>,\n name: string,\n args: unknown,\n options: { context?: unknown; signal?: AbortSignal } = {},\n): Promise<unknown> {\n const tool = tools.find((candidate) => candidate.name === name)\n if (!tool?.execute) {\n return Promise.reject(new Error(`Unknown tool: ${name}`))\n }\n return Promise.resolve(\n tool.execute(args ?? {}, {\n context: options.context,\n abortSignal: options.signal,\n }),\n )\n}\n"],"mappings":";;AAkDA,SAAgB,kBAAkB,OAA0C;CAC1E,OACE,UAAU,QACV,OAAO,UAAU,YACjB,UAAU,SACV,OAAO,MAAM,SAAS;AAE1B;;;;;;;;AASA,SAAgB,gBACd,aACA,UACgB;CAChB,OAAO,YAAY,KAAK,gBAAgB;EACtC,MAAM,WAAW;EACjB,aAAa,WAAW,eAAe;EACvC,aAAa,WAAW;EACxB,UAAU,MAAe,YACvB,SAAS,QACP,WAAW,MACX,MACA,SAAS,gBAAgB,KAAA,IACrB,EAAE,QAAQ,QAAQ,YAAY,IAC9B,CAAC,CACP;CACJ,EAAE;AACJ;;;;;;AAOA,SAAgB,gBAAgB,OAA8C;CAC5E,OAAO,MAAM,KAAK,UAAU;EAC1B,MAAM,KAAK;EACX,aAAa,KAAK;EAClB,aAAa,mBAAmB,KAAK,WAAW,IAC5C,KAAK,cACL;GAAE,MAAM;GAAU,YAAY,CAAC;EAAE;CACvC,EAAE;AACJ;AAEA,SAAS,mBACP,OACqD;CACrD,OACE,UAAU,QACV,OAAO,UAAU,YACjB,UAAU,SACT,MAA6B,SAAS;AAE3C;AAOA,SAAS,mBAAmB,OAA2C;CACrE,OAAO,UAAU,QAAQ,OAAO,UAAU,YAAY,YAAY;AACpE;;;;;;;AAQA,SAAgB,uBACd,KACA,OACoB;CACpB,OAAO,EACL,MAAM,QAAQ,MAAM,MAAM,SAAS;EACjC,MAAM,MAAM,MAAM,MAAM,KAAK;GAC3B,QAAQ;GACR,SAAS;IACP,gBAAgB;IAChB,eAAe,UAAU;GAC3B;GACA,MAAM,KAAK,UAAU;IAAE;IAAM;GAAK,CAAC;GACnC,GAAI,SAAS,WAAW,KAAA,IAAY,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;EACpE,CAAC;EACD,IAAI,CAAC,IAAI,IAAI;GACX,MAAM,OAAO,MAAM,IAAI,KAAK;GAC5B,MAAM,IAAI,MACR,gBAAgB,KAAK,YAAY,IAAI,OAAO,GAAG,KAAK,MAAM,GAAG,GAAG,GAClE;EACF;EACA,MAAM,OAAgB,MAAM,IAAI,KAAK;EACrC,IAAI,CAAC,mBAAmB,IAAI,GAC1B,MAAM,IAAI,MACR,gBAAgB,KAAK,mCACvB;EAEF,OAAO,KAAK;CACd,EACF;AACF;;;;;;AAOA,SAAgB,gBACd,OACA,MACA,MACA,UAAuD,CAAC,GACtC;CAClB,MAAM,OAAO,MAAM,MAAM,cAAc,UAAU,SAAS,IAAI;CAC9D,IAAI,CAAC,MAAM,SACT,OAAO,QAAQ,uBAAO,IAAI,MAAM,iBAAiB,MAAM,CAAC;CAE1D,OAAO,QAAQ,QACb,KAAK,QAAQ,QAAQ,CAAC,GAAG;EACvB,SAAS,QAAQ;EACjB,aAAa,QAAQ;CACvB,CAAC,CACH;AACF"}
|