@tanstack/ai-sandbox 0.3.2 → 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1 +1 @@
1
- {"version":3,"file":"bootstrap.js","names":[],"sources":["../../src/bootstrap.ts"],"sourcesContent":["/**\n * Workspace bootstrap engine — provider-agnostic because it only uses the\n * {@link SandboxHandle} contract. Runs once when a sandbox is freshly created\n * (or restored without its working tree): land the source, inject secrets,\n * detect the package manager, and run setup commands.\n *\n * Harness-specific projection (CLAUDE.md, agent skills, MCP config) is NOT done\n * here — that's each adapter's `projectWorkspace()` hook, since the format\n * differs per harness.\n */\nimport { resolveHarnessCwd } from './harness-cwd'\nimport { buildSetupPlan } from './setup-plan'\nimport { createBootstrapShell } from './shell'\nimport {\n mergeAgentsContent,\n resolveGitSkillDir,\n writeAgentsFile,\n} from './agents-file'\nimport { resolveAllSecrets, resolveSecret } from './secrets'\nimport type { SandboxHandle } from './contracts'\nimport type { PackageManager, WorkspaceDefinition } from './workspace'\n\nconst LOCKFILES: Record<Exclude<PackageManager, 'auto'>, string> = {\n pnpm: 'pnpm-lock.yaml',\n yarn: 'yarn.lock',\n bun: 'bun.lockb',\n npm: 'package-lock.json',\n}\n\nexport const DEFAULT_WORKSPACE_ROOT = '/workspace'\n\n/** Resolve the package manager, detecting from a lockfile when `'auto'`. */\nexport async function detectPackageManager(\n handle: SandboxHandle,\n workspace: WorkspaceDefinition,\n root: string,\n): Promise<Exclude<PackageManager, 'auto'> | undefined> {\n const pm = workspace.packageManager ?? 'auto'\n if (pm !== 'auto') return pm\n for (const [manager, lockfile] of Object.entries(LOCKFILES) as Array<\n [Exclude<PackageManager, 'auto'>, string]\n >) {\n if (await handle.fs.exists(`${root}/${lockfile}`)) return manager\n }\n return undefined\n}\n\nexport interface BootstrapResult {\n packageManager?: Exclude<PackageManager, 'auto'>\n ranSetup: Array<string>\n}\n\n/**\n * Bootstrap a freshly created sandbox's workspace. Idempotent enough to be safe\n * on restore: a git clone into a populated dir is skipped by checking for the\n * target dir first.\n */\nexport async function bootstrapWorkspace(\n handle: SandboxHandle,\n workspace: WorkspaceDefinition,\n options: { signal?: AbortSignal } = {},\n): Promise<BootstrapResult> {\n const root = workspace.root ?? DEFAULT_WORKSPACE_ROOT\n\n // Secrets live only in the running sandbox env (never persisted).\n if (workspace.secrets !== undefined) {\n const resolved = resolveAllSecrets(workspace.secrets)\n if (Object.keys(resolved).length > 0) {\n await handle.env.set(resolved)\n }\n }\n\n // Land the source. Clone into the handle's own default root (each provider\n // maps the conventional `/workspace` virtual root to its real backing dir),\n // rather than passing a virtual `dir` that can't be remapped inside a shell\n // command string.\n if (workspace.source.type === 'git') {\n const alreadyCloned = await handle.fs.exists(`${root}/.git`)\n if (!alreadyCloned) {\n await handle.git.clone({\n url: workspace.source.url,\n ref: workspace.source.ref,\n auth: workspace.source.auth,\n ...(workspace.source.depth !== undefined\n ? { depth: workspace.source.depth }\n : {}),\n })\n }\n }\n // 'local' is provider-pre-populated at create; 'none' starts empty.\n\n // Clone git-skill repos so setup steps (and the harness projector) can use\n // them. gitSkill clones are always shallow (depth 1) unless the skill's own\n // repo entry carries a depth override — the WorkspaceSkill `git` variant\n // does not expose one, so depth always defaults to 1 inside git.clone.\n const skills = workspace.skills ?? []\n for (const skill of skills) {\n if (skill.kind === 'git') {\n const url = skill.repo.startsWith('http')\n ? skill.repo\n : `https://github.com/${skill.repo}.git`\n const dir = resolveHarnessCwd(\n handle,\n skill.into ?? resolveGitSkillDir(root, skill),\n )\n const auth =\n skill.secret !== undefined && workspace.secrets !== undefined\n ? { token: resolveSecret(workspace.secrets, skill.secret) }\n : undefined\n await handle.git.clone({\n url,\n dir,\n ...(auth !== undefined ? { auth } : {}),\n depth: 1,\n })\n }\n }\n\n // Write AGENTS.md (and its per-CLI symlinks) when instructions are provided\n // directly on the workspace, via a fileSkill whose path is `AGENTS.md`, or\n // when named workspace scripts should be surfaced for the agent.\n let agentsContent: string | undefined\n if (\n workspace.instructions !== undefined &&\n workspace.instructions.length > 0\n ) {\n agentsContent = workspace.instructions\n } else {\n const agentsFileSkill = skills.find(\n (s): s is Extract<typeof s, { kind: 'file' }> =>\n s.kind === 'file' && s.path === 'AGENTS.md',\n )\n if (agentsFileSkill !== undefined) {\n agentsContent = agentsFileSkill.content\n }\n }\n agentsContent = mergeAgentsContent(agentsContent, workspace.scripts)\n if (agentsContent !== undefined) {\n await writeAgentsFile(handle, root, agentsContent)\n }\n\n // Write all other fileSkills directly into the workspace root.\n for (const skill of skills) {\n if (skill.kind === 'file' && skill.path !== 'AGENTS.md') {\n await handle.fs.write(`${root}/${skill.path}`, skill.content)\n }\n }\n\n const packageManager = await detectPackageManager(handle, workspace, root)\n\n // Run setup over a single persistent shell so `cd`/exports persist across\n // serial steps. Parallel groups fork the shell's current cwd+env into\n // concurrent one-shot exec calls.\n const ranSetup: Array<string> = []\n const plan = buildSetupPlan(workspace.setup)\n if (plan.length > 0) {\n const shell = await createBootstrapShell(handle, { cwd: root })\n try {\n for (const group of plan) {\n if (group.kind === 'serial') {\n const result = await shell.run(group.command)\n if (result.exitCode !== 0) {\n const tail = result.stdout.trim().slice(-1500)\n throw new Error(\n `setup step failed: ${group.command} (exit ${result.exitCode})${tail ? `\\n${tail}` : ''}`,\n )\n }\n ranSetup.push(group.command)\n } else {\n const { cwd, env } = await shell.forkState()\n const results = await Promise.all(\n group.commands.map((command) =>\n handle.process\n .exec(command, {\n cwd,\n env,\n ...(options.signal ? { signal: options.signal } : {}),\n })\n .then((res) => ({ command, res })),\n ),\n )\n const failed = results.find((entry) => entry.res.exitCode !== 0)\n if (failed !== undefined) {\n const tail = `${failed.res.stdout}\\n${failed.res.stderr}`\n .trim()\n .slice(-1500)\n throw new Error(\n `setup step failed: ${failed.command} (exit ${failed.res.exitCode})${tail ? `\\n${tail}` : ''}`,\n )\n }\n ranSetup.push(...group.commands)\n }\n }\n } finally {\n await shell.dispose()\n }\n }\n\n return { packageManager, ranSetup }\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAsBA,IAAM,YAA6D;CACjE,MAAM;CACN,MAAM;CACN,KAAK;CACL,KAAK;AACP;AAEA,IAAa,yBAAyB;;AAGtC,eAAsB,qBACpB,QACA,WACA,MACsD;CACtD,MAAM,KAAK,UAAU,kBAAkB;CACvC,IAAI,OAAO,QAAQ,OAAO;CAC1B,KAAK,MAAM,CAAC,SAAS,aAAa,OAAO,QAAQ,SAAS,GAGxD,IAAI,MAAM,OAAO,GAAG,OAAO,GAAG,KAAK,GAAG,UAAU,GAAG,OAAO;AAG9D;;;;;;AAYA,eAAsB,mBACpB,QACA,WACA,UAAoC,CAAC,GACX;CAC1B,MAAM,OAAO,UAAU,QAAA;CAGvB,IAAI,UAAU,YAAY,KAAA,GAAW;EACnC,MAAM,WAAW,kBAAkB,UAAU,OAAO;EACpD,IAAI,OAAO,KAAK,QAAQ,CAAC,CAAC,SAAS,GACjC,MAAM,OAAO,IAAI,IAAI,QAAQ;CAEjC;CAMA,IAAI,UAAU,OAAO,SAAS;MAExB,CAAC,MADuB,OAAO,GAAG,OAAO,GAAG,KAAK,MAAM,GAEzD,MAAM,OAAO,IAAI,MAAM;GACrB,KAAK,UAAU,OAAO;GACtB,KAAK,UAAU,OAAO;GACtB,MAAM,UAAU,OAAO;GACvB,GAAI,UAAU,OAAO,UAAU,KAAA,IAC3B,EAAE,OAAO,UAAU,OAAO,MAAM,IAChC,CAAC;EACP,CAAC;CAAA;CASL,MAAM,SAAS,UAAU,UAAU,CAAC;CACpC,KAAK,MAAM,SAAS,QAClB,IAAI,MAAM,SAAS,OAAO;EACxB,MAAM,MAAM,MAAM,KAAK,WAAW,MAAM,IACpC,MAAM,OACN,sBAAsB,MAAM,KAAK;EACrC,MAAM,MAAM,kBACV,QACA,MAAM,QAAQ,mBAAmB,MAAM,KAAK,CAC9C;EACA,MAAM,OACJ,MAAM,WAAW,KAAA,KAAa,UAAU,YAAY,KAAA,IAChD,EAAE,OAAO,cAAc,UAAU,SAAS,MAAM,MAAM,EAAE,IACxD,KAAA;EACN,MAAM,OAAO,IAAI,MAAM;GACrB;GACA;GACA,GAAI,SAAS,KAAA,IAAY,EAAE,KAAK,IAAI,CAAC;GACrC,OAAO;EACT,CAAC;CACH;CAMF,IAAI;CACJ,IACE,UAAU,iBAAiB,KAAA,KAC3B,UAAU,aAAa,SAAS,GAEhC,gBAAgB,UAAU;MACrB;EACL,MAAM,kBAAkB,OAAO,MAC5B,MACC,EAAE,SAAS,UAAU,EAAE,SAAS,WACpC;EACA,IAAI,oBAAoB,KAAA,GACtB,gBAAgB,gBAAgB;CAEpC;CACA,gBAAgB,mBAAmB,eAAe,UAAU,OAAO;CACnE,IAAI,kBAAkB,KAAA,GACpB,MAAM,gBAAgB,QAAQ,MAAM,aAAa;CAInD,KAAK,MAAM,SAAS,QAClB,IAAI,MAAM,SAAS,UAAU,MAAM,SAAS,aAC1C,MAAM,OAAO,GAAG,MAAM,GAAG,KAAK,GAAG,MAAM,QAAQ,MAAM,OAAO;CAIhE,MAAM,iBAAiB,MAAM,qBAAqB,QAAQ,WAAW,IAAI;CAKzE,MAAM,WAA0B,CAAC;CACjC,MAAM,OAAO,eAAe,UAAU,KAAK;CAC3C,IAAI,KAAK,SAAS,GAAG;EACnB,MAAM,QAAQ,MAAM,qBAAqB,QAAQ,EAAE,KAAK,KAAK,CAAC;EAC9D,IAAI;GACF,KAAK,MAAM,SAAS,MAClB,IAAI,MAAM,SAAS,UAAU;IAC3B,MAAM,SAAS,MAAM,MAAM,IAAI,MAAM,OAAO;IAC5C,IAAI,OAAO,aAAa,GAAG;KACzB,MAAM,OAAO,OAAO,OAAO,KAAK,CAAC,CAAC,MAAM,KAAK;KAC7C,MAAM,IAAI,MACR,sBAAsB,MAAM,QAAQ,SAAS,OAAO,SAAS,GAAG,OAAO,KAAK,SAAS,IACvF;IACF;IACA,SAAS,KAAK,MAAM,OAAO;GAC7B,OAAO;IACL,MAAM,EAAE,KAAK,QAAQ,MAAM,MAAM,UAAU;IAY3C,MAAM,UAAS,MAXO,QAAQ,IAC5B,MAAM,SAAS,KAAK,YAClB,OAAO,QACJ,KAAK,SAAS;KACb;KACA;KACA,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;IACrD,CAAC,CAAC,CACD,MAAM,SAAS;KAAE;KAAS;IAAI,EAAE,CACrC,CACF,EAAA,CACuB,MAAM,UAAU,MAAM,IAAI,aAAa,CAAC;IAC/D,IAAI,WAAW,KAAA,GAAW;KACxB,MAAM,OAAO,GAAG,OAAO,IAAI,OAAO,IAAI,OAAO,IAAI,SAC9C,KAAK,CAAC,CACN,MAAM,KAAK;KACd,MAAM,IAAI,MACR,sBAAsB,OAAO,QAAQ,SAAS,OAAO,IAAI,SAAS,GAAG,OAAO,KAAK,SAAS,IAC5F;IACF;IACA,SAAS,KAAK,GAAG,MAAM,QAAQ;GACjC;EAEJ,UAAU;GACR,MAAM,MAAM,QAAQ;EACtB;CACF;CAEA,OAAO;EAAE;EAAgB;CAAS;AACpC"}
1
+ {"version":3,"file":"bootstrap.js","names":[],"sources":["../../src/bootstrap.ts"],"sourcesContent":["/**\n * Workspace bootstrap engine — provider-agnostic because it only uses the\n * {@link SandboxHandle} contract. Runs once when a sandbox is freshly created\n * (or restored without its working tree): land the source, inject secrets,\n * detect the package manager, and run setup commands.\n *\n * Harness-specific projection (CLAUDE.md, agent skills, MCP config) is NOT done\n * here — that's each adapter's `projectWorkspace()` hook, since the format\n * differs per harness.\n */\nimport { resolveHarnessCwd } from './harness-cwd'\nimport { buildSetupPlan } from './setup-plan'\nimport { createBootstrapShell } from './shell'\nimport {\n mergeAgentsContent,\n resolveGitSkillDir,\n writeAgentsFile,\n} from './agents-file'\nimport { resolveAllSecrets, resolveSecret } from './secrets'\nimport type { SandboxHandle } from './contracts'\nimport type { PackageManager, WorkspaceDefinition } from './workspace'\n\nconst LOCKFILES: Record<Exclude<PackageManager, 'auto'>, string> = {\n pnpm: 'pnpm-lock.yaml',\n yarn: 'yarn.lock',\n bun: 'bun.lockb',\n npm: 'package-lock.json',\n}\n\nexport const DEFAULT_WORKSPACE_ROOT = '/workspace'\n\n/** Resolve the package manager, detecting from a lockfile when `'auto'`. */\nexport async function detectPackageManager(\n handle: SandboxHandle,\n workspace: WorkspaceDefinition,\n root: string,\n): Promise<Exclude<PackageManager, 'auto'> | undefined> {\n const pm = workspace.packageManager ?? 'auto'\n if (pm !== 'auto') return pm\n for (const [manager, lockfile] of Object.entries(LOCKFILES) as Array<\n [Exclude<PackageManager, 'auto'>, string]\n >) {\n if (await handle.fs.exists(`${root}/${lockfile}`)) return manager\n }\n return undefined\n}\n\nexport interface BootstrapResult {\n packageManager?: Exclude<PackageManager, 'auto'>\n ranSetup: Array<string>\n}\n\n/**\n * Bootstrap a freshly created sandbox's workspace. Idempotent enough to be safe\n * on restore: a git clone into a populated dir is skipped by checking for the\n * target dir first.\n */\nexport async function bootstrapWorkspace(\n handle: SandboxHandle,\n workspace: WorkspaceDefinition,\n options: { signal?: AbortSignal } = {},\n): Promise<BootstrapResult> {\n const root = workspace.root ?? DEFAULT_WORKSPACE_ROOT\n\n // Secrets live only in the running sandbox env (never persisted).\n if (workspace.secrets !== undefined) {\n const resolved = resolveAllSecrets(workspace.secrets)\n if (Object.keys(resolved).length > 0) {\n await handle.env.set(resolved)\n }\n }\n\n // Land the source. Clone into the handle's own default root (each provider\n // maps the conventional `/workspace` virtual root to its real backing dir),\n // rather than passing a virtual `dir` that can't be remapped inside a shell\n // command string.\n if (workspace.source.type === 'git') {\n const alreadyCloned = await handle.fs.exists(`${root}/.git`)\n if (!alreadyCloned) {\n await handle.git.clone({\n url: workspace.source.url,\n ref: workspace.source.ref,\n auth: workspace.source.auth,\n ...(workspace.source.depth !== undefined\n ? { depth: workspace.source.depth }\n : {}),\n })\n }\n }\n // 'local' is provider-pre-populated at create; 'none' starts empty.\n\n // Clone git-skill repos so setup steps (and the harness projector) can use\n // them. gitSkill clones are always shallow (depth 1) unless the skill's own\n // repo entry carries a depth override — the WorkspaceSkill `git` variant\n // does not expose one, so depth always defaults to 1 inside git.clone.\n const skills = workspace.skills ?? []\n for (const skill of skills) {\n if (skill.kind === 'git') {\n const url = skill.repo.startsWith('http')\n ? skill.repo\n : `https://github.com/${skill.repo}.git`\n const dir = resolveHarnessCwd(\n handle,\n skill.into ?? resolveGitSkillDir(root, skill),\n )\n const auth =\n skill.secret !== undefined && workspace.secrets !== undefined\n ? { token: resolveSecret(workspace.secrets, skill.secret) }\n : undefined\n await handle.git.clone({\n url,\n dir,\n ...(auth !== undefined ? { auth } : {}),\n depth: 1,\n })\n }\n }\n\n // Write AGENTS.md (and its per-CLI symlinks) when instructions are provided\n // directly on the workspace, via a fileSkill whose path is `AGENTS.md`, or\n // when named workspace scripts should be surfaced for the agent.\n let agentsContent: string | undefined\n if (\n workspace.instructions !== undefined &&\n workspace.instructions.length > 0\n ) {\n agentsContent = workspace.instructions\n } else {\n const agentsFileSkill = skills.find(\n (s): s is Extract<typeof s, { kind: 'file' }> =>\n s.kind === 'file' && s.path === 'AGENTS.md',\n )\n if (agentsFileSkill !== undefined) {\n agentsContent = agentsFileSkill.content\n }\n }\n agentsContent = mergeAgentsContent(agentsContent, workspace.scripts)\n if (agentsContent !== undefined) {\n await writeAgentsFile(handle, root, agentsContent)\n }\n\n // Write all other fileSkills directly into the workspace root.\n for (const skill of skills) {\n if (skill.kind === 'file' && skill.path !== 'AGENTS.md') {\n await handle.fs.write(`${root}/${skill.path}`, skill.content)\n }\n }\n\n const packageManager = await detectPackageManager(handle, workspace, root)\n\n // Run setup over a single persistent shell so `cd`/exports persist across\n // serial steps. Parallel groups fork the shell's current cwd+env into\n // concurrent one-shot exec calls.\n const ranSetup: Array<string> = []\n const plan = buildSetupPlan(workspace.setup)\n if (plan.length > 0) {\n const shell = await createBootstrapShell(handle, { cwd: root })\n try {\n for (const group of plan) {\n if (group.kind === 'serial') {\n const result = await shell.run(group.command)\n if (result.exitCode !== 0) {\n const tail = result.stdout.trim().slice(-1500)\n throw new Error(\n `setup step failed: ${group.command} (exit ${result.exitCode})${tail ? `\\n${tail}` : ''}`,\n )\n }\n ranSetup.push(group.command)\n } else {\n const { cwd, env } = await shell.forkState()\n const results = await Promise.all(\n group.commands.map((command) =>\n handle.process\n .exec(command, {\n cwd,\n env,\n ...(options.signal ? { signal: options.signal } : {}),\n })\n .then((res) => ({ command, res })),\n ),\n )\n const failed = results.find((entry) => entry.res.exitCode !== 0)\n if (failed !== undefined) {\n const tail = `${failed.res.stdout}\\n${failed.res.stderr}`\n .trim()\n .slice(-1500)\n throw new Error(\n `setup step failed: ${failed.command} (exit ${failed.res.exitCode})${tail ? `\\n${tail}` : ''}`,\n )\n }\n ranSetup.push(...group.commands)\n }\n }\n } finally {\n await shell.dispose()\n }\n }\n\n return { packageManager, ranSetup }\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAsBA,IAAM,YAA6D;CACjE,MAAM;CACN,MAAM;CACN,KAAK;CACL,KAAK;AACP;AAEA,IAAa,yBAAyB;;AAGtC,eAAsB,qBACpB,QACA,WACA,MACsD;CACtD,MAAM,KAAK,UAAU,kBAAkB;CACvC,IAAI,OAAO,QAAQ,OAAO;CAC1B,KAAK,MAAM,CAAC,SAAS,aAAa,OAAO,QAAQ,SAAS,GAGxD,IAAI,MAAM,OAAO,GAAG,OAAO,GAAG,KAAK,GAAG,UAAU,GAAG,OAAO;AAG9D;;;;;;AAYA,eAAsB,mBACpB,QACA,WACA,UAAoC,CAAC,GACX;CAC1B,MAAM,OAAO,UAAU,QAAA;CAGvB,IAAI,UAAU,YAAY,KAAA,GAAW;EACnC,MAAM,WAAW,kBAAkB,UAAU,OAAO;EACpD,IAAI,OAAO,KAAK,QAAQ,CAAC,CAAC,SAAS,GACjC,MAAM,OAAO,IAAI,IAAI,QAAQ;CAEjC;CAMA,IAAI,UAAU,OAAO,SAAS,OAExB;MAAA,CAAC,MADuB,OAAO,GAAG,OAAO,GAAG,KAAK,MAAM,GAEzD,MAAM,OAAO,IAAI,MAAM;GACrB,KAAK,UAAU,OAAO;GACtB,KAAK,UAAU,OAAO;GACtB,MAAM,UAAU,OAAO;GACvB,GAAI,UAAU,OAAO,UAAU,KAAA,IAC3B,EAAE,OAAO,UAAU,OAAO,MAAM,IAChC,CAAC;EACP,CAAC;CAAA;CASL,MAAM,SAAS,UAAU,UAAU,CAAC;CACpC,KAAK,MAAM,SAAS,QAClB,IAAI,MAAM,SAAS,OAAO;EACxB,MAAM,MAAM,MAAM,KAAK,WAAW,MAAM,IACpC,MAAM,OACN,sBAAsB,MAAM,KAAK;EACrC,MAAM,MAAM,kBACV,QACA,MAAM,QAAQ,mBAAmB,MAAM,KAAK,CAC9C;EACA,MAAM,OACJ,MAAM,WAAW,KAAA,KAAa,UAAU,YAAY,KAAA,IAChD,EAAE,OAAO,cAAc,UAAU,SAAS,MAAM,MAAM,EAAE,IACxD,KAAA;EACN,MAAM,OAAO,IAAI,MAAM;GACrB;GACA;GACA,GAAI,SAAS,KAAA,IAAY,EAAE,KAAK,IAAI,CAAC;GACrC,OAAO;EACT,CAAC;CACH;CAMF,IAAI;CACJ,IACE,UAAU,iBAAiB,KAAA,KAC3B,UAAU,aAAa,SAAS,GAEhC,gBAAgB,UAAU;MACrB;EACL,MAAM,kBAAkB,OAAO,MAC5B,MACC,EAAE,SAAS,UAAU,EAAE,SAAS,WACpC;EACA,IAAI,oBAAoB,KAAA,GACtB,gBAAgB,gBAAgB;CAEpC;CACA,gBAAgB,mBAAmB,eAAe,UAAU,OAAO;CACnE,IAAI,kBAAkB,KAAA,GACpB,MAAM,gBAAgB,QAAQ,MAAM,aAAa;CAInD,KAAK,MAAM,SAAS,QAClB,IAAI,MAAM,SAAS,UAAU,MAAM,SAAS,aAC1C,MAAM,OAAO,GAAG,MAAM,GAAG,KAAK,GAAG,MAAM,QAAQ,MAAM,OAAO;CAIhE,MAAM,iBAAiB,MAAM,qBAAqB,QAAQ,WAAW,IAAI;CAKzE,MAAM,WAA0B,CAAC;CACjC,MAAM,OAAO,eAAe,UAAU,KAAK;CAC3C,IAAI,KAAK,SAAS,GAAG;EACnB,MAAM,QAAQ,MAAM,qBAAqB,QAAQ,EAAE,KAAK,KAAK,CAAC;EAC9D,IAAI;GACF,KAAK,MAAM,SAAS,MAClB,IAAI,MAAM,SAAS,UAAU;IAC3B,MAAM,SAAS,MAAM,MAAM,IAAI,MAAM,OAAO;IAC5C,IAAI,OAAO,aAAa,GAAG;KACzB,MAAM,OAAO,OAAO,OAAO,KAAK,CAAC,CAAC,MAAM,KAAK;KAC7C,MAAM,IAAI,MACR,sBAAsB,MAAM,QAAQ,SAAS,OAAO,SAAS,GAAG,OAAO,KAAK,SAAS,IACvF;IACF;IACA,SAAS,KAAK,MAAM,OAAO;GAC7B,OAAO;IACL,MAAM,EAAE,KAAK,QAAQ,MAAM,MAAM,UAAU;IAY3C,MAAM,UAAS,MAXO,QAAQ,IAC5B,MAAM,SAAS,KAAK,YAClB,OAAO,QACJ,KAAK,SAAS;KACb;KACA;KACA,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;IACrD,CAAC,CAAC,CACD,MAAM,SAAS;KAAE;KAAS;IAAI,EAAE,CACrC,CACF,EAAA,CACuB,MAAM,UAAU,MAAM,IAAI,aAAa,CAAC;IAC/D,IAAI,WAAW,KAAA,GAAW;KACxB,MAAM,OAAO,GAAG,OAAO,IAAI,OAAO,IAAI,OAAO,IAAI,SAC9C,KAAK,CAAC,CACN,MAAM,KAAK;KACd,MAAM,IAAI,MACR,sBAAsB,OAAO,QAAQ,SAAS,OAAO,IAAI,SAAS,GAAG,OAAO,KAAK,SAAS,IAC5F;IACF;IACA,SAAS,KAAK,GAAG,MAAM,QAAQ;GACjC;EAEJ,UAAU;GACR,MAAM,MAAM,QAAQ;EACtB;CACF;CAEA,OAAO;EAAE;EAAgB;CAAS;AACpC"}
@@ -80,7 +80,7 @@ import { isTerminalRunStatus } from "@tanstack/ai";
80
80
  * sandbox's disk survives. Erring long is the cheap direction: the cost of too
81
81
  * long is bytes, the cost of too short is a destroyed live run.
82
82
  */
83
- var DEFAULT_ORPHAN_TTL_MS = 3600 * 1e3;
83
+ var DEFAULT_ORPHAN_TTL_MS = 36e5;
84
84
  /**
85
85
  * Ceiling on deletions per sweep. A cron-driven sweep runs unattended, so a
86
86
  * mistake — a store that answers `terminal` for everything, a misconfigured
@@ -163,7 +163,8 @@ async function pruneJournals(options) {
163
163
  let ageGate = "unavailable";
164
164
  const mtimes = /* @__PURE__ */ new Map();
165
165
  try {
166
- const parsed = parseJournalMtimeListing((await options.handle.process.exec(journalMtimeListCommand(dir))).stdout, dir);
166
+ const probe = await options.handle.process.exec(journalMtimeListCommand(dir));
167
+ const parsed = parseJournalMtimeListing(probe.stdout, dir);
167
168
  if (parsed.kind === "listed") {
168
169
  ageGate = "listed";
169
170
  for (const entry of parsed.entries) mtimes.set(entry.name, entry.mtimeMs);
@@ -1 +1 @@
1
- {"version":3,"file":"journal-sweep.js","names":[],"sources":["../../src/journal-sweep.ts"],"sourcesContent":["/**\n * Bound the journal directory: delete the journals nobody will ever read again,\n * and — far more importantly — refuse to delete anything else.\n *\n * `journalCleanupCommand` already deletes ONE run's journal at the moment its\n * `{\"__exit\":N}` sentinel is observed. That covers every run a host watched to\n * completion and covers nothing else: a run that reaches its sentinel while\n * DETACHED has no host reading its journal, so nothing observes the sentinel and\n * nothing calls the cleanup. Those journals accumulate in\n * {@link DEFAULT_JOURNAL_DIR} until the sandbox dies, which on a `keepAlive`\n * sandbox may be never. This module is the sweep that bounds them, driven from a\n * cron or a reaper rather than from a run.\n *\n * **Why deleting is dangerous, and therefore why almost every branch keeps.**\n * The journal is the ONLY copy of the bytes a successor host needs to replay a\n * run a dead host abandoned mid-flight. Delete a live run's journal and that run\n * becomes unresumable — silently, because the reader will simply deliver nothing.\n * There is no undo and no second copy. So the decision procedure here is not\n * \"delete unless I have a reason to keep\"; it is the opposite, and every arm that\n * is not a PROVEN-safe deletion keeps:\n *\n * | the store says… | action | why |\n * | ---------------------------------- | ------ | --- |\n * | terminal (`isTerminalRunStatus`) | DELETE | the delivery log, not the journal, is the record |\n * | non-terminal, INCLUDING `'interrupted'` | KEEP | an interrupt-resume continues from it |\n * | nothing (unknown runId) | KEEP until `orphanTtlMs` | the reader creates the journal BEFORE the record exists |\n * | the lookup threw | KEEP | never delete on an unanswered question |\n * | (the name did not decode) | KEEP | a truncated name decodes to a plausible WRONG runId |\n * | (no mtime listing) | KEEP every age-gated entry | cannot age-gate ⇒ cannot expire |\n *\n * Deleting a TERMINAL run's journal is safe because a late takeover of a terminal\n * run aligns against the delivery LOG, not the journal: `align.ts`'s\n * `alignToStoredLog` takes a `StreamDurability` plus an\n * `AsyncIterable<StreamChunk>`, has no `SandboxHandle` and no `JournalPaths` in\n * its signature, and reads the already-delivered prefix with\n * `durability.snapshot()`. It *cannot* read a journal, so removing one cannot\n * break it. A non-zero exit is terminal too — `{\"__exit\":7}` is as final as\n * `{\"__exit\":0}`.\n *\n * The unknown-runId arm is the subtle one, and it is why an age gate exists at\n * all. `journalFollowCommand` opens the journal with `: >> file`, which CREATES\n * it; the reader and the run record are written by two independent code paths and\n * nothing orders them. So \"a journal exists whose runId the store has never heard\n * of\" is the NORMAL state of a run that started moments ago, not an anomaly.\n * Treating unknown as deletable would race every single run start. The journal is\n * therefore kept until it has been untouched for `orphanTtlMs`, which is the only\n * evidence available that no one is writing to it.\n *\n * **The fail-closed trap this module exists to not fall into.** BusyBox `find`\n * prints its \"unrecognized option\" diagnostic to *stderr* and exits **1 with\n * empty stdout**. A capability probe that ignores the exit code reads that as \"no\n * files matched\", i.e. \"no file is newer than the cutoff\" — and code that then\n * concludes \"therefore every file is old\" **deletes the entire directory**, live\n * runs included. {@link parseJournalMtimeListing} is built to make that\n * impossible: it passes the directory as `stat`'s own first operand as a\n * self-witness and returns `{ kind: 'unavailable' }` when that witness line is\n * absent, never `[]`. This module's whole obligation on that front is to honor\n * `unavailable` as \"I cannot age-gate, so I keep\" rather than as an empty\n * listing. See the `age-gate-unavailable` reason.\n *\n * **Shell only, never `handle.fs.*`.** On local-process, `fs.*` resolves `/tmp`\n * under the sandbox root while a shell redirect hits the real host `/tmp`, so an\n * `fs.remove` would delete a DIFFERENT path than the one `journaledCommand`\n * wrote — silently doing nothing while reporting success. Every filesystem touch\n * here goes through `handle.process.exec` with a command composed in\n * `journal.ts`.\n */\nimport { isTerminalRunStatus } from '@tanstack/ai'\nimport {\n DEFAULT_JOURNAL_DIR,\n decodeJournalRunId,\n journalCleanupCommand,\n journalListCommand,\n journalMtimeListCommand,\n journalPaths,\n parseJournalMtimeListing,\n} from './journal'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { RunStore } from '@tanstack/ai'\nimport type { SandboxHandle } from './contracts'\n\n/**\n * How long a journal whose runId the store does not know must go untouched\n * before the sweep will delete it.\n *\n * One hour, chosen against what the window actually protects: the gap between a\n * reader creating the journal with `: >> file` and the run record appearing in\n * the store. That gap is milliseconds in the normal case and seconds in the worst\n * case (a slow store, a retried write). An hour is three orders of magnitude of\n * headroom on the race, while still bounding a leaked journal to something a\n * sandbox's disk survives. Erring long is the cheap direction: the cost of too\n * long is bytes, the cost of too short is a destroyed live run.\n */\nexport const DEFAULT_ORPHAN_TTL_MS = 60 * 60 * 1000\n\n/**\n * Ceiling on deletions per sweep. A cron-driven sweep runs unattended, so a\n * mistake — a store that answers `terminal` for everything, a misconfigured\n * directory — is bounded by this rather than by how many journals happen to\n * exist. The remainder is reported as kept with reason `max-deletes` and picked\n * up by the next sweep.\n */\nexport const DEFAULT_MAX_DELETES = 200\n\n/** Why {@link pruneJournals} left a journal in place. */\nexport type KeptJournalReason =\n /** The store answered with a non-terminal status (`'running'`, `'interrupted'`). */\n | 'non-terminal'\n /** The store has never heard of this runId and the journal is still fresh. */\n | 'orphan-too-recent'\n /**\n * The store has never heard of this runId and the age gate could not run at\n * all — {@link parseJournalMtimeListing} returned `unavailable`. THE\n * FAIL-CLOSED ARM: an unavailable listing is not an empty one and says nothing\n * about any file's age.\n */\n | 'age-gate-unavailable'\n /**\n * The age gate ran but reported no mtime for this file, so its age is unknown.\n * (A file created between the two `exec`s, or a name the glob missed.)\n */\n | 'age-gate-missing-entry'\n /** {@link decodeJournalRunId} refused the name (`truncated` or `malformed`). */\n | 'undecodable-name'\n /** The store lookup threw. A question that was not answered is not a licence to delete. */\n | 'store-error'\n /** The `rm` itself failed or exited non-zero. */\n | 'delete-failed'\n /** {@link PruneJournalsOptions.maxDeletes} was already reached this sweep. */\n | 'max-deletes'\n\n/** One journal (or one runId's journal + sidecar) the sweep declined to delete. */\nexport interface KeptJournal {\n /** The decoded runId; absent exactly when `reason` is `'undecodable-name'`. */\n runId?: string\n /** Every listed filename this entry covers — the journal and its `.err` sidecar. */\n names: Array<string>\n reason: KeptJournalReason\n}\n\n/** A non-fatal failure the sweep folded into its result instead of throwing. */\nexport interface PruneJournalsFailure {\n stage: 'list' | 'mtime-list' | 'store' | 'delete'\n /** Present when the failure is attributable to one run. */\n runId?: string\n message: string\n}\n\n/** What one {@link pruneJournals} sweep did. */\nexport interface PruneJournalsResult {\n /** Filenames `ls -1` reported, before de-duplication by runId. */\n listed: number\n /** Distinct runIds those filenames decoded to. */\n runIds: number\n /** runIds whose journal AND sidecar were deleted, in the order deleted. */\n deleted: Array<string>\n /** Everything left in place, with the reason. */\n kept: Array<KeptJournal>\n /**\n * Whether the mtime age gate was usable this sweep. `'unavailable'` means no\n * orphan could be expired, by design.\n */\n ageGate: 'listed' | 'unavailable'\n failures: Array<PruneJournalsFailure>\n}\n\nexport interface PruneJournalsOptions {\n /** Sandbox holding the journal directory. Touched only via `process.exec`. */\n handle: SandboxHandle\n /**\n * Run lookup. Only `get` is used: the sweep asks about the runIds it found on\n * disk and never enumerates the store, so no optional `RunStore` method is\n * required of a backend.\n */\n runs: Pick<RunStore, 'get'>\n /** Journal directory. Defaults to {@link DEFAULT_JOURNAL_DIR}. */\n dir?: string\n /** Age-gate reference time. Defaults to `Date.now()`; injectable for tests. */\n now?: number\n /** See {@link DEFAULT_ORPHAN_TTL_MS}. */\n orphanTtlMs?: number\n /** See {@link DEFAULT_MAX_DELETES}. */\n maxDeletes?: number\n logger?: InternalLogger\n}\n\nfunction errorMessage(error: unknown): string {\n return error instanceof Error ? error.message : String(error)\n}\n\n/**\n * Group listed filenames by the runId they decode to, so a journal and its\n * `.err` sidecar are ONE decision and ONE `rm`, not two.\n *\n * De-duplication is not a tidiness measure: `journalCleanupCommand` deletes both\n * paths for a runId at once, so iterating raw names would ask the store twice per\n * run and then issue a second `rm` for files the first one already removed —\n * doubling the store load and reporting one run as two deletions.\n */\nfunction groupByRunId(names: Array<string>): {\n byRunId: Map<string, Array<string>>\n undecodable: Array<string>\n} {\n const byRunId = new Map<string, Array<string>>()\n const undecodable: Array<string> = []\n for (const name of names) {\n const decoded = decodeJournalRunId(name)\n if (decoded.kind !== 'runId') {\n undecodable.push(name)\n continue\n }\n const existing = byRunId.get(decoded.runId)\n if (existing === undefined) byRunId.set(decoded.runId, [name])\n else existing.push(name)\n }\n return { byRunId, undecodable }\n}\n\n/**\n * Sweep the journal directory, deleting only journals whose runs the store\n * reports terminal (plus orphans that have been untouched past `orphanTtlMs`).\n *\n * **Never rejects.** This runs unattended from a cron, where a rejected promise\n * is an unhandled rejection and, worse, hides which journals were and were not\n * swept. Every failure — a listing that errored, a store that threw, an `rm` that\n * exited non-zero — is folded into\n * {@link PruneJournalsResult.failures} and the sweep continues with the entries\n * it can still decide about.\n */\nexport async function pruneJournals(\n options: PruneJournalsOptions,\n): Promise<PruneJournalsResult> {\n const dir = options.dir ?? DEFAULT_JOURNAL_DIR\n const now = options.now ?? Date.now()\n const orphanTtlMs = options.orphanTtlMs ?? DEFAULT_ORPHAN_TTL_MS\n const maxDeletes = options.maxDeletes ?? DEFAULT_MAX_DELETES\n const logger = options.logger\n\n const deleted: Array<string> = []\n const kept: Array<KeptJournal> = []\n const failures: Array<PruneJournalsFailure> = []\n\n // `ls -1` is the authoritative name list. It is a SEPARATE command from the\n // mtime listing on purpose: `stat -c` may not exist on the provider's\n // busybox, and a sweep that could not enumerate at all when the age gate is\n // unavailable would never delete the terminal journals it is safe to delete.\n let names: Array<string> = []\n try {\n const listing = await options.handle.process.exec(journalListCommand(dir))\n names = listing.stdout\n .split('\\n')\n .map((line) => line.trim())\n .filter((line) => line !== '')\n } catch (error) {\n // Nothing was enumerated, so nothing can be deleted. Report and stop —\n // there is no partial-listing arm, because a partial listing is\n // indistinguishable from a complete one and we only ever DELETE from it.\n failures.push({ stage: 'list', message: errorMessage(error) })\n logger?.warn('journal sweep: listing the journal directory failed', {\n dir,\n error,\n })\n return {\n listed: 0,\n runIds: 0,\n deleted,\n kept,\n ageGate: 'unavailable',\n failures,\n }\n }\n\n // The age gate. `unavailable` is a first-class outcome, NOT an empty listing:\n // see the module doc's BusyBox `find` trap. It disables orphan expiry for this\n // sweep and disables nothing else.\n let ageGate: 'listed' | 'unavailable' = 'unavailable'\n const mtimes = new Map<string, number>()\n try {\n const probe = await options.handle.process.exec(\n journalMtimeListCommand(dir),\n )\n const parsed = parseJournalMtimeListing(probe.stdout, dir)\n if (parsed.kind === 'listed') {\n ageGate = 'listed'\n for (const entry of parsed.entries) mtimes.set(entry.name, entry.mtimeMs)\n } else {\n logger?.warn(\n 'journal sweep: mtime listing unavailable; keeping every orphan',\n { dir },\n )\n }\n } catch (error) {\n failures.push({ stage: 'mtime-list', message: errorMessage(error) })\n logger?.warn('journal sweep: mtime listing failed; keeping every orphan', {\n dir,\n error,\n })\n }\n\n const { byRunId, undecodable } = groupByRunId(names)\n\n // Undecodable names are kept unconditionally and without asking the store.\n // A truncated name decodes to a PLAUSIBLE BUT WRONG runId, so consulting the\n // store about it would answer a question about some other run — possibly a\n // live one — and a `terminal` answer would then delete this run's journal.\n for (const name of undecodable) {\n kept.push({ names: [name], reason: 'undecodable-name' })\n }\n\n const orphanCutoff = now - orphanTtlMs\n\n for (const [runId, runNames] of byRunId) {\n if (deleted.length >= maxDeletes) {\n kept.push({ runId, names: runNames, reason: 'max-deletes' })\n continue\n }\n\n let record: Awaited<ReturnType<RunStore['get']>>\n try {\n record = await options.runs.get(runId)\n } catch (error) {\n failures.push({ stage: 'store', runId, message: errorMessage(error) })\n logger?.warn('journal sweep: run lookup failed; keeping the journal', {\n runId,\n error,\n })\n kept.push({ runId, names: runNames, reason: 'store-error' })\n continue\n }\n\n if (record === null) {\n // Unknown to the store: either the record has not been written yet (the\n // normal case for a run that just started) or it was deleted after the\n // run ended. Only age distinguishes them.\n if (ageGate === 'unavailable') {\n kept.push({ runId, names: runNames, reason: 'age-gate-unavailable' })\n continue\n }\n const observed = runNames.map((name) => mtimes.get(name))\n if (observed.some((mtimeMs) => mtimeMs === undefined)) {\n kept.push({ runId, names: runNames, reason: 'age-gate-missing-entry' })\n continue\n }\n // The NEWEST of the run's files decides: a journal whose sidecar was\n // written a second ago is being written to, whatever the journal's own\n // mtime says.\n const newest = Math.max(...observed.filter(isDefined))\n if (newest > orphanCutoff) {\n kept.push({ runId, names: runNames, reason: 'orphan-too-recent' })\n continue\n }\n } else if (!isTerminalRunStatus(record.status)) {\n // `'interrupted'` lands here, and must: it is a human-in-the-loop PAUSE\n // that interrupt-resume continues from, not an end state.\n kept.push({ runId, names: runNames, reason: 'non-terminal' })\n continue\n }\n\n // Shell `rm`, never `handle.fs.remove`: module doc, and `journalPaths`\n // re-derives byte-identical paths from the runId alone.\n const command = journalCleanupCommand(journalPaths(runId, dir))\n try {\n const result = await options.handle.process.exec(command)\n if (result.exitCode !== 0) {\n failures.push({\n stage: 'delete',\n runId,\n message: `rm exited ${result.exitCode}`,\n })\n kept.push({ runId, names: runNames, reason: 'delete-failed' })\n continue\n }\n } catch (error) {\n // A failed cleanup must never fail the sweep: the journal is still there\n // and the next sweep will see it again.\n failures.push({ stage: 'delete', runId, message: errorMessage(error) })\n logger?.warn('journal sweep: deleting a journal failed', { runId, error })\n kept.push({ runId, names: runNames, reason: 'delete-failed' })\n continue\n }\n deleted.push(runId)\n }\n\n logger?.sandbox('journal sweep complete', {\n dir,\n listed: names.length,\n runIds: byRunId.size,\n deleted: deleted.length,\n kept: kept.length,\n ageGate,\n })\n\n return {\n listed: names.length,\n runIds: byRunId.size,\n deleted,\n kept,\n ageGate,\n failures,\n }\n}\n\n/** Narrowing predicate: `Array<number | undefined>` → `Array<number>`. */\nfunction isDefined(value: number | undefined): value is number {\n return value !== undefined\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA6FA,IAAa,wBAAwB,OAAU;;;;;;;;AAS/C,IAAa,sBAAsB;AAoFnC,SAAS,aAAa,OAAwB;CAC5C,OAAO,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK;AAC9D;;;;;;;;;;AAWA,SAAS,aAAa,OAGpB;CACA,MAAM,0BAAU,IAAI,IAA2B;CAC/C,MAAM,cAA6B,CAAC;CACpC,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,UAAU,mBAAmB,IAAI;EACvC,IAAI,QAAQ,SAAS,SAAS;GAC5B,YAAY,KAAK,IAAI;GACrB;EACF;EACA,MAAM,WAAW,QAAQ,IAAI,QAAQ,KAAK;EAC1C,IAAI,aAAa,KAAA,GAAW,QAAQ,IAAI,QAAQ,OAAO,CAAC,IAAI,CAAC;OACxD,SAAS,KAAK,IAAI;CACzB;CACA,OAAO;EAAE;EAAS;CAAY;AAChC;;;;;;;;;;;;AAaA,eAAsB,cACpB,SAC8B;CAC9B,MAAM,MAAM,QAAQ,OAAA;CACpB,MAAM,MAAM,QAAQ,OAAO,KAAK,IAAI;CACpC,MAAM,cAAc,QAAQ,eAAA;CAC5B,MAAM,aAAa,QAAQ,cAAA;CAC3B,MAAM,SAAS,QAAQ;CAEvB,MAAM,UAAyB,CAAC;CAChC,MAAM,OAA2B,CAAC;CAClC,MAAM,WAAwC,CAAC;CAM/C,IAAI,QAAuB,CAAC;CAC5B,IAAI;EAEF,SAAQ,MADc,QAAQ,OAAO,QAAQ,KAAK,mBAAmB,GAAG,CAAC,EAAA,CACzD,OACb,MAAM,IAAI,CAAC,CACX,KAAK,SAAS,KAAK,KAAK,CAAC,CAAC,CAC1B,QAAQ,SAAS,SAAS,EAAE;CACjC,SAAS,OAAO;EAId,SAAS,KAAK;GAAE,OAAO;GAAQ,SAAS,aAAa,KAAK;EAAE,CAAC;EAC7D,QAAQ,KAAK,uDAAuD;GAClE;GACA;EACF,CAAC;EACD,OAAO;GACL,QAAQ;GACR,QAAQ;GACR;GACA;GACA,SAAS;GACT;EACF;CACF;CAKA,IAAI,UAAoC;CACxC,MAAM,yBAAS,IAAI,IAAoB;CACvC,IAAI;EAIF,MAAM,SAAS,0BAAyB,MAHpB,QAAQ,OAAO,QAAQ,KACzC,wBAAwB,GAAG,CAC7B,EAAA,CAC8C,QAAQ,GAAG;EACzD,IAAI,OAAO,SAAS,UAAU;GAC5B,UAAU;GACV,KAAK,MAAM,SAAS,OAAO,SAAS,OAAO,IAAI,MAAM,MAAM,MAAM,OAAO;EAC1E,OACE,QAAQ,KACN,kEACA,EAAE,IAAI,CACR;CAEJ,SAAS,OAAO;EACd,SAAS,KAAK;GAAE,OAAO;GAAc,SAAS,aAAa,KAAK;EAAE,CAAC;EACnE,QAAQ,KAAK,6DAA6D;GACxE;GACA;EACF,CAAC;CACH;CAEA,MAAM,EAAE,SAAS,gBAAgB,aAAa,KAAK;CAMnD,KAAK,MAAM,QAAQ,aACjB,KAAK,KAAK;EAAE,OAAO,CAAC,IAAI;EAAG,QAAQ;CAAmB,CAAC;CAGzD,MAAM,eAAe,MAAM;CAE3B,KAAK,MAAM,CAAC,OAAO,aAAa,SAAS;EACvC,IAAI,QAAQ,UAAU,YAAY;GAChC,KAAK,KAAK;IAAE;IAAO,OAAO;IAAU,QAAQ;GAAc,CAAC;GAC3D;EACF;EAEA,IAAI;EACJ,IAAI;GACF,SAAS,MAAM,QAAQ,KAAK,IAAI,KAAK;EACvC,SAAS,OAAO;GACd,SAAS,KAAK;IAAE,OAAO;IAAS;IAAO,SAAS,aAAa,KAAK;GAAE,CAAC;GACrE,QAAQ,KAAK,yDAAyD;IACpE;IACA;GACF,CAAC;GACD,KAAK,KAAK;IAAE;IAAO,OAAO;IAAU,QAAQ;GAAc,CAAC;GAC3D;EACF;EAEA,IAAI,WAAW,MAAM;GAInB,IAAI,YAAY,eAAe;IAC7B,KAAK,KAAK;KAAE;KAAO,OAAO;KAAU,QAAQ;IAAuB,CAAC;IACpE;GACF;GACA,MAAM,WAAW,SAAS,KAAK,SAAS,OAAO,IAAI,IAAI,CAAC;GACxD,IAAI,SAAS,MAAM,YAAY,YAAY,KAAA,CAAS,GAAG;IACrD,KAAK,KAAK;KAAE;KAAO,OAAO;KAAU,QAAQ;IAAyB,CAAC;IACtE;GACF;GAKA,IADe,KAAK,IAAI,GAAG,SAAS,OAAO,SAAS,CAChD,IAAS,cAAc;IACzB,KAAK,KAAK;KAAE;KAAO,OAAO;KAAU,QAAQ;IAAoB,CAAC;IACjE;GACF;EACF,OAAO,IAAI,CAAC,oBAAoB,OAAO,MAAM,GAAG;GAG9C,KAAK,KAAK;IAAE;IAAO,OAAO;IAAU,QAAQ;GAAe,CAAC;GAC5D;EACF;EAIA,MAAM,UAAU,sBAAsB,aAAa,OAAO,GAAG,CAAC;EAC9D,IAAI;GACF,MAAM,SAAS,MAAM,QAAQ,OAAO,QAAQ,KAAK,OAAO;GACxD,IAAI,OAAO,aAAa,GAAG;IACzB,SAAS,KAAK;KACZ,OAAO;KACP;KACA,SAAS,aAAa,OAAO;IAC/B,CAAC;IACD,KAAK,KAAK;KAAE;KAAO,OAAO;KAAU,QAAQ;IAAgB,CAAC;IAC7D;GACF;EACF,SAAS,OAAO;GAGd,SAAS,KAAK;IAAE,OAAO;IAAU;IAAO,SAAS,aAAa,KAAK;GAAE,CAAC;GACtE,QAAQ,KAAK,4CAA4C;IAAE;IAAO;GAAM,CAAC;GACzE,KAAK,KAAK;IAAE;IAAO,OAAO;IAAU,QAAQ;GAAgB,CAAC;GAC7D;EACF;EACA,QAAQ,KAAK,KAAK;CACpB;CAEA,QAAQ,QAAQ,0BAA0B;EACxC;EACA,QAAQ,MAAM;EACd,QAAQ,QAAQ;EAChB,SAAS,QAAQ;EACjB,MAAM,KAAK;EACX;CACF,CAAC;CAED,OAAO;EACL,QAAQ,MAAM;EACd,QAAQ,QAAQ;EAChB;EACA;EACA;EACA;CACF;AACF;;AAGA,SAAS,UAAU,OAA4C;CAC7D,OAAO,UAAU,KAAA;AACnB"}
1
+ {"version":3,"file":"journal-sweep.js","names":[],"sources":["../../src/journal-sweep.ts"],"sourcesContent":["/**\n * Bound the journal directory: delete the journals nobody will ever read again,\n * and — far more importantly — refuse to delete anything else.\n *\n * `journalCleanupCommand` already deletes ONE run's journal at the moment its\n * `{\"__exit\":N}` sentinel is observed. That covers every run a host watched to\n * completion and covers nothing else: a run that reaches its sentinel while\n * DETACHED has no host reading its journal, so nothing observes the sentinel and\n * nothing calls the cleanup. Those journals accumulate in\n * {@link DEFAULT_JOURNAL_DIR} until the sandbox dies, which on a `keepAlive`\n * sandbox may be never. This module is the sweep that bounds them, driven from a\n * cron or a reaper rather than from a run.\n *\n * **Why deleting is dangerous, and therefore why almost every branch keeps.**\n * The journal is the ONLY copy of the bytes a successor host needs to replay a\n * run a dead host abandoned mid-flight. Delete a live run's journal and that run\n * becomes unresumable — silently, because the reader will simply deliver nothing.\n * There is no undo and no second copy. So the decision procedure here is not\n * \"delete unless I have a reason to keep\"; it is the opposite, and every arm that\n * is not a PROVEN-safe deletion keeps:\n *\n * | the store says… | action | why |\n * | ---------------------------------- | ------ | --- |\n * | terminal (`isTerminalRunStatus`) | DELETE | the delivery log, not the journal, is the record |\n * | non-terminal, INCLUDING `'interrupted'` | KEEP | an interrupt-resume continues from it |\n * | nothing (unknown runId) | KEEP until `orphanTtlMs` | the reader creates the journal BEFORE the record exists |\n * | the lookup threw | KEEP | never delete on an unanswered question |\n * | (the name did not decode) | KEEP | a truncated name decodes to a plausible WRONG runId |\n * | (no mtime listing) | KEEP every age-gated entry | cannot age-gate ⇒ cannot expire |\n *\n * Deleting a TERMINAL run's journal is safe because a late takeover of a terminal\n * run aligns against the delivery LOG, not the journal: `align.ts`'s\n * `alignToStoredLog` takes a `StreamDurability` plus an\n * `AsyncIterable<StreamChunk>`, has no `SandboxHandle` and no `JournalPaths` in\n * its signature, and reads the already-delivered prefix with\n * `durability.snapshot()`. It *cannot* read a journal, so removing one cannot\n * break it. A non-zero exit is terminal too — `{\"__exit\":7}` is as final as\n * `{\"__exit\":0}`.\n *\n * The unknown-runId arm is the subtle one, and it is why an age gate exists at\n * all. `journalFollowCommand` opens the journal with `: >> file`, which CREATES\n * it; the reader and the run record are written by two independent code paths and\n * nothing orders them. So \"a journal exists whose runId the store has never heard\n * of\" is the NORMAL state of a run that started moments ago, not an anomaly.\n * Treating unknown as deletable would race every single run start. The journal is\n * therefore kept until it has been untouched for `orphanTtlMs`, which is the only\n * evidence available that no one is writing to it.\n *\n * **The fail-closed trap this module exists to not fall into.** BusyBox `find`\n * prints its \"unrecognized option\" diagnostic to *stderr* and exits **1 with\n * empty stdout**. A capability probe that ignores the exit code reads that as \"no\n * files matched\", i.e. \"no file is newer than the cutoff\" — and code that then\n * concludes \"therefore every file is old\" **deletes the entire directory**, live\n * runs included. {@link parseJournalMtimeListing} is built to make that\n * impossible: it passes the directory as `stat`'s own first operand as a\n * self-witness and returns `{ kind: 'unavailable' }` when that witness line is\n * absent, never `[]`. This module's whole obligation on that front is to honor\n * `unavailable` as \"I cannot age-gate, so I keep\" rather than as an empty\n * listing. See the `age-gate-unavailable` reason.\n *\n * **Shell only, never `handle.fs.*`.** On local-process, `fs.*` resolves `/tmp`\n * under the sandbox root while a shell redirect hits the real host `/tmp`, so an\n * `fs.remove` would delete a DIFFERENT path than the one `journaledCommand`\n * wrote — silently doing nothing while reporting success. Every filesystem touch\n * here goes through `handle.process.exec` with a command composed in\n * `journal.ts`.\n */\nimport { isTerminalRunStatus } from '@tanstack/ai'\nimport {\n DEFAULT_JOURNAL_DIR,\n decodeJournalRunId,\n journalCleanupCommand,\n journalListCommand,\n journalMtimeListCommand,\n journalPaths,\n parseJournalMtimeListing,\n} from './journal'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { RunStore } from '@tanstack/ai'\nimport type { SandboxHandle } from './contracts'\n\n/**\n * How long a journal whose runId the store does not know must go untouched\n * before the sweep will delete it.\n *\n * One hour, chosen against what the window actually protects: the gap between a\n * reader creating the journal with `: >> file` and the run record appearing in\n * the store. That gap is milliseconds in the normal case and seconds in the worst\n * case (a slow store, a retried write). An hour is three orders of magnitude of\n * headroom on the race, while still bounding a leaked journal to something a\n * sandbox's disk survives. Erring long is the cheap direction: the cost of too\n * long is bytes, the cost of too short is a destroyed live run.\n */\nexport const DEFAULT_ORPHAN_TTL_MS = 60 * 60 * 1000\n\n/**\n * Ceiling on deletions per sweep. A cron-driven sweep runs unattended, so a\n * mistake — a store that answers `terminal` for everything, a misconfigured\n * directory — is bounded by this rather than by how many journals happen to\n * exist. The remainder is reported as kept with reason `max-deletes` and picked\n * up by the next sweep.\n */\nexport const DEFAULT_MAX_DELETES = 200\n\n/** Why {@link pruneJournals} left a journal in place. */\nexport type KeptJournalReason =\n /** The store answered with a non-terminal status (`'running'`, `'interrupted'`). */\n | 'non-terminal'\n /** The store has never heard of this runId and the journal is still fresh. */\n | 'orphan-too-recent'\n /**\n * The store has never heard of this runId and the age gate could not run at\n * all — {@link parseJournalMtimeListing} returned `unavailable`. THE\n * FAIL-CLOSED ARM: an unavailable listing is not an empty one and says nothing\n * about any file's age.\n */\n | 'age-gate-unavailable'\n /**\n * The age gate ran but reported no mtime for this file, so its age is unknown.\n * (A file created between the two `exec`s, or a name the glob missed.)\n */\n | 'age-gate-missing-entry'\n /** {@link decodeJournalRunId} refused the name (`truncated` or `malformed`). */\n | 'undecodable-name'\n /** The store lookup threw. A question that was not answered is not a licence to delete. */\n | 'store-error'\n /** The `rm` itself failed or exited non-zero. */\n | 'delete-failed'\n /** {@link PruneJournalsOptions.maxDeletes} was already reached this sweep. */\n | 'max-deletes'\n\n/** One journal (or one runId's journal + sidecar) the sweep declined to delete. */\nexport interface KeptJournal {\n /** The decoded runId; absent exactly when `reason` is `'undecodable-name'`. */\n runId?: string\n /** Every listed filename this entry covers — the journal and its `.err` sidecar. */\n names: Array<string>\n reason: KeptJournalReason\n}\n\n/** A non-fatal failure the sweep folded into its result instead of throwing. */\nexport interface PruneJournalsFailure {\n stage: 'list' | 'mtime-list' | 'store' | 'delete'\n /** Present when the failure is attributable to one run. */\n runId?: string\n message: string\n}\n\n/** What one {@link pruneJournals} sweep did. */\nexport interface PruneJournalsResult {\n /** Filenames `ls -1` reported, before de-duplication by runId. */\n listed: number\n /** Distinct runIds those filenames decoded to. */\n runIds: number\n /** runIds whose journal AND sidecar were deleted, in the order deleted. */\n deleted: Array<string>\n /** Everything left in place, with the reason. */\n kept: Array<KeptJournal>\n /**\n * Whether the mtime age gate was usable this sweep. `'unavailable'` means no\n * orphan could be expired, by design.\n */\n ageGate: 'listed' | 'unavailable'\n failures: Array<PruneJournalsFailure>\n}\n\nexport interface PruneJournalsOptions {\n /** Sandbox holding the journal directory. Touched only via `process.exec`. */\n handle: SandboxHandle\n /**\n * Run lookup. Only `get` is used: the sweep asks about the runIds it found on\n * disk and never enumerates the store, so no optional `RunStore` method is\n * required of a backend.\n */\n runs: Pick<RunStore, 'get'>\n /** Journal directory. Defaults to {@link DEFAULT_JOURNAL_DIR}. */\n dir?: string\n /** Age-gate reference time. Defaults to `Date.now()`; injectable for tests. */\n now?: number\n /** See {@link DEFAULT_ORPHAN_TTL_MS}. */\n orphanTtlMs?: number\n /** See {@link DEFAULT_MAX_DELETES}. */\n maxDeletes?: number\n logger?: InternalLogger\n}\n\nfunction errorMessage(error: unknown): string {\n return error instanceof Error ? error.message : String(error)\n}\n\n/**\n * Group listed filenames by the runId they decode to, so a journal and its\n * `.err` sidecar are ONE decision and ONE `rm`, not two.\n *\n * De-duplication is not a tidiness measure: `journalCleanupCommand` deletes both\n * paths for a runId at once, so iterating raw names would ask the store twice per\n * run and then issue a second `rm` for files the first one already removed —\n * doubling the store load and reporting one run as two deletions.\n */\nfunction groupByRunId(names: Array<string>): {\n byRunId: Map<string, Array<string>>\n undecodable: Array<string>\n} {\n const byRunId = new Map<string, Array<string>>()\n const undecodable: Array<string> = []\n for (const name of names) {\n const decoded = decodeJournalRunId(name)\n if (decoded.kind !== 'runId') {\n undecodable.push(name)\n continue\n }\n const existing = byRunId.get(decoded.runId)\n if (existing === undefined) byRunId.set(decoded.runId, [name])\n else existing.push(name)\n }\n return { byRunId, undecodable }\n}\n\n/**\n * Sweep the journal directory, deleting only journals whose runs the store\n * reports terminal (plus orphans that have been untouched past `orphanTtlMs`).\n *\n * **Never rejects.** This runs unattended from a cron, where a rejected promise\n * is an unhandled rejection and, worse, hides which journals were and were not\n * swept. Every failure — a listing that errored, a store that threw, an `rm` that\n * exited non-zero — is folded into\n * {@link PruneJournalsResult.failures} and the sweep continues with the entries\n * it can still decide about.\n */\nexport async function pruneJournals(\n options: PruneJournalsOptions,\n): Promise<PruneJournalsResult> {\n const dir = options.dir ?? DEFAULT_JOURNAL_DIR\n const now = options.now ?? Date.now()\n const orphanTtlMs = options.orphanTtlMs ?? DEFAULT_ORPHAN_TTL_MS\n const maxDeletes = options.maxDeletes ?? DEFAULT_MAX_DELETES\n const logger = options.logger\n\n const deleted: Array<string> = []\n const kept: Array<KeptJournal> = []\n const failures: Array<PruneJournalsFailure> = []\n\n // `ls -1` is the authoritative name list. It is a SEPARATE command from the\n // mtime listing on purpose: `stat -c` may not exist on the provider's\n // busybox, and a sweep that could not enumerate at all when the age gate is\n // unavailable would never delete the terminal journals it is safe to delete.\n let names: Array<string> = []\n try {\n const listing = await options.handle.process.exec(journalListCommand(dir))\n names = listing.stdout\n .split('\\n')\n .map((line) => line.trim())\n .filter((line) => line !== '')\n } catch (error) {\n // Nothing was enumerated, so nothing can be deleted. Report and stop —\n // there is no partial-listing arm, because a partial listing is\n // indistinguishable from a complete one and we only ever DELETE from it.\n failures.push({ stage: 'list', message: errorMessage(error) })\n logger?.warn('journal sweep: listing the journal directory failed', {\n dir,\n error,\n })\n return {\n listed: 0,\n runIds: 0,\n deleted,\n kept,\n ageGate: 'unavailable',\n failures,\n }\n }\n\n // The age gate. `unavailable` is a first-class outcome, NOT an empty listing:\n // see the module doc's BusyBox `find` trap. It disables orphan expiry for this\n // sweep and disables nothing else.\n let ageGate: 'listed' | 'unavailable' = 'unavailable'\n const mtimes = new Map<string, number>()\n try {\n const probe = await options.handle.process.exec(\n journalMtimeListCommand(dir),\n )\n const parsed = parseJournalMtimeListing(probe.stdout, dir)\n if (parsed.kind === 'listed') {\n ageGate = 'listed'\n for (const entry of parsed.entries) mtimes.set(entry.name, entry.mtimeMs)\n } else {\n logger?.warn(\n 'journal sweep: mtime listing unavailable; keeping every orphan',\n { dir },\n )\n }\n } catch (error) {\n failures.push({ stage: 'mtime-list', message: errorMessage(error) })\n logger?.warn('journal sweep: mtime listing failed; keeping every orphan', {\n dir,\n error,\n })\n }\n\n const { byRunId, undecodable } = groupByRunId(names)\n\n // Undecodable names are kept unconditionally and without asking the store.\n // A truncated name decodes to a PLAUSIBLE BUT WRONG runId, so consulting the\n // store about it would answer a question about some other run — possibly a\n // live one — and a `terminal` answer would then delete this run's journal.\n for (const name of undecodable) {\n kept.push({ names: [name], reason: 'undecodable-name' })\n }\n\n const orphanCutoff = now - orphanTtlMs\n\n for (const [runId, runNames] of byRunId) {\n if (deleted.length >= maxDeletes) {\n kept.push({ runId, names: runNames, reason: 'max-deletes' })\n continue\n }\n\n let record: Awaited<ReturnType<RunStore['get']>>\n try {\n record = await options.runs.get(runId)\n } catch (error) {\n failures.push({ stage: 'store', runId, message: errorMessage(error) })\n logger?.warn('journal sweep: run lookup failed; keeping the journal', {\n runId,\n error,\n })\n kept.push({ runId, names: runNames, reason: 'store-error' })\n continue\n }\n\n if (record === null) {\n // Unknown to the store: either the record has not been written yet (the\n // normal case for a run that just started) or it was deleted after the\n // run ended. Only age distinguishes them.\n if (ageGate === 'unavailable') {\n kept.push({ runId, names: runNames, reason: 'age-gate-unavailable' })\n continue\n }\n const observed = runNames.map((name) => mtimes.get(name))\n if (observed.some((mtimeMs) => mtimeMs === undefined)) {\n kept.push({ runId, names: runNames, reason: 'age-gate-missing-entry' })\n continue\n }\n // The NEWEST of the run's files decides: a journal whose sidecar was\n // written a second ago is being written to, whatever the journal's own\n // mtime says.\n const newest = Math.max(...observed.filter(isDefined))\n if (newest > orphanCutoff) {\n kept.push({ runId, names: runNames, reason: 'orphan-too-recent' })\n continue\n }\n } else if (!isTerminalRunStatus(record.status)) {\n // `'interrupted'` lands here, and must: it is a human-in-the-loop PAUSE\n // that interrupt-resume continues from, not an end state.\n kept.push({ runId, names: runNames, reason: 'non-terminal' })\n continue\n }\n\n // Shell `rm`, never `handle.fs.remove`: module doc, and `journalPaths`\n // re-derives byte-identical paths from the runId alone.\n const command = journalCleanupCommand(journalPaths(runId, dir))\n try {\n const result = await options.handle.process.exec(command)\n if (result.exitCode !== 0) {\n failures.push({\n stage: 'delete',\n runId,\n message: `rm exited ${result.exitCode}`,\n })\n kept.push({ runId, names: runNames, reason: 'delete-failed' })\n continue\n }\n } catch (error) {\n // A failed cleanup must never fail the sweep: the journal is still there\n // and the next sweep will see it again.\n failures.push({ stage: 'delete', runId, message: errorMessage(error) })\n logger?.warn('journal sweep: deleting a journal failed', { runId, error })\n kept.push({ runId, names: runNames, reason: 'delete-failed' })\n continue\n }\n deleted.push(runId)\n }\n\n logger?.sandbox('journal sweep complete', {\n dir,\n listed: names.length,\n runIds: byRunId.size,\n deleted: deleted.length,\n kept: kept.length,\n ageGate,\n })\n\n return {\n listed: names.length,\n runIds: byRunId.size,\n deleted,\n kept,\n ageGate,\n failures,\n }\n}\n\n/** Narrowing predicate: `Array<number | undefined>` → `Array<number>`. */\nfunction isDefined(value: number | undefined): value is number {\n return value !== undefined\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA6FA,IAAa,wBAAwB;;;;;;;;AASrC,IAAa,sBAAsB;AAoFnC,SAAS,aAAa,OAAwB;CAC5C,OAAO,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK;AAC9D;;;;;;;;;;AAWA,SAAS,aAAa,OAGpB;CACA,MAAM,0BAAU,IAAI,IAA2B;CAC/C,MAAM,cAA6B,CAAC;CACpC,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,UAAU,mBAAmB,IAAI;EACvC,IAAI,QAAQ,SAAS,SAAS;GAC5B,YAAY,KAAK,IAAI;GACrB;EACF;EACA,MAAM,WAAW,QAAQ,IAAI,QAAQ,KAAK;EAC1C,IAAI,aAAa,KAAA,GAAW,QAAQ,IAAI,QAAQ,OAAO,CAAC,IAAI,CAAC;OACxD,SAAS,KAAK,IAAI;CACzB;CACA,OAAO;EAAE;EAAS;CAAY;AAChC;;;;;;;;;;;;AAaA,eAAsB,cACpB,SAC8B;CAC9B,MAAM,MAAM,QAAQ,OAAA;CACpB,MAAM,MAAM,QAAQ,OAAO,KAAK,IAAI;CACpC,MAAM,cAAc,QAAQ,eAAA;CAC5B,MAAM,aAAa,QAAQ,cAAA;CAC3B,MAAM,SAAS,QAAQ;CAEvB,MAAM,UAAyB,CAAC;CAChC,MAAM,OAA2B,CAAC;CAClC,MAAM,WAAwC,CAAC;CAM/C,IAAI,QAAuB,CAAC;CAC5B,IAAI;EAEF,SAAQ,MADc,QAAQ,OAAO,QAAQ,KAAK,mBAAmB,GAAG,CAAC,EAAA,CACzD,OACb,MAAM,IAAI,CAAC,CACX,KAAK,SAAS,KAAK,KAAK,CAAC,CAAC,CAC1B,QAAQ,SAAS,SAAS,EAAE;CACjC,SAAS,OAAO;EAId,SAAS,KAAK;GAAE,OAAO;GAAQ,SAAS,aAAa,KAAK;EAAE,CAAC;EAC7D,QAAQ,KAAK,uDAAuD;GAClE;GACA;EACF,CAAC;EACD,OAAO;GACL,QAAQ;GACR,QAAQ;GACR;GACA;GACA,SAAS;GACT;EACF;CACF;CAKA,IAAI,UAAoC;CACxC,MAAM,yBAAS,IAAI,IAAoB;CACvC,IAAI;EACF,MAAM,QAAQ,MAAM,QAAQ,OAAO,QAAQ,KACzC,wBAAwB,GAAG,CAC7B;EACA,MAAM,SAAS,yBAAyB,MAAM,QAAQ,GAAG;EACzD,IAAI,OAAO,SAAS,UAAU;GAC5B,UAAU;GACV,KAAK,MAAM,SAAS,OAAO,SAAS,OAAO,IAAI,MAAM,MAAM,MAAM,OAAO;EAC1E,OACE,QAAQ,KACN,kEACA,EAAE,IAAI,CACR;CAEJ,SAAS,OAAO;EACd,SAAS,KAAK;GAAE,OAAO;GAAc,SAAS,aAAa,KAAK;EAAE,CAAC;EACnE,QAAQ,KAAK,6DAA6D;GACxE;GACA;EACF,CAAC;CACH;CAEA,MAAM,EAAE,SAAS,gBAAgB,aAAa,KAAK;CAMnD,KAAK,MAAM,QAAQ,aACjB,KAAK,KAAK;EAAE,OAAO,CAAC,IAAI;EAAG,QAAQ;CAAmB,CAAC;CAGzD,MAAM,eAAe,MAAM;CAE3B,KAAK,MAAM,CAAC,OAAO,aAAa,SAAS;EACvC,IAAI,QAAQ,UAAU,YAAY;GAChC,KAAK,KAAK;IAAE;IAAO,OAAO;IAAU,QAAQ;GAAc,CAAC;GAC3D;EACF;EAEA,IAAI;EACJ,IAAI;GACF,SAAS,MAAM,QAAQ,KAAK,IAAI,KAAK;EACvC,SAAS,OAAO;GACd,SAAS,KAAK;IAAE,OAAO;IAAS;IAAO,SAAS,aAAa,KAAK;GAAE,CAAC;GACrE,QAAQ,KAAK,yDAAyD;IACpE;IACA;GACF,CAAC;GACD,KAAK,KAAK;IAAE;IAAO,OAAO;IAAU,QAAQ;GAAc,CAAC;GAC3D;EACF;EAEA,IAAI,WAAW,MAAM;GAInB,IAAI,YAAY,eAAe;IAC7B,KAAK,KAAK;KAAE;KAAO,OAAO;KAAU,QAAQ;IAAuB,CAAC;IACpE;GACF;GACA,MAAM,WAAW,SAAS,KAAK,SAAS,OAAO,IAAI,IAAI,CAAC;GACxD,IAAI,SAAS,MAAM,YAAY,YAAY,KAAA,CAAS,GAAG;IACrD,KAAK,KAAK;KAAE;KAAO,OAAO;KAAU,QAAQ;IAAyB,CAAC;IACtE;GACF;GAKA,IADe,KAAK,IAAI,GAAG,SAAS,OAAO,SAAS,CAChD,IAAS,cAAc;IACzB,KAAK,KAAK;KAAE;KAAO,OAAO;KAAU,QAAQ;IAAoB,CAAC;IACjE;GACF;EACF,OAAO,IAAI,CAAC,oBAAoB,OAAO,MAAM,GAAG;GAG9C,KAAK,KAAK;IAAE;IAAO,OAAO;IAAU,QAAQ;GAAe,CAAC;GAC5D;EACF;EAIA,MAAM,UAAU,sBAAsB,aAAa,OAAO,GAAG,CAAC;EAC9D,IAAI;GACF,MAAM,SAAS,MAAM,QAAQ,OAAO,QAAQ,KAAK,OAAO;GACxD,IAAI,OAAO,aAAa,GAAG;IACzB,SAAS,KAAK;KACZ,OAAO;KACP;KACA,SAAS,aAAa,OAAO;IAC/B,CAAC;IACD,KAAK,KAAK;KAAE;KAAO,OAAO;KAAU,QAAQ;IAAgB,CAAC;IAC7D;GACF;EACF,SAAS,OAAO;GAGd,SAAS,KAAK;IAAE,OAAO;IAAU;IAAO,SAAS,aAAa,KAAK;GAAE,CAAC;GACtE,QAAQ,KAAK,4CAA4C;IAAE;IAAO;GAAM,CAAC;GACzE,KAAK,KAAK;IAAE;IAAO,OAAO;IAAU,QAAQ;GAAgB,CAAC;GAC7D;EACF;EACA,QAAQ,KAAK,KAAK;CACpB;CAEA,QAAQ,QAAQ,0BAA0B;EACxC;EACA,QAAQ,MAAM;EACd,QAAQ,QAAQ;EAChB,SAAS,QAAQ;EACjB,MAAM,KAAK;EACX;CACF,CAAC;CAED,OAAO;EACL,QAAQ,MAAM;EACd,QAAQ,QAAQ;EAChB;EACA;EACA;EACA;CACF;AACF;;AAGA,SAAS,UAAU,OAA4C;CAC7D,OAAO,UAAU,KAAA;AACnB"}
@@ -230,7 +230,8 @@ function withSandbox(definition, options) {
230
230
  }
231
231
  const workspace = definition.workspace;
232
232
  if (workspace !== void 0) {
233
- const root = resolveHarnessCwd(handle, workspace.root ?? "/workspace");
233
+ const virtualRoot = workspace.root ?? "/workspace";
234
+ const root = resolveHarnessCwd(handle, virtualRoot);
234
235
  const workspaceHash = computeWorkspaceHash(workspace);
235
236
  const secrets = workspace.secrets;
236
237
  provideWorkspaceProjection(ctx, {
@@ -1 +1 @@
1
- {"version":3,"file":"middleware.js","names":[],"sources":["../../src/middleware.ts"],"sourcesContent":["/**\n * `withSandbox(definition, options?)` — the middleware that PROVIDES the\n * {@link SandboxCapability} a harness adapter requires.\n *\n * - `setup`: resume-or-create the sandbox (via the definition's ensure\n * algorithm), provide the handle, using the durability seams from\n * {@link SandboxMiddlewareOptions} (or, failing that, a bus-provided\n * SandboxInstanceStoreCapability / LocksCapability, then an in-memory\n * fallback). If `fileEvents` is not false, starts a\n * watcher that dispatches to sandbox-scoped hooks and forwards to the runtime\n * sink.\n * - `onFinish`/`onAbort`/`onError`: stop the watcher, snapshot (`after-run`)\n * and/or destroy per lifecycle.\n *\n * NOTE: streamed sandbox lifecycle events (sandbox.created, workspace.setup.*)\n * are emitted by the harness adapter's chatStream (which can yield CUSTOM\n * chunks), not from here — middleware setup runs before streaming begins.\n */\nimport {\n defineChatMiddleware,\n provideDetachableRun,\n provideRunDetached,\n wasCancelRequested,\n} from '@tanstack/ai'\nimport { InMemoryLockStore, LocksCapability } from '@tanstack/ai/locks'\nimport {\n getPendingTurn,\n getRunDisconnect,\n getSandboxRuntime,\n} from '@tanstack/ai/adapter-internals'\nimport {\n SandboxCapability,\n provideSandbox,\n provideSandboxPolicy,\n} from './capabilities'\nimport {\n provideSandboxDurability,\n resolveSandboxDurability,\n} from './durability'\nimport { SandboxInstanceStoreCapability } from './instance-store'\nimport { computeWorkspaceHash } from './key'\nimport { buildFileHookEvent, resolveFileEvents } from './file-diff'\nimport { ProjectionCapability, provideWorkspaceProjection } from './projection'\nimport { resolveSecret } from './secrets'\nimport {\n createToolHistoryRecorder,\n stripObservedToolCalls,\n} from './tool-history'\nimport { watchWorkspace } from './watch'\nimport { DEFAULT_WORKSPACE_ROOT } from './bootstrap'\nimport { resolveHarnessCwd } from './harness-cwd'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { LockStore } from '@tanstack/ai/locks'\nimport type {\n AbortInfo,\n ChatMiddlewareContext,\n DefinedChatMiddleware,\n RunStore,\n SandboxFileEvent,\n SandboxFileHookEvent,\n} from '@tanstack/ai'\nimport type {\n SandboxDurabilityOptions,\n SandboxRunDurability,\n} from './durability'\nimport type { SandboxInstanceStore } from './instance-store'\nimport type { ToolHistoryRecorder } from './tool-history'\nimport type { SandboxHandle } from './contracts'\nimport type {\n SandboxDefinition,\n SandboxEnsureContext,\n SandboxHooks,\n} from './sandbox'\nimport type { SandboxWatchHandle } from './watch'\n\n/** Per-request state we need to carry from `setup` to the terminal hooks. */\ninterface SandboxRunState {\n /**\n * OPTIONAL because the state is registered BEFORE `definition.ensure()` is\n * awaited, and `ensure` is the slowest thing in the whole run — cloning a repo\n * into a fresh sandbox is minutes wide. That window is where the most common\n * disconnect of all lands (a user starts a run and switches away while the UI\n * still says \"starting the sandbox\"), so it is the one window the teardown and\n * disconnect hooks most need to be able to act in. Registering only after the\n * handle exists left exactly that window uncovered.\n *\n * Nothing the disconnect path does needs the handle: `detachedSince` and\n * `sandboxKey` come from `ensureCtx`, which is built before `ensure` is called.\n * Only `onFinish`'s snapshot needs it, and that cannot run before `setup` has\n * completed.\n */\n handle?: SandboxHandle\n ensureCtx: SandboxEnsureContext\n watcher?: SandboxWatchHandle\n /** In-flight `enriched.diff()` promises queued by the `fileEvents.diff`\n * watcher callback, awaited before teardown so a pending diff isn't\n * dropped when the run finishes/aborts/errors mid-computation. */\n pendingDiffs: Array<Promise<void>>\n /** Logger captured at setup, so terminal hooks can log watcher teardown. */\n logger?: InternalLogger\n /**\n * Durability resolved once at setup (absent when the run is not durable), so\n * `onAbort` cannot reach a different verdict than the one `setup` published\n * on the capability bus.\n */\n durability?: SandboxRunDurability\n /**\n * Records the harness's own tool calls into the transcript, so a finished run\n * restores its tool cards from the message store instead of only from the (live,\n * rejoin-only) delivery log. See `./tool-history`.\n */\n toolHistory: ToolHistoryRecorder\n}\n\nconst runState = new WeakMap<object, SandboxRunState>()\n\n/**\n * Stop the watcher and drain any in-flight `diff()` promises before teardown,\n * so the final file's diff isn't dropped when a run finishes/aborts/errors\n * mid-computation. The `pendingDiffs` await is the load-bearing line — without\n * it a deferred diff resolves after the run is gone and its chunk is lost.\n */\nasync function drainWatcher(\n state: SandboxRunState,\n phase: 'finish' | 'abort' | 'error',\n): Promise<void> {\n // Guard `stop()`: a rejecting watcher teardown must NOT propagate out of\n // here, or the caller skips the `definition.destroy(...)` that follows —\n // leaking the sandbox on exactly the abort path that must ALWAYS tear down.\n try {\n await state.watcher?.stop()\n } catch (error) {\n state.logger?.warn('sandbox watcher stop failed', { phase, error })\n }\n await Promise.allSettled(state.pendingDiffs)\n if (state.watcher) state.logger?.sandbox('sandbox watcher stopped', { phase })\n}\n\n/**\n * Record the two facts a later attach and the reaper both need, then publish the\n * detach verdict core reads.\n *\n * Shared by the DISCONNECT subscriber registered in `setup` (the run is still\n * going — the normal case) and `onAbort`'s detach branch (the run is being torn\n * down while detachable), so the two can never write a different shape of detach.\n *\n * GUARDED, and reports failure rather than throwing. `update` is a documented\n * no-op for an unknown runId, so a vanished record does not turn teardown into a\n * throw; a genuinely rejecting store is the caller's to react to — `onAbort` falls\n * through to destroying the sandbox, because a DESTROYED sandbox beats an\n * unreachable one, while the disconnect subscriber has nothing to fall back to\n * (the run is alive and still using the sandbox) and simply leaves the verdict\n * unpublished.\n *\n * The verdict is published ONLY on success. Publishing it after a failed record\n * write would leave core holding the log open for a takeover that can never be\n * found, since nothing in the store points at the run.\n */\nasync function recordDetach(\n definition: SandboxDefinition,\n state: SandboxRunState,\n durability: SandboxRunDurability,\n ctx: ChatMiddlewareContext,\n phase: 'disconnect' | 'abort',\n): Promise<boolean> {\n try {\n // The record already exists: `setup` pre-creates it for every durable run\n // BEFORE `ensure`, precisely so this stamp cannot land on a runId the store has\n // never heard of — `RunStore.update` is a documented no-op for an unknown\n // runId, which is how the detach used to be lost silently (measured against the\n // browser repro: `detached_since` and `sandbox_key` both stayed NULL for a run\n // that had genuinely detached). If it has since vanished, that no-op is the\n // correct outcome and this must not throw.\n await durability.runs.update(ctx.runId, {\n detachedSince: Date.now(),\n sandboxKey: definition.key(state.ensureCtx),\n })\n } catch (error) {\n state.logger?.warn('sandbox detach record write failed', {\n runId: ctx.runId,\n phase,\n error,\n })\n return false\n }\n // Core's durable delivery sink reads this (see `RunDetachedCapability`) and\n // leaves the run's log OPEN instead of appending a synthetic terminal\n // `RUN_ERROR` and closing it — a terminalized log ends a later attach's replay\n // at the prefix and diverges the takeover's journal replay, which recorded a\n // healthy detached run as `'failed'`.\n provideRunDetached(ctx, true)\n return true\n}\n\n/**\n * Whether an out-of-band cancel has been recorded for this run, in EITHER band.\n * A user pressing Stop and a user closing the tab produce the IDENTICAL\n * connection close, so intent is never inferred from the disconnect itself: it\n * arrives in-process (the abort reason carried the cancel sentinel) or durably\n * (another host recorded it on the run record).\n */\nasync function cancelIntent(\n durability: SandboxRunDurability | undefined,\n runId: string,\n inProcess: boolean,\n): Promise<boolean> {\n if (inProcess) return true\n if (durability === undefined) return false\n // No guard needed here, and one would be dead code: `wasCancelRequested` already\n // answers `false` for a store read that rejects. That matters on this path,\n // because a rejection escaping into `onAbort` would skip BOTH of its branches at\n // once, leaving a sandbox that is neither reclaimable nor destroyed. The test\n // 'DETACHES when the cancel probe REJECTS' pins the composition.\n return wasCancelRequested(durability.runs, runId)\n}\n\n/** Defensively pull tenant scoping out of the runtime context, if present. */\nfunction tenantFrom(\n context: unknown,\n): { userId?: string; orgId?: string } | undefined {\n if (context === null || typeof context !== 'object') return undefined\n const c = context as Record<string, unknown>\n const userId = typeof c.userId === 'string' ? c.userId : undefined\n const orgId = typeof c.orgId === 'string' ? c.orgId : undefined\n if (userId === undefined && orgId === undefined) return undefined\n return { userId, orgId }\n}\n\n/**\n * Durability seams for a sandboxed run. Both are optional; each independently\n * falls back to a process-lifetime in-memory default, which is correct for a\n * single process but NOT across replicas.\n */\nexport interface SandboxMiddlewareOptions<TOffset extends string = string> {\n /**\n * Durable instance map (which provider sandbox to resume for a key). Pass\n * your own store to make resume survive across processes/replicas.\n *\n * Takes precedence over a store provided on the capability bus (see\n * `provideSandboxInstanceStore`), so the call site wins over ambient wiring.\n */\n instances?: SandboxInstanceStore\n /**\n * Distributed lock serializing resume-or-create for one key. Needed for\n * multi-replica correctness so two concurrent runs don't both create.\n *\n * Prefer `withLocks` from `@tanstack/ai/locks` when other middleware also\n * needs the lock; use this option to scope one to this sandbox. Takes\n * precedence over a bus-provided lock.\n */\n locks?: LockStore\n /**\n * Run lifecycle records. Pair with `durability.adapter` to make a run\n * DETACHABLE: a client disconnect then leaves the agent running and records\n * `detachedSince` instead of destroying the sandbox.\n *\n * Pass the SAME store chat persistence uses (`persistence.stores.runs`) so\n * one record describes the run instead of two that can disagree.\n *\n * Defaults to `undefined`: an app that passes neither this nor `durability`\n * keeps today's destroy-on-disconnect behavior exactly.\n */\n runs?: RunStore\n /**\n * Delivery durability for the run's event log, plus the journal and detach\n * knobs. Requires `runs`; either alone is not durable.\n *\n * `TOffset` is inferred from the adapter passed here, so a branded-cursor\n * backend (`durableStream`) wires without a cast and without the call site\n * ever naming the parameter.\n */\n durability?: SandboxDurabilityOptions<TOffset>\n}\n\n/**\n * Resolve the ensure seams. Precedence is explicit option → capability bus →\n * (in `ensure`) the in-memory fallback. The option wins because it is visible\n * at the call site; the bus remains for platform/framework injection.\n */\nfunction buildEnsureCtx(\n ctx: ChatMiddlewareContext,\n // Narrowed to the two seams it reads rather than taking the whole options\n // object: `SandboxMiddlewareOptions` is now generic in the durability offset,\n // and `SandboxMiddlewareOptions<TOffset>` is not assignable to\n // `SandboxMiddlewareOptions<string>`. Both members here are offset-free, so\n // the narrowing keeps this helper independent of that parameter entirely.\n options: Pick<SandboxMiddlewareOptions, 'instances' | 'locks'> | undefined,\n): SandboxEnsureContext {\n return {\n threadId: ctx.threadId,\n runId: ctx.runId,\n store:\n options?.instances ?? ctx.getOptional(SandboxInstanceStoreCapability),\n locks: options?.locks ?? ctx.getOptional(LocksCapability),\n tenant: tenantFrom(ctx.context),\n signal: ctx.signal,\n adapterName: ctx.provider,\n }\n}\n\n/**\n * Dispatch a sandbox file event to the per-type hooks declared on the\n * definition. Errors in individual hooks are swallowed so one bad hook\n * cannot break the run — but are logged under the `errors` category first, so\n * a throwing hook is observable (matching the run-scoped path in the engine\n * and the behavior the observability docs promise).\n */\nasync function dispatchDefinitionHooks(\n hooks: SandboxHooks | undefined,\n event: SandboxFileHookEvent,\n logger?: InternalLogger,\n): Promise<void> {\n if (!hooks) return\n const typed = (\n {\n create: 'onFileCreate',\n change: 'onFileChange',\n delete: 'onFileDelete',\n } as const\n )[event.type]\n for (const fn of [hooks.onFile, hooks[typed]]) {\n if (!fn) continue\n try {\n await fn(event)\n } catch (error) {\n // swallowed — one bad hook must not break the run — but logged so the\n // failure isn't invisible.\n logger?.errors('sandbox file hook failed', {\n path: event.path,\n type: event.type,\n error,\n })\n }\n }\n}\n\nexport function withSandbox<TOffset extends string = string>(\n definition: SandboxDefinition,\n options?: SandboxMiddlewareOptions<TOffset>,\n): DefinedChatMiddleware<\n unknown,\n readonly [],\n readonly [typeof SandboxCapability, typeof ProjectionCapability]\n> {\n return defineChatMiddleware({\n name: 'sandbox',\n provides: [SandboxCapability, ProjectionCapability],\n // SandboxPolicyCapability is provided conditionally (only when the\n // definition has a policy), so it is intentionally NOT declared here —\n // consumers read it via `getOptional`. SandboxDurabilityCapability and\n // DetachableRunCapability are conditional for the same reason (only when\n // `runs` + `durability` are both wired), so they are intentionally NOT\n // declared here either.\n optionalRequires: [SandboxInstanceStoreCapability, LocksCapability],\n\n async setup(ctx) {\n const ensureCtx = buildEnsureCtx(ctx, options)\n\n // Resolving here (not lazily on the abort path) is what keeps `setup` and\n // `onAbort` on one verdict: the payload the bus carries is the same object\n // the teardown path consults.\n // `TOffset` is passed explicitly: `options` is possibly `undefined` here,\n // so inference has nothing to work from on that branch and would fall\n // back to the `= string` default, re-erecting the very wall this\n // parameter exists to remove.\n const durability = resolveSandboxDurability<TOffset>(options)\n if (durability !== undefined) {\n provideSandboxDurability(ctx, durability)\n // A neutral boolean core owns, so `@tanstack/ai-persistence` can ask\n // \"is this run detachable?\" without depending on this package.\n provideDetachableRun(ctx, true)\n }\n\n // Pull the runtime (and its logger) up front so `baseSha` capture and\n // hook dispatch below can log through the same `sandbox`/`errors`\n // categories the engine uses.\n const runtime = getSandboxRuntime(ctx, { optional: true })\n const logger = runtime?.logger\n\n // REGISTER THE RUN STATE NOW — before `definition.ensure()`, not merely\n // before the end of `setup`.\n //\n // `onAbort` and the disconnect subscriber both need this state, so until\n // this map is populated they are silent no-ops. `ensure` is the LONGEST\n // await in the entire run (create a sandbox, clone a repo — minutes), and it\n // is where the most common disconnect of all lands: a user starts a run and\n // switches away while the UI still says \"starting the sandbox\". Registering\n // after `ensure` returned still left that whole window uncovered.\n //\n // Leaving it uncovered loses every teardown behavior at once: no\n // `detachedSince`/`sandboxKey`, so `listReclaimable` can never surface the\n // run and the reaper can never reclaim it; no `definition.destroy`, so the\n // sandbox leaks; and no detach verdict for core to read.\n //\n // Everything those hooks read is already resolved above: the ensure context\n // (which is all `definition.key` needs), the durability verdict, and the\n // logger. The fields discovered later (`handle`, `watcher`) are ASSIGNED onto\n // this same object as they become available, so the teardown path always\n // sees the most complete state that exists at the moment it runs.\n const state: SandboxRunState = {\n ensureCtx,\n pendingDiffs: [],\n toolHistory: createToolHistoryRecorder(),\n ...(logger ? { logger } : {}),\n ...(durability ? { durability } : {}),\n }\n runState.set(ctx, state)\n\n // MAKE THE RUN FINDABLE BEFORE `ensure`, not after the run finally streams.\n //\n // Chat persistence creates the run record from `onConfig`, which runs after\n // EVERY middleware `setup` — so for the whole of `definition.ensure` (create a\n // sandbox, clone a repo: minutes) the run has no record at all, and\n // `findActiveRun` answers \"no active run\" for a run that is demonstrably\n // starting. Measured: a status sidebar read straight off `findActiveRun`\n // reported `idle` for 6.5 minutes while the sandbox was being built, and a\n // client returning to the thread in that window had nothing to tell it a run\n // was in flight — so it rendered an empty pane instead of \"starting sandbox\".\n //\n // A crash in the same window is worse: no record means `listReclaimable` can\n // never surface the run, so the sandbox leaks with no recovery path.\n //\n // `createOrResume` is idempotent and never resurrects a finished run, so\n // persistence's own later call stays correct and simply finds this record.\n if (durability !== undefined) {\n try {\n await durability.runs.createOrResume({\n runId: ctx.runId,\n threadId: ctx.threadId,\n startedAt: Date.now(),\n })\n } catch (error) {\n // Best-effort: a store blip must not stop a run that is otherwise fine.\n // The run is simply invisible until persistence's own `onConfig` call.\n logger?.warn('sandbox run record pre-create failed', {\n runId: ctx.runId,\n error,\n })\n }\n\n // NO ATTACH MARKER HERE. A joiner does need a chunk in the log before the\n // harness has emitted anything — an empty log fails every joiner's\n // fast-fail (`memoryStream`'s first-chunk deadline, the client's rejoin\n // connect deadline) and flushes no HTTP headers, so a reload during\n // `ensure` reads a live run as gone. Core does it: a fresh durable producer\n // appends `RUN_ACCEPTED_EVENT` before the producer stream is first pulled,\n // for EVERY durable run rather than only sandboxed ones, and never on an\n // attach. A second marker from here would only land mid-stream in a run\n // that is already producing.\n\n // STORE THE USER'S TURN NOW, before `ensure` takes minutes.\n //\n // Chat persistence stores it from `onStart`, which runs after every\n // middleware `setup` — so without this the thread holds NOTHING for the\n // whole sandbox build. Measured: a reload during the build asked the server\n // for the conversation and got `{\"messages\":[],…}`, so the user saw no sign\n // of the message they had just sent, and a second device saw an empty\n // thread.\n //\n // The persistence layer owns WHAT gets stored (see `PendingTurnCapability`):\n // `saveThread` replaces the thread, so deciding the list here would risk\n // deleting the history. Absent when the app wires no persistence, which is\n // simply a run with no transcript to store.\n try {\n await getPendingTurn(ctx, { optional: true })?.snapshot()\n } catch (error) {\n // Best-effort: the run is still worth doing, and `onStart` stores the\n // turn again once setup completes.\n logger?.warn('sandbox pending-turn snapshot failed', {\n runId: ctx.runId,\n error,\n })\n }\n }\n\n // SUBSCRIBE BEFORE `ensure`, for the same reason the state is registered\n // before it: `ensure` is the minutes-wide await a disconnect actually lands\n // in. Core calls back immediately if the socket has already closed, so\n // subscribing here cannot miss a disconnect that beat us to it.\n //\n // This is what makes a durable run SURVIVE losing its viewer. The only route\n // a disconnect previously had into this middleware was the application\n // mirroring `request.signal` into `chat()`'s `abortController` — which aborts\n // the run, so `chat()` returned right after this `setup` and the harness\n // adapter's `chatStream` was never called: the agent in the sandbox we just\n // spent minutes creating was NEVER LAUNCHED, and no takeover could recover it\n // because an agent that never ran wrote no journal to replay.\n if (durability !== undefined && durability.detachOnDisconnect) {\n getRunDisconnect(ctx, { optional: true })?.subscribe(async () => {\n // BOOKKEEPING ONLY — the run is still executing. Deliberately absent:\n // `drainWatcher` (would blind a live agent's file events for the whole\n // remainder) and `definition.destroy` (the run is still using the\n // sandbox). Both belong to the terminal hooks, which still run exactly\n // once afterwards.\n //\n // A run with a cancel already recorded is left alone: that is `onAbort`'s\n // path, and stamping `detachedSince` on a deliberately-stopped run would\n // hand it to the reaper as reclaimable work.\n if (await cancelIntent(durability, ctx.runId, false)) return\n if (\n await recordDetach(definition, state, durability, ctx, 'disconnect')\n ) {\n state.logger?.sandbox(\n 'sandbox run detached on disconnect; the run continues',\n { runId: ctx.runId },\n )\n }\n })\n }\n\n const handle = await definition.ensure(ensureCtx)\n // MUTATE, don't re-`set`: a disconnect that landed during `ensure` already\n // captured this object.\n state.handle = handle\n provideSandbox(ctx, handle)\n if (definition.policy) provideSandboxPolicy(ctx, definition.policy)\n\n // Deliberately placed AFTER `logger` is in scope rather than next to the\n // `provideSandboxDurability` call above — there is no logger to warn\n // through until the runtime has been read.\n //\n // `ensureCtx.locks === undefined` counts as in-memory: `defineSandbox`'s\n // `ensure` falls back to a process-lifetime `InMemoryLockStore` when no\n // lock is wired, so an unwired lock has exactly the deficiency being\n // warned about — it is the MOST in-memory case, not an exempt one.\n if (\n durability !== undefined &&\n (ensureCtx.locks === undefined ||\n ensureCtx.locks instanceof InMemoryLockStore)\n ) {\n logger?.warn(\n 'sandbox durability is wired over an InMemoryLockStore: run claims are ' +\n 'serialized within this process only and the lease never signals loss, ' +\n 'so two hosts can drive one run and duplicate its event log. Use a ' +\n 'distributed LockStore via withLocks for any multi-replica deploy.',\n { runId: ctx.runId },\n )\n }\n\n const watchRoot = definition.workspace?.root ?? DEFAULT_WORKSPACE_ROOT\n let baseSha = ''\n try {\n const shaRes = await handle.process.exec('git rev-parse HEAD', {\n cwd: watchRoot,\n })\n if (shaRes.exitCode === 0) {\n baseSha = shaRes.stdout.trim()\n logger?.sandbox('sandbox git baseline captured', {\n root: watchRoot,\n baseSha,\n })\n } else {\n // Non-zero exit: either not a git repository (non-git workspace) or a\n // repo with no commits (no HEAD). Expected, but it silently degrades\n // every subsequent diff to a full-file add-patch, so surface it\n // under `sandbox` (with stderr) rather than leaving nothing to grep.\n logger?.sandbox('sandbox git baseline unavailable (non-zero exit)', {\n root: watchRoot,\n exitCode: shaRes.exitCode,\n stderr: shaRes.stderr,\n })\n }\n } catch (error) {\n // exec rejected (git not on PATH, exec seam broken) → baseSha stays ''\n // and accessors fall back, but this is a real anomaly, not a plain\n // non-git workspace, so warn.\n logger?.warn('sandbox git baseline capture failed', {\n root: watchRoot,\n error,\n })\n }\n\n const workspace = definition.workspace\n if (workspace !== undefined) {\n const virtualRoot = workspace.root ?? DEFAULT_WORKSPACE_ROOT\n const root = resolveHarnessCwd(handle, virtualRoot)\n const workspaceHash = computeWorkspaceHash(workspace)\n const secrets = workspace.secrets\n provideWorkspaceProjection(ctx, {\n skills: workspace.skills ?? [],\n plugins: workspace.plugins ?? [],\n resolveSecret: (ref) => {\n if (secrets === undefined) {\n throw new Error(\n `resolveSecret: no secrets defined on this workspace (ref: \"${ref.__secretName}\")`,\n )\n }\n return resolveSecret(secrets, ref)\n },\n markerPath: `${root}/.tanstack-projected-${workspaceHash}`,\n root,\n ...(workspace.scripts !== undefined\n ? { scripts: workspace.scripts }\n : {}),\n })\n }\n\n const hooks = definition.hooks\n await hooks?.onReady?.(handle)\n\n const fe = resolveFileEvents(definition.fileEvents)\n // THE SAME array the run state already holds, not a fresh one. The watcher\n // callback below closes over this reference, and `drainWatcher` awaits\n // `state.pendingDiffs` — a second array would silently drop every in-flight\n // diff from the teardown drain.\n const pendingDiffs = state.pendingDiffs\n let watcher: SandboxWatchHandle | undefined\n if (fe.enabled) {\n watcher = await watchWorkspace(handle, {\n onEvent: (event: SandboxFileEvent) => {\n const enriched = buildFileHookEvent(\n handle,\n watchRoot,\n baseSha,\n event,\n logger,\n )\n void dispatchDefinitionHooks(hooks, enriched, logger)\n runtime?.emit(enriched)\n if (fe.diff) {\n pendingDiffs.push(\n enriched\n .diff()\n .then((diff) => {\n runtime?.emitFileDiff({ path: event.path, diff })\n })\n .catch((error: unknown) => {\n logger?.warn('sandbox file diff emit failed', {\n path: event.path,\n error,\n })\n }),\n )\n }\n },\n // Watch the SAME root the enrichment layer relativizes against\n // (`buildFileHookEvent(handle, watchRoot, …)` and the `baseSha`\n // capture). Without this the watcher defaults to `/workspace` while\n // enrichment uses `watchRoot`, so a custom `workspace.root` makes the\n // two look at different directories and git pathspecs break.\n root: watchRoot,\n ...(ctx.signal !== undefined ? { signal: ctx.signal } : {}),\n ...(logger !== undefined ? { logger } : {}),\n })\n logger?.sandbox('sandbox watcher started', {\n root: watchRoot,\n diff: fe.diff,\n })\n }\n\n // MUTATE the object registered above rather than `set`-ing a second one: an\n // abort that landed mid-setup already captured a reference to it (and may\n // already be draining `pendingDiffs`), so replacing the entry would hand the\n // teardown path a different object than the watcher writes into.\n // `pendingDiffs` needs no copying — it IS `state.pendingDiffs`.\n if (watcher) state.watcher = watcher\n },\n\n // Keep the recorded tool history OUT of the request to the model. It is stored\n // history for the next turn, it names tools the provider was never given, and one\n // triage-sized run is hundreds of kilobytes — so replaying it is wasteful at best\n // and rejected at worst. `ctx.messages` keeps it (that is what gets stored and\n // rendered); only `config.messages` loses it.\n onConfig(_ctx, config) {\n const messages = stripObservedToolCalls(config.messages)\n if (messages.length === config.messages.length) return\n return { messages }\n },\n\n // The engine re-syncs `middlewareCtx.messages` from its own array once per agent\n // iteration, which drops whatever the recorder appended during the previous\n // iteration's stream. Restoring it here — AFTER that sync — is what makes a\n // multi-iteration run keep its full history without depending on where this\n // middleware sits relative to persistence in the middleware array.\n onIteration(ctx) {\n runState.get(ctx)?.toolHistory.reconcile(ctx)\n },\n\n // Record the harness's own tool calls as transcript messages. Observe only:\n // returning nothing passes every chunk through untouched.\n onChunk(ctx, chunk) {\n runState.get(ctx)?.toolHistory.observe(chunk, ctx)\n },\n\n async onFinish(ctx) {\n const state = runState.get(ctx)\n if (!state) return\n const { handle, ensureCtx } = state\n\n // Last chance before persistence writes the transcript. Only matters if a\n // config sync landed after the final tool chunk; the recorder is idempotent, so\n // in the normal case this changes nothing.\n state.toolHistory.reconcile(ctx)\n\n await drainWatcher(state, 'finish')\n\n const lifecycle = definition.lifecycle\n\n // `handle` is absent only if `setup` never got past `definition.ensure`, in\n // which case there is no sandbox to snapshot.\n if (\n lifecycle?.snapshot === 'after-run' &&\n handle?.capabilities.snapshots &&\n handle.snapshot\n ) {\n const snapshot = await handle.snapshot(`after-run-${ctx.runId}`)\n const store = ensureCtx.store\n if (store) {\n const key = definition.key(ensureCtx)\n const existing = await store.get(key)\n if (existing) {\n await store.upsert({\n ...existing,\n latestSnapshotId: snapshot.id,\n updatedAt: Date.now(),\n })\n }\n }\n }\n\n if (lifecycle?.destroyOnComplete) {\n await definition.destroy(ensureCtx)\n await definition.hooks?.onDestroy?.()\n }\n },\n\n async onAbort(ctx, info: AbortInfo) {\n const state = runState.get(ctx)\n if (!state) return\n\n // First on BOTH branches: a diff still in flight must be drained whether\n // the sandbox is about to be destroyed or merely detached, or the final\n // file's diff is dropped.\n await drainWatcher(state, 'abort')\n\n const durability = state.durability\n const cancelled = await cancelIntent(\n durability,\n ctx.runId,\n info.cancelRequested === true,\n )\n\n if (\n durability !== undefined &&\n !cancelled &&\n durability.detachOnDisconnect\n ) {\n // DETACH on the teardown path. Reached when the run is aborted for a\n // reason that is NOT an out-of-band cancel while detachable — a genuine\n // stop from elsewhere, or a host going down. The ordinary disconnect is\n // handled by the disconnect subscriber in `setup`, which does not end the\n // run at all.\n //\n // On a failed record write this branch is ABANDONED for the destroy one\n // below, because a rejection here is the worst shape available: the\n // verdict is unpublished, so core terminalizes the log and records a\n // healthy detached run as failed; `detachedSince`/`sandboxKey` are\n // unwritten, so `listReclaimable` can never surface the run and\n // `reapDetachedRuns` can never reclaim it. A DESTROYED sandbox beats an\n // unreachable one — the same reasoning `drainWatcher` applies to its own\n // guarded `stop()`.\n if (await recordDetach(definition, state, durability, ctx, 'abort')) {\n return\n }\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n return\n }\n\n // ALWAYS tear down on an explicit abort, regardless of `destroyOnComplete`.\n // The in-sandbox agent process is not killed by closing its IO stream\n // (e.g. a Docker exec survives client disconnect), so the only reliable way\n // to stop it — and the token/cost drain of its ongoing API calls — is to\n // destroy the sandbox (stop the container/VM). `keepAlive` /\n // `destroyOnComplete:false` governs *successful completion*, never cancel.\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n },\n\n async onError(ctx, info) {\n const state = runState.get(ctx)\n if (!state) return\n\n await drainWatcher(state, 'error')\n await definition.hooks?.onError?.(info.error)\n\n // On failure, only tear down when the lifecycle says so; otherwise leave\n // the sandbox for a resumed retry.\n if (definition.lifecycle?.destroyOnComplete) {\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n }\n },\n })\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAkHA,IAAM,2BAAW,IAAI,QAAiC;;;;;;;AAQtD,eAAe,aACb,OACA,OACe;CAIf,IAAI;EACF,MAAM,MAAM,SAAS,KAAK;CAC5B,SAAS,OAAO;EACd,MAAM,QAAQ,KAAK,+BAA+B;GAAE;GAAO;EAAM,CAAC;CACpE;CACA,MAAM,QAAQ,WAAW,MAAM,YAAY;CAC3C,IAAI,MAAM,SAAS,MAAM,QAAQ,QAAQ,2BAA2B,EAAE,MAAM,CAAC;AAC/E;;;;;;;;;;;;;;;;;;;;;AAsBA,eAAe,aACb,YACA,OACA,YACA,KACA,OACkB;CAClB,IAAI;EAQF,MAAM,WAAW,KAAK,OAAO,IAAI,OAAO;GACtC,eAAe,KAAK,IAAI;GACxB,YAAY,WAAW,IAAI,MAAM,SAAS;EAC5C,CAAC;CACH,SAAS,OAAO;EACd,MAAM,QAAQ,KAAK,sCAAsC;GACvD,OAAO,IAAI;GACX;GACA;EACF,CAAC;EACD,OAAO;CACT;CAMA,mBAAmB,KAAK,IAAI;CAC5B,OAAO;AACT;;;;;;;;AASA,eAAe,aACb,YACA,OACA,WACkB;CAClB,IAAI,WAAW,OAAO;CACtB,IAAI,eAAe,KAAA,GAAW,OAAO;CAMrC,OAAO,mBAAmB,WAAW,MAAM,KAAK;AAClD;;AAGA,SAAS,WACP,SACiD;CACjD,IAAI,YAAY,QAAQ,OAAO,YAAY,UAAU,OAAO,KAAA;CAC5D,MAAM,IAAI;CACV,MAAM,SAAS,OAAO,EAAE,WAAW,WAAW,EAAE,SAAS,KAAA;CACzD,MAAM,QAAQ,OAAO,EAAE,UAAU,WAAW,EAAE,QAAQ,KAAA;CACtD,IAAI,WAAW,KAAA,KAAa,UAAU,KAAA,GAAW,OAAO,KAAA;CACxD,OAAO;EAAE;EAAQ;CAAM;AACzB;;;;;;AAqDA,SAAS,eACP,KAMA,SACsB;CACtB,OAAO;EACL,UAAU,IAAI;EACd,OAAO,IAAI;EACX,OACE,SAAS,aAAa,IAAI,YAAY,8BAA8B;EACtE,OAAO,SAAS,SAAS,IAAI,YAAY,eAAe;EACxD,QAAQ,WAAW,IAAI,OAAO;EAC9B,QAAQ,IAAI;EACZ,aAAa,IAAI;CACnB;AACF;;;;;;;;AASA,eAAe,wBACb,OACA,OACA,QACe;CACf,IAAI,CAAC,OAAO;CACZ,MAAM,QACJ;EACE,QAAQ;EACR,QAAQ;EACR,QAAQ;CACV,EACA,MAAM;CACR,KAAK,MAAM,MAAM,CAAC,MAAM,QAAQ,MAAM,MAAM,GAAG;EAC7C,IAAI,CAAC,IAAI;EACT,IAAI;GACF,MAAM,GAAG,KAAK;EAChB,SAAS,OAAO;GAGd,QAAQ,OAAO,4BAA4B;IACzC,MAAM,MAAM;IACZ,MAAM,MAAM;IACZ;GACF,CAAC;EACH;CACF;AACF;AAEA,SAAgB,YACd,YACA,SAKA;CACA,OAAO,qBAAqB;EAC1B,MAAM;EACN,UAAU,CAAC,mBAAmB,oBAAoB;EAOlD,kBAAkB,CAAC,gCAAgC,eAAe;EAElE,MAAM,MAAM,KAAK;GACf,MAAM,YAAY,eAAe,KAAK,OAAO;GAS7C,MAAM,aAAa,yBAAkC,OAAO;GAC5D,IAAI,eAAe,KAAA,GAAW;IAC5B,yBAAyB,KAAK,UAAU;IAGxC,qBAAqB,KAAK,IAAI;GAChC;GAKA,MAAM,UAAU,kBAAkB,KAAK,EAAE,UAAU,KAAK,CAAC;GACzD,MAAM,SAAS,SAAS;GAsBxB,MAAM,QAAyB;IAC7B;IACA,cAAc,CAAC;IACf,aAAa,0BAA0B;IACvC,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC;IAC3B,GAAI,aAAa,EAAE,WAAW,IAAI,CAAC;GACrC;GACA,SAAS,IAAI,KAAK,KAAK;GAkBvB,IAAI,eAAe,KAAA,GAAW;IAC5B,IAAI;KACF,MAAM,WAAW,KAAK,eAAe;MACnC,OAAO,IAAI;MACX,UAAU,IAAI;MACd,WAAW,KAAK,IAAI;KACtB,CAAC;IACH,SAAS,OAAO;KAGd,QAAQ,KAAK,wCAAwC;MACnD,OAAO,IAAI;MACX;KACF,CAAC;IACH;IAyBA,IAAI;KACF,MAAM,eAAe,KAAK,EAAE,UAAU,KAAK,CAAC,CAAC,EAAE,SAAS;IAC1D,SAAS,OAAO;KAGd,QAAQ,KAAK,wCAAwC;MACnD,OAAO,IAAI;MACX;KACF,CAAC;IACH;GACF;GAcA,IAAI,eAAe,KAAA,KAAa,WAAW,oBACzC,iBAAiB,KAAK,EAAE,UAAU,KAAK,CAAC,CAAC,EAAE,UAAU,YAAY;IAU/D,IAAI,MAAM,aAAa,YAAY,IAAI,OAAO,KAAK,GAAG;IACtD,IACE,MAAM,aAAa,YAAY,OAAO,YAAY,KAAK,YAAY,GAEnE,MAAM,QAAQ,QACZ,yDACA,EAAE,OAAO,IAAI,MAAM,CACrB;GAEJ,CAAC;GAGH,MAAM,SAAS,MAAM,WAAW,OAAO,SAAS;GAGhD,MAAM,SAAS;GACf,eAAe,KAAK,MAAM;GAC1B,IAAI,WAAW,QAAQ,qBAAqB,KAAK,WAAW,MAAM;GAUlE,IACE,eAAe,KAAA,MACd,UAAU,UAAU,KAAA,KACnB,UAAU,iBAAiB,oBAE7B,QAAQ,KACN,mRAIA,EAAE,OAAO,IAAI,MAAM,CACrB;GAGF,MAAM,YAAY,WAAW,WAAW,QAAA;GACxC,IAAI,UAAU;GACd,IAAI;IACF,MAAM,SAAS,MAAM,OAAO,QAAQ,KAAK,sBAAsB,EAC7D,KAAK,UACP,CAAC;IACD,IAAI,OAAO,aAAa,GAAG;KACzB,UAAU,OAAO,OAAO,KAAK;KAC7B,QAAQ,QAAQ,iCAAiC;MAC/C,MAAM;MACN;KACF,CAAC;IACH,OAKE,QAAQ,QAAQ,oDAAoD;KAClE,MAAM;KACN,UAAU,OAAO;KACjB,QAAQ,OAAO;IACjB,CAAC;GAEL,SAAS,OAAO;IAId,QAAQ,KAAK,uCAAuC;KAClD,MAAM;KACN;IACF,CAAC;GACH;GAEA,MAAM,YAAY,WAAW;GAC7B,IAAI,cAAc,KAAA,GAAW;IAE3B,MAAM,OAAO,kBAAkB,QADX,UAAU,QAAA,YACoB;IAClD,MAAM,gBAAgB,qBAAqB,SAAS;IACpD,MAAM,UAAU,UAAU;IAC1B,2BAA2B,KAAK;KAC9B,QAAQ,UAAU,UAAU,CAAC;KAC7B,SAAS,UAAU,WAAW,CAAC;KAC/B,gBAAgB,QAAQ;MACtB,IAAI,YAAY,KAAA,GACd,MAAM,IAAI,MACR,8DAA8D,IAAI,aAAa,GACjF;MAEF,OAAO,cAAc,SAAS,GAAG;KACnC;KACA,YAAY,GAAG,KAAK,uBAAuB;KAC3C;KACA,GAAI,UAAU,YAAY,KAAA,IACtB,EAAE,SAAS,UAAU,QAAQ,IAC7B,CAAC;IACP,CAAC;GACH;GAEA,MAAM,QAAQ,WAAW;GACzB,MAAM,OAAO,UAAU,MAAM;GAE7B,MAAM,KAAK,kBAAkB,WAAW,UAAU;GAKlD,MAAM,eAAe,MAAM;GAC3B,IAAI;GACJ,IAAI,GAAG,SAAS;IACd,UAAU,MAAM,eAAe,QAAQ;KACrC,UAAU,UAA4B;MACpC,MAAM,WAAW,mBACf,QACA,WACA,SACA,OACA,MACF;MACA,wBAA6B,OAAO,UAAU,MAAM;MACpD,SAAS,KAAK,QAAQ;MACtB,IAAI,GAAG,MACL,aAAa,KACX,SACG,KAAK,CAAC,CACN,MAAM,SAAS;OACd,SAAS,aAAa;QAAE,MAAM,MAAM;QAAM;OAAK,CAAC;MAClD,CAAC,CAAC,CACD,OAAO,UAAmB;OACzB,QAAQ,KAAK,iCAAiC;QAC5C,MAAM,MAAM;QACZ;OACF,CAAC;MACH,CAAC,CACL;KAEJ;KAMA,MAAM;KACN,GAAI,IAAI,WAAW,KAAA,IAAY,EAAE,QAAQ,IAAI,OAAO,IAAI,CAAC;KACzD,GAAI,WAAW,KAAA,IAAY,EAAE,OAAO,IAAI,CAAC;IAC3C,CAAC;IACD,QAAQ,QAAQ,2BAA2B;KACzC,MAAM;KACN,MAAM,GAAG;IACX,CAAC;GACH;GAOA,IAAI,SAAS,MAAM,UAAU;EAC/B;EAOA,SAAS,MAAM,QAAQ;GACrB,MAAM,WAAW,uBAAuB,OAAO,QAAQ;GACvD,IAAI,SAAS,WAAW,OAAO,SAAS,QAAQ;GAChD,OAAO,EAAE,SAAS;EACpB;EAOA,YAAY,KAAK;GACf,SAAS,IAAI,GAAG,CAAC,EAAE,YAAY,UAAU,GAAG;EAC9C;EAIA,QAAQ,KAAK,OAAO;GAClB,SAAS,IAAI,GAAG,CAAC,EAAE,YAAY,QAAQ,OAAO,GAAG;EACnD;EAEA,MAAM,SAAS,KAAK;GAClB,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GACZ,MAAM,EAAE,QAAQ,cAAc;GAK9B,MAAM,YAAY,UAAU,GAAG;GAE/B,MAAM,aAAa,OAAO,QAAQ;GAElC,MAAM,YAAY,WAAW;GAI7B,IACE,WAAW,aAAa,eACxB,QAAQ,aAAa,aACrB,OAAO,UACP;IACA,MAAM,WAAW,MAAM,OAAO,SAAS,aAAa,IAAI,OAAO;IAC/D,MAAM,QAAQ,UAAU;IACxB,IAAI,OAAO;KACT,MAAM,MAAM,WAAW,IAAI,SAAS;KACpC,MAAM,WAAW,MAAM,MAAM,IAAI,GAAG;KACpC,IAAI,UACF,MAAM,MAAM,OAAO;MACjB,GAAG;MACH,kBAAkB,SAAS;MAC3B,WAAW,KAAK,IAAI;KACtB,CAAC;IAEL;GACF;GAEA,IAAI,WAAW,mBAAmB;IAChC,MAAM,WAAW,QAAQ,SAAS;IAClC,MAAM,WAAW,OAAO,YAAY;GACtC;EACF;EAEA,MAAM,QAAQ,KAAK,MAAiB;GAClC,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GAKZ,MAAM,aAAa,OAAO,OAAO;GAEjC,MAAM,aAAa,MAAM;GACzB,MAAM,YAAY,MAAM,aACtB,YACA,IAAI,OACJ,KAAK,oBAAoB,IAC3B;GAEA,IACE,eAAe,KAAA,KACf,CAAC,aACD,WAAW,oBACX;IAeA,IAAI,MAAM,aAAa,YAAY,OAAO,YAAY,KAAK,OAAO,GAChE;IAEF,MAAM,WAAW,QAAQ,MAAM,SAAS;IACxC,MAAM,WAAW,OAAO,YAAY;IACpC;GACF;GAQA,MAAM,WAAW,QAAQ,MAAM,SAAS;GACxC,MAAM,WAAW,OAAO,YAAY;EACtC;EAEA,MAAM,QAAQ,KAAK,MAAM;GACvB,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GAEZ,MAAM,aAAa,OAAO,OAAO;GACjC,MAAM,WAAW,OAAO,UAAU,KAAK,KAAK;GAI5C,IAAI,WAAW,WAAW,mBAAmB;IAC3C,MAAM,WAAW,QAAQ,MAAM,SAAS;IACxC,MAAM,WAAW,OAAO,YAAY;GACtC;EACF;CACF,CAAC;AACH"}
1
+ {"version":3,"file":"middleware.js","names":[],"sources":["../../src/middleware.ts"],"sourcesContent":["/**\n * `withSandbox(definition, options?)` — the middleware that PROVIDES the\n * {@link SandboxCapability} a harness adapter requires.\n *\n * - `setup`: resume-or-create the sandbox (via the definition's ensure\n * algorithm), provide the handle, using the durability seams from\n * {@link SandboxMiddlewareOptions} (or, failing that, a bus-provided\n * SandboxInstanceStoreCapability / LocksCapability, then an in-memory\n * fallback). If `fileEvents` is not false, starts a\n * watcher that dispatches to sandbox-scoped hooks and forwards to the runtime\n * sink.\n * - `onFinish`/`onAbort`/`onError`: stop the watcher, snapshot (`after-run`)\n * and/or destroy per lifecycle.\n *\n * NOTE: streamed sandbox lifecycle events (sandbox.created, workspace.setup.*)\n * are emitted by the harness adapter's chatStream (which can yield CUSTOM\n * chunks), not from here — middleware setup runs before streaming begins.\n */\nimport {\n defineChatMiddleware,\n provideDetachableRun,\n provideRunDetached,\n wasCancelRequested,\n} from '@tanstack/ai'\nimport { InMemoryLockStore, LocksCapability } from '@tanstack/ai/locks'\nimport {\n getPendingTurn,\n getRunDisconnect,\n getSandboxRuntime,\n} from '@tanstack/ai/adapter-internals'\nimport {\n SandboxCapability,\n provideSandbox,\n provideSandboxPolicy,\n} from './capabilities'\nimport {\n provideSandboxDurability,\n resolveSandboxDurability,\n} from './durability'\nimport { SandboxInstanceStoreCapability } from './instance-store'\nimport { computeWorkspaceHash } from './key'\nimport { buildFileHookEvent, resolveFileEvents } from './file-diff'\nimport { ProjectionCapability, provideWorkspaceProjection } from './projection'\nimport { resolveSecret } from './secrets'\nimport {\n createToolHistoryRecorder,\n stripObservedToolCalls,\n} from './tool-history'\nimport { watchWorkspace } from './watch'\nimport { DEFAULT_WORKSPACE_ROOT } from './bootstrap'\nimport { resolveHarnessCwd } from './harness-cwd'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { LockStore } from '@tanstack/ai/locks'\nimport type {\n AbortInfo,\n ChatMiddlewareContext,\n DefinedChatMiddleware,\n RunStore,\n SandboxFileEvent,\n SandboxFileHookEvent,\n} from '@tanstack/ai'\nimport type {\n SandboxDurabilityOptions,\n SandboxRunDurability,\n} from './durability'\nimport type { SandboxInstanceStore } from './instance-store'\nimport type { ToolHistoryRecorder } from './tool-history'\nimport type { SandboxHandle } from './contracts'\nimport type {\n SandboxDefinition,\n SandboxEnsureContext,\n SandboxHooks,\n} from './sandbox'\nimport type { SandboxWatchHandle } from './watch'\n\n/** Per-request state we need to carry from `setup` to the terminal hooks. */\ninterface SandboxRunState {\n /**\n * OPTIONAL because the state is registered BEFORE `definition.ensure()` is\n * awaited, and `ensure` is the slowest thing in the whole run — cloning a repo\n * into a fresh sandbox is minutes wide. That window is where the most common\n * disconnect of all lands (a user starts a run and switches away while the UI\n * still says \"starting the sandbox\"), so it is the one window the teardown and\n * disconnect hooks most need to be able to act in. Registering only after the\n * handle exists left exactly that window uncovered.\n *\n * Nothing the disconnect path does needs the handle: `detachedSince` and\n * `sandboxKey` come from `ensureCtx`, which is built before `ensure` is called.\n * Only `onFinish`'s snapshot needs it, and that cannot run before `setup` has\n * completed.\n */\n handle?: SandboxHandle\n ensureCtx: SandboxEnsureContext\n watcher?: SandboxWatchHandle\n /** In-flight `enriched.diff()` promises queued by the `fileEvents.diff`\n * watcher callback, awaited before teardown so a pending diff isn't\n * dropped when the run finishes/aborts/errors mid-computation. */\n pendingDiffs: Array<Promise<void>>\n /** Logger captured at setup, so terminal hooks can log watcher teardown. */\n logger?: InternalLogger\n /**\n * Durability resolved once at setup (absent when the run is not durable), so\n * `onAbort` cannot reach a different verdict than the one `setup` published\n * on the capability bus.\n */\n durability?: SandboxRunDurability\n /**\n * Records the harness's own tool calls into the transcript, so a finished run\n * restores its tool cards from the message store instead of only from the (live,\n * rejoin-only) delivery log. See `./tool-history`.\n */\n toolHistory: ToolHistoryRecorder\n}\n\nconst runState = new WeakMap<object, SandboxRunState>()\n\n/**\n * Stop the watcher and drain any in-flight `diff()` promises before teardown,\n * so the final file's diff isn't dropped when a run finishes/aborts/errors\n * mid-computation. The `pendingDiffs` await is the load-bearing line — without\n * it a deferred diff resolves after the run is gone and its chunk is lost.\n */\nasync function drainWatcher(\n state: SandboxRunState,\n phase: 'finish' | 'abort' | 'error',\n): Promise<void> {\n // Guard `stop()`: a rejecting watcher teardown must NOT propagate out of\n // here, or the caller skips the `definition.destroy(...)` that follows —\n // leaking the sandbox on exactly the abort path that must ALWAYS tear down.\n try {\n await state.watcher?.stop()\n } catch (error) {\n state.logger?.warn('sandbox watcher stop failed', { phase, error })\n }\n await Promise.allSettled(state.pendingDiffs)\n if (state.watcher) state.logger?.sandbox('sandbox watcher stopped', { phase })\n}\n\n/**\n * Record the two facts a later attach and the reaper both need, then publish the\n * detach verdict core reads.\n *\n * Shared by the DISCONNECT subscriber registered in `setup` (the run is still\n * going — the normal case) and `onAbort`'s detach branch (the run is being torn\n * down while detachable), so the two can never write a different shape of detach.\n *\n * GUARDED, and reports failure rather than throwing. `update` is a documented\n * no-op for an unknown runId, so a vanished record does not turn teardown into a\n * throw; a genuinely rejecting store is the caller's to react to — `onAbort` falls\n * through to destroying the sandbox, because a DESTROYED sandbox beats an\n * unreachable one, while the disconnect subscriber has nothing to fall back to\n * (the run is alive and still using the sandbox) and simply leaves the verdict\n * unpublished.\n *\n * The verdict is published ONLY on success. Publishing it after a failed record\n * write would leave core holding the log open for a takeover that can never be\n * found, since nothing in the store points at the run.\n */\nasync function recordDetach(\n definition: SandboxDefinition,\n state: SandboxRunState,\n durability: SandboxRunDurability,\n ctx: ChatMiddlewareContext,\n phase: 'disconnect' | 'abort',\n): Promise<boolean> {\n try {\n // The record already exists: `setup` pre-creates it for every durable run\n // BEFORE `ensure`, precisely so this stamp cannot land on a runId the store has\n // never heard of — `RunStore.update` is a documented no-op for an unknown\n // runId, which is how the detach used to be lost silently (measured against the\n // browser repro: `detached_since` and `sandbox_key` both stayed NULL for a run\n // that had genuinely detached). If it has since vanished, that no-op is the\n // correct outcome and this must not throw.\n await durability.runs.update(ctx.runId, {\n detachedSince: Date.now(),\n sandboxKey: definition.key(state.ensureCtx),\n })\n } catch (error) {\n state.logger?.warn('sandbox detach record write failed', {\n runId: ctx.runId,\n phase,\n error,\n })\n return false\n }\n // Core's durable delivery sink reads this (see `RunDetachedCapability`) and\n // leaves the run's log OPEN instead of appending a synthetic terminal\n // `RUN_ERROR` and closing it — a terminalized log ends a later attach's replay\n // at the prefix and diverges the takeover's journal replay, which recorded a\n // healthy detached run as `'failed'`.\n provideRunDetached(ctx, true)\n return true\n}\n\n/**\n * Whether an out-of-band cancel has been recorded for this run, in EITHER band.\n * A user pressing Stop and a user closing the tab produce the IDENTICAL\n * connection close, so intent is never inferred from the disconnect itself: it\n * arrives in-process (the abort reason carried the cancel sentinel) or durably\n * (another host recorded it on the run record).\n */\nasync function cancelIntent(\n durability: SandboxRunDurability | undefined,\n runId: string,\n inProcess: boolean,\n): Promise<boolean> {\n if (inProcess) return true\n if (durability === undefined) return false\n // No guard needed here, and one would be dead code: `wasCancelRequested` already\n // answers `false` for a store read that rejects. That matters on this path,\n // because a rejection escaping into `onAbort` would skip BOTH of its branches at\n // once, leaving a sandbox that is neither reclaimable nor destroyed. The test\n // 'DETACHES when the cancel probe REJECTS' pins the composition.\n return wasCancelRequested(durability.runs, runId)\n}\n\n/** Defensively pull tenant scoping out of the runtime context, if present. */\nfunction tenantFrom(\n context: unknown,\n): { userId?: string; orgId?: string } | undefined {\n if (context === null || typeof context !== 'object') return undefined\n const c = context as Record<string, unknown>\n const userId = typeof c.userId === 'string' ? c.userId : undefined\n const orgId = typeof c.orgId === 'string' ? c.orgId : undefined\n if (userId === undefined && orgId === undefined) return undefined\n return { userId, orgId }\n}\n\n/**\n * Durability seams for a sandboxed run. Both are optional; each independently\n * falls back to a process-lifetime in-memory default, which is correct for a\n * single process but NOT across replicas.\n */\nexport interface SandboxMiddlewareOptions<TOffset extends string = string> {\n /**\n * Durable instance map (which provider sandbox to resume for a key). Pass\n * your own store to make resume survive across processes/replicas.\n *\n * Takes precedence over a store provided on the capability bus (see\n * `provideSandboxInstanceStore`), so the call site wins over ambient wiring.\n */\n instances?: SandboxInstanceStore\n /**\n * Distributed lock serializing resume-or-create for one key. Needed for\n * multi-replica correctness so two concurrent runs don't both create.\n *\n * Prefer `withLocks` from `@tanstack/ai/locks` when other middleware also\n * needs the lock; use this option to scope one to this sandbox. Takes\n * precedence over a bus-provided lock.\n */\n locks?: LockStore\n /**\n * Run lifecycle records. Pair with `durability.adapter` to make a run\n * DETACHABLE: a client disconnect then leaves the agent running and records\n * `detachedSince` instead of destroying the sandbox.\n *\n * Pass the SAME store chat persistence uses (`persistence.stores.runs`) so\n * one record describes the run instead of two that can disagree.\n *\n * Defaults to `undefined`: an app that passes neither this nor `durability`\n * keeps today's destroy-on-disconnect behavior exactly.\n */\n runs?: RunStore\n /**\n * Delivery durability for the run's event log, plus the journal and detach\n * knobs. Requires `runs`; either alone is not durable.\n *\n * `TOffset` is inferred from the adapter passed here, so a branded-cursor\n * backend (`durableStream`) wires without a cast and without the call site\n * ever naming the parameter.\n */\n durability?: SandboxDurabilityOptions<TOffset>\n}\n\n/**\n * Resolve the ensure seams. Precedence is explicit option → capability bus →\n * (in `ensure`) the in-memory fallback. The option wins because it is visible\n * at the call site; the bus remains for platform/framework injection.\n */\nfunction buildEnsureCtx(\n ctx: ChatMiddlewareContext,\n // Narrowed to the two seams it reads rather than taking the whole options\n // object: `SandboxMiddlewareOptions` is now generic in the durability offset,\n // and `SandboxMiddlewareOptions<TOffset>` is not assignable to\n // `SandboxMiddlewareOptions<string>`. Both members here are offset-free, so\n // the narrowing keeps this helper independent of that parameter entirely.\n options: Pick<SandboxMiddlewareOptions, 'instances' | 'locks'> | undefined,\n): SandboxEnsureContext {\n return {\n threadId: ctx.threadId,\n runId: ctx.runId,\n store:\n options?.instances ?? ctx.getOptional(SandboxInstanceStoreCapability),\n locks: options?.locks ?? ctx.getOptional(LocksCapability),\n tenant: tenantFrom(ctx.context),\n signal: ctx.signal,\n adapterName: ctx.provider,\n }\n}\n\n/**\n * Dispatch a sandbox file event to the per-type hooks declared on the\n * definition. Errors in individual hooks are swallowed so one bad hook\n * cannot break the run — but are logged under the `errors` category first, so\n * a throwing hook is observable (matching the run-scoped path in the engine\n * and the behavior the observability docs promise).\n */\nasync function dispatchDefinitionHooks(\n hooks: SandboxHooks | undefined,\n event: SandboxFileHookEvent,\n logger?: InternalLogger,\n): Promise<void> {\n if (!hooks) return\n const typed = (\n {\n create: 'onFileCreate',\n change: 'onFileChange',\n delete: 'onFileDelete',\n } as const\n )[event.type]\n for (const fn of [hooks.onFile, hooks[typed]]) {\n if (!fn) continue\n try {\n await fn(event)\n } catch (error) {\n // swallowed — one bad hook must not break the run — but logged so the\n // failure isn't invisible.\n logger?.errors('sandbox file hook failed', {\n path: event.path,\n type: event.type,\n error,\n })\n }\n }\n}\n\nexport function withSandbox<TOffset extends string = string>(\n definition: SandboxDefinition,\n options?: SandboxMiddlewareOptions<TOffset>,\n): DefinedChatMiddleware<\n unknown,\n readonly [],\n readonly [typeof SandboxCapability, typeof ProjectionCapability]\n> {\n return defineChatMiddleware({\n name: 'sandbox',\n provides: [SandboxCapability, ProjectionCapability],\n // SandboxPolicyCapability is provided conditionally (only when the\n // definition has a policy), so it is intentionally NOT declared here —\n // consumers read it via `getOptional`. SandboxDurabilityCapability and\n // DetachableRunCapability are conditional for the same reason (only when\n // `runs` + `durability` are both wired), so they are intentionally NOT\n // declared here either.\n optionalRequires: [SandboxInstanceStoreCapability, LocksCapability],\n\n async setup(ctx) {\n const ensureCtx = buildEnsureCtx(ctx, options)\n\n // Resolving here (not lazily on the abort path) is what keeps `setup` and\n // `onAbort` on one verdict: the payload the bus carries is the same object\n // the teardown path consults.\n // `TOffset` is passed explicitly: `options` is possibly `undefined` here,\n // so inference has nothing to work from on that branch and would fall\n // back to the `= string` default, re-erecting the very wall this\n // parameter exists to remove.\n const durability = resolveSandboxDurability<TOffset>(options)\n if (durability !== undefined) {\n provideSandboxDurability(ctx, durability)\n // A neutral boolean core owns, so `@tanstack/ai-persistence` can ask\n // \"is this run detachable?\" without depending on this package.\n provideDetachableRun(ctx, true)\n }\n\n // Pull the runtime (and its logger) up front so `baseSha` capture and\n // hook dispatch below can log through the same `sandbox`/`errors`\n // categories the engine uses.\n const runtime = getSandboxRuntime(ctx, { optional: true })\n const logger = runtime?.logger\n\n // REGISTER THE RUN STATE NOW — before `definition.ensure()`, not merely\n // before the end of `setup`.\n //\n // `onAbort` and the disconnect subscriber both need this state, so until\n // this map is populated they are silent no-ops. `ensure` is the LONGEST\n // await in the entire run (create a sandbox, clone a repo — minutes), and it\n // is where the most common disconnect of all lands: a user starts a run and\n // switches away while the UI still says \"starting the sandbox\". Registering\n // after `ensure` returned still left that whole window uncovered.\n //\n // Leaving it uncovered loses every teardown behavior at once: no\n // `detachedSince`/`sandboxKey`, so `listReclaimable` can never surface the\n // run and the reaper can never reclaim it; no `definition.destroy`, so the\n // sandbox leaks; and no detach verdict for core to read.\n //\n // Everything those hooks read is already resolved above: the ensure context\n // (which is all `definition.key` needs), the durability verdict, and the\n // logger. The fields discovered later (`handle`, `watcher`) are ASSIGNED onto\n // this same object as they become available, so the teardown path always\n // sees the most complete state that exists at the moment it runs.\n const state: SandboxRunState = {\n ensureCtx,\n pendingDiffs: [],\n toolHistory: createToolHistoryRecorder(),\n ...(logger ? { logger } : {}),\n ...(durability ? { durability } : {}),\n }\n runState.set(ctx, state)\n\n // MAKE THE RUN FINDABLE BEFORE `ensure`, not after the run finally streams.\n //\n // Chat persistence creates the run record from `onConfig`, which runs after\n // EVERY middleware `setup` — so for the whole of `definition.ensure` (create a\n // sandbox, clone a repo: minutes) the run has no record at all, and\n // `findActiveRun` answers \"no active run\" for a run that is demonstrably\n // starting. Measured: a status sidebar read straight off `findActiveRun`\n // reported `idle` for 6.5 minutes while the sandbox was being built, and a\n // client returning to the thread in that window had nothing to tell it a run\n // was in flight — so it rendered an empty pane instead of \"starting sandbox\".\n //\n // A crash in the same window is worse: no record means `listReclaimable` can\n // never surface the run, so the sandbox leaks with no recovery path.\n //\n // `createOrResume` is idempotent and never resurrects a finished run, so\n // persistence's own later call stays correct and simply finds this record.\n if (durability !== undefined) {\n try {\n await durability.runs.createOrResume({\n runId: ctx.runId,\n threadId: ctx.threadId,\n startedAt: Date.now(),\n })\n } catch (error) {\n // Best-effort: a store blip must not stop a run that is otherwise fine.\n // The run is simply invisible until persistence's own `onConfig` call.\n logger?.warn('sandbox run record pre-create failed', {\n runId: ctx.runId,\n error,\n })\n }\n\n // NO ATTACH MARKER HERE. A joiner does need a chunk in the log before the\n // harness has emitted anything — an empty log fails every joiner's\n // fast-fail (`memoryStream`'s first-chunk deadline, the client's rejoin\n // connect deadline) and flushes no HTTP headers, so a reload during\n // `ensure` reads a live run as gone. Core does it: a fresh durable producer\n // appends `RUN_ACCEPTED_EVENT` before the producer stream is first pulled,\n // for EVERY durable run rather than only sandboxed ones, and never on an\n // attach. A second marker from here would only land mid-stream in a run\n // that is already producing.\n\n // STORE THE USER'S TURN NOW, before `ensure` takes minutes.\n //\n // Chat persistence stores it from `onStart`, which runs after every\n // middleware `setup` — so without this the thread holds NOTHING for the\n // whole sandbox build. Measured: a reload during the build asked the server\n // for the conversation and got `{\"messages\":[],…}`, so the user saw no sign\n // of the message they had just sent, and a second device saw an empty\n // thread.\n //\n // The persistence layer owns WHAT gets stored (see `PendingTurnCapability`):\n // `saveThread` replaces the thread, so deciding the list here would risk\n // deleting the history. Absent when the app wires no persistence, which is\n // simply a run with no transcript to store.\n try {\n await getPendingTurn(ctx, { optional: true })?.snapshot()\n } catch (error) {\n // Best-effort: the run is still worth doing, and `onStart` stores the\n // turn again once setup completes.\n logger?.warn('sandbox pending-turn snapshot failed', {\n runId: ctx.runId,\n error,\n })\n }\n }\n\n // SUBSCRIBE BEFORE `ensure`, for the same reason the state is registered\n // before it: `ensure` is the minutes-wide await a disconnect actually lands\n // in. Core calls back immediately if the socket has already closed, so\n // subscribing here cannot miss a disconnect that beat us to it.\n //\n // This is what makes a durable run SURVIVE losing its viewer. The only route\n // a disconnect previously had into this middleware was the application\n // mirroring `request.signal` into `chat()`'s `abortController` — which aborts\n // the run, so `chat()` returned right after this `setup` and the harness\n // adapter's `chatStream` was never called: the agent in the sandbox we just\n // spent minutes creating was NEVER LAUNCHED, and no takeover could recover it\n // because an agent that never ran wrote no journal to replay.\n if (durability !== undefined && durability.detachOnDisconnect) {\n getRunDisconnect(ctx, { optional: true })?.subscribe(async () => {\n // BOOKKEEPING ONLY — the run is still executing. Deliberately absent:\n // `drainWatcher` (would blind a live agent's file events for the whole\n // remainder) and `definition.destroy` (the run is still using the\n // sandbox). Both belong to the terminal hooks, which still run exactly\n // once afterwards.\n //\n // A run with a cancel already recorded is left alone: that is `onAbort`'s\n // path, and stamping `detachedSince` on a deliberately-stopped run would\n // hand it to the reaper as reclaimable work.\n if (await cancelIntent(durability, ctx.runId, false)) return\n if (\n await recordDetach(definition, state, durability, ctx, 'disconnect')\n ) {\n state.logger?.sandbox(\n 'sandbox run detached on disconnect; the run continues',\n { runId: ctx.runId },\n )\n }\n })\n }\n\n const handle = await definition.ensure(ensureCtx)\n // MUTATE, don't re-`set`: a disconnect that landed during `ensure` already\n // captured this object.\n state.handle = handle\n provideSandbox(ctx, handle)\n if (definition.policy) provideSandboxPolicy(ctx, definition.policy)\n\n // Deliberately placed AFTER `logger` is in scope rather than next to the\n // `provideSandboxDurability` call above — there is no logger to warn\n // through until the runtime has been read.\n //\n // `ensureCtx.locks === undefined` counts as in-memory: `defineSandbox`'s\n // `ensure` falls back to a process-lifetime `InMemoryLockStore` when no\n // lock is wired, so an unwired lock has exactly the deficiency being\n // warned about — it is the MOST in-memory case, not an exempt one.\n if (\n durability !== undefined &&\n (ensureCtx.locks === undefined ||\n ensureCtx.locks instanceof InMemoryLockStore)\n ) {\n logger?.warn(\n 'sandbox durability is wired over an InMemoryLockStore: run claims are ' +\n 'serialized within this process only and the lease never signals loss, ' +\n 'so two hosts can drive one run and duplicate its event log. Use a ' +\n 'distributed LockStore via withLocks for any multi-replica deploy.',\n { runId: ctx.runId },\n )\n }\n\n const watchRoot = definition.workspace?.root ?? DEFAULT_WORKSPACE_ROOT\n let baseSha = ''\n try {\n const shaRes = await handle.process.exec('git rev-parse HEAD', {\n cwd: watchRoot,\n })\n if (shaRes.exitCode === 0) {\n baseSha = shaRes.stdout.trim()\n logger?.sandbox('sandbox git baseline captured', {\n root: watchRoot,\n baseSha,\n })\n } else {\n // Non-zero exit: either not a git repository (non-git workspace) or a\n // repo with no commits (no HEAD). Expected, but it silently degrades\n // every subsequent diff to a full-file add-patch, so surface it\n // under `sandbox` (with stderr) rather than leaving nothing to grep.\n logger?.sandbox('sandbox git baseline unavailable (non-zero exit)', {\n root: watchRoot,\n exitCode: shaRes.exitCode,\n stderr: shaRes.stderr,\n })\n }\n } catch (error) {\n // exec rejected (git not on PATH, exec seam broken) → baseSha stays ''\n // and accessors fall back, but this is a real anomaly, not a plain\n // non-git workspace, so warn.\n logger?.warn('sandbox git baseline capture failed', {\n root: watchRoot,\n error,\n })\n }\n\n const workspace = definition.workspace\n if (workspace !== undefined) {\n const virtualRoot = workspace.root ?? DEFAULT_WORKSPACE_ROOT\n const root = resolveHarnessCwd(handle, virtualRoot)\n const workspaceHash = computeWorkspaceHash(workspace)\n const secrets = workspace.secrets\n provideWorkspaceProjection(ctx, {\n skills: workspace.skills ?? [],\n plugins: workspace.plugins ?? [],\n resolveSecret: (ref) => {\n if (secrets === undefined) {\n throw new Error(\n `resolveSecret: no secrets defined on this workspace (ref: \"${ref.__secretName}\")`,\n )\n }\n return resolveSecret(secrets, ref)\n },\n markerPath: `${root}/.tanstack-projected-${workspaceHash}`,\n root,\n ...(workspace.scripts !== undefined\n ? { scripts: workspace.scripts }\n : {}),\n })\n }\n\n const hooks = definition.hooks\n await hooks?.onReady?.(handle)\n\n const fe = resolveFileEvents(definition.fileEvents)\n // THE SAME array the run state already holds, not a fresh one. The watcher\n // callback below closes over this reference, and `drainWatcher` awaits\n // `state.pendingDiffs` — a second array would silently drop every in-flight\n // diff from the teardown drain.\n const pendingDiffs = state.pendingDiffs\n let watcher: SandboxWatchHandle | undefined\n if (fe.enabled) {\n watcher = await watchWorkspace(handle, {\n onEvent: (event: SandboxFileEvent) => {\n const enriched = buildFileHookEvent(\n handle,\n watchRoot,\n baseSha,\n event,\n logger,\n )\n void dispatchDefinitionHooks(hooks, enriched, logger)\n runtime?.emit(enriched)\n if (fe.diff) {\n pendingDiffs.push(\n enriched\n .diff()\n .then((diff) => {\n runtime?.emitFileDiff({ path: event.path, diff })\n })\n .catch((error: unknown) => {\n logger?.warn('sandbox file diff emit failed', {\n path: event.path,\n error,\n })\n }),\n )\n }\n },\n // Watch the SAME root the enrichment layer relativizes against\n // (`buildFileHookEvent(handle, watchRoot, …)` and the `baseSha`\n // capture). Without this the watcher defaults to `/workspace` while\n // enrichment uses `watchRoot`, so a custom `workspace.root` makes the\n // two look at different directories and git pathspecs break.\n root: watchRoot,\n ...(ctx.signal !== undefined ? { signal: ctx.signal } : {}),\n ...(logger !== undefined ? { logger } : {}),\n })\n logger?.sandbox('sandbox watcher started', {\n root: watchRoot,\n diff: fe.diff,\n })\n }\n\n // MUTATE the object registered above rather than `set`-ing a second one: an\n // abort that landed mid-setup already captured a reference to it (and may\n // already be draining `pendingDiffs`), so replacing the entry would hand the\n // teardown path a different object than the watcher writes into.\n // `pendingDiffs` needs no copying — it IS `state.pendingDiffs`.\n if (watcher) state.watcher = watcher\n },\n\n // Keep the recorded tool history OUT of the request to the model. It is stored\n // history for the next turn, it names tools the provider was never given, and one\n // triage-sized run is hundreds of kilobytes — so replaying it is wasteful at best\n // and rejected at worst. `ctx.messages` keeps it (that is what gets stored and\n // rendered); only `config.messages` loses it.\n onConfig(_ctx, config) {\n const messages = stripObservedToolCalls(config.messages)\n if (messages.length === config.messages.length) return\n return { messages }\n },\n\n // The engine re-syncs `middlewareCtx.messages` from its own array once per agent\n // iteration, which drops whatever the recorder appended during the previous\n // iteration's stream. Restoring it here — AFTER that sync — is what makes a\n // multi-iteration run keep its full history without depending on where this\n // middleware sits relative to persistence in the middleware array.\n onIteration(ctx) {\n runState.get(ctx)?.toolHistory.reconcile(ctx)\n },\n\n // Record the harness's own tool calls as transcript messages. Observe only:\n // returning nothing passes every chunk through untouched.\n onChunk(ctx, chunk) {\n runState.get(ctx)?.toolHistory.observe(chunk, ctx)\n },\n\n async onFinish(ctx) {\n const state = runState.get(ctx)\n if (!state) return\n const { handle, ensureCtx } = state\n\n // Last chance before persistence writes the transcript. Only matters if a\n // config sync landed after the final tool chunk; the recorder is idempotent, so\n // in the normal case this changes nothing.\n state.toolHistory.reconcile(ctx)\n\n await drainWatcher(state, 'finish')\n\n const lifecycle = definition.lifecycle\n\n // `handle` is absent only if `setup` never got past `definition.ensure`, in\n // which case there is no sandbox to snapshot.\n if (\n lifecycle?.snapshot === 'after-run' &&\n handle?.capabilities.snapshots &&\n handle.snapshot\n ) {\n const snapshot = await handle.snapshot(`after-run-${ctx.runId}`)\n const store = ensureCtx.store\n if (store) {\n const key = definition.key(ensureCtx)\n const existing = await store.get(key)\n if (existing) {\n await store.upsert({\n ...existing,\n latestSnapshotId: snapshot.id,\n updatedAt: Date.now(),\n })\n }\n }\n }\n\n if (lifecycle?.destroyOnComplete) {\n await definition.destroy(ensureCtx)\n await definition.hooks?.onDestroy?.()\n }\n },\n\n async onAbort(ctx, info: AbortInfo) {\n const state = runState.get(ctx)\n if (!state) return\n\n // First on BOTH branches: a diff still in flight must be drained whether\n // the sandbox is about to be destroyed or merely detached, or the final\n // file's diff is dropped.\n await drainWatcher(state, 'abort')\n\n const durability = state.durability\n const cancelled = await cancelIntent(\n durability,\n ctx.runId,\n info.cancelRequested === true,\n )\n\n if (\n durability !== undefined &&\n !cancelled &&\n durability.detachOnDisconnect\n ) {\n // DETACH on the teardown path. Reached when the run is aborted for a\n // reason that is NOT an out-of-band cancel while detachable — a genuine\n // stop from elsewhere, or a host going down. The ordinary disconnect is\n // handled by the disconnect subscriber in `setup`, which does not end the\n // run at all.\n //\n // On a failed record write this branch is ABANDONED for the destroy one\n // below, because a rejection here is the worst shape available: the\n // verdict is unpublished, so core terminalizes the log and records a\n // healthy detached run as failed; `detachedSince`/`sandboxKey` are\n // unwritten, so `listReclaimable` can never surface the run and\n // `reapDetachedRuns` can never reclaim it. A DESTROYED sandbox beats an\n // unreachable one — the same reasoning `drainWatcher` applies to its own\n // guarded `stop()`.\n if (await recordDetach(definition, state, durability, ctx, 'abort')) {\n return\n }\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n return\n }\n\n // ALWAYS tear down on an explicit abort, regardless of `destroyOnComplete`.\n // The in-sandbox agent process is not killed by closing its IO stream\n // (e.g. a Docker exec survives client disconnect), so the only reliable way\n // to stop it — and the token/cost drain of its ongoing API calls — is to\n // destroy the sandbox (stop the container/VM). `keepAlive` /\n // `destroyOnComplete:false` governs *successful completion*, never cancel.\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n },\n\n async onError(ctx, info) {\n const state = runState.get(ctx)\n if (!state) return\n\n await drainWatcher(state, 'error')\n await definition.hooks?.onError?.(info.error)\n\n // On failure, only tear down when the lifecycle says so; otherwise leave\n // the sandbox for a resumed retry.\n if (definition.lifecycle?.destroyOnComplete) {\n await definition.destroy(state.ensureCtx)\n await definition.hooks?.onDestroy?.()\n }\n },\n })\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAkHA,IAAM,2BAAW,IAAI,QAAiC;;;;;;;AAQtD,eAAe,aACb,OACA,OACe;CAIf,IAAI;EACF,MAAM,MAAM,SAAS,KAAK;CAC5B,SAAS,OAAO;EACd,MAAM,QAAQ,KAAK,+BAA+B;GAAE;GAAO;EAAM,CAAC;CACpE;CACA,MAAM,QAAQ,WAAW,MAAM,YAAY;CAC3C,IAAI,MAAM,SAAS,MAAM,QAAQ,QAAQ,2BAA2B,EAAE,MAAM,CAAC;AAC/E;;;;;;;;;;;;;;;;;;;;;AAsBA,eAAe,aACb,YACA,OACA,YACA,KACA,OACkB;CAClB,IAAI;EAQF,MAAM,WAAW,KAAK,OAAO,IAAI,OAAO;GACtC,eAAe,KAAK,IAAI;GACxB,YAAY,WAAW,IAAI,MAAM,SAAS;EAC5C,CAAC;CACH,SAAS,OAAO;EACd,MAAM,QAAQ,KAAK,sCAAsC;GACvD,OAAO,IAAI;GACX;GACA;EACF,CAAC;EACD,OAAO;CACT;CAMA,mBAAmB,KAAK,IAAI;CAC5B,OAAO;AACT;;;;;;;;AASA,eAAe,aACb,YACA,OACA,WACkB;CAClB,IAAI,WAAW,OAAO;CACtB,IAAI,eAAe,KAAA,GAAW,OAAO;CAMrC,OAAO,mBAAmB,WAAW,MAAM,KAAK;AAClD;;AAGA,SAAS,WACP,SACiD;CACjD,IAAI,YAAY,QAAQ,OAAO,YAAY,UAAU,OAAO,KAAA;CAC5D,MAAM,IAAI;CACV,MAAM,SAAS,OAAO,EAAE,WAAW,WAAW,EAAE,SAAS,KAAA;CACzD,MAAM,QAAQ,OAAO,EAAE,UAAU,WAAW,EAAE,QAAQ,KAAA;CACtD,IAAI,WAAW,KAAA,KAAa,UAAU,KAAA,GAAW,OAAO,KAAA;CACxD,OAAO;EAAE;EAAQ;CAAM;AACzB;;;;;;AAqDA,SAAS,eACP,KAMA,SACsB;CACtB,OAAO;EACL,UAAU,IAAI;EACd,OAAO,IAAI;EACX,OACE,SAAS,aAAa,IAAI,YAAY,8BAA8B;EACtE,OAAO,SAAS,SAAS,IAAI,YAAY,eAAe;EACxD,QAAQ,WAAW,IAAI,OAAO;EAC9B,QAAQ,IAAI;EACZ,aAAa,IAAI;CACnB;AACF;;;;;;;;AASA,eAAe,wBACb,OACA,OACA,QACe;CACf,IAAI,CAAC,OAAO;CACZ,MAAM,QACJ;EACE,QAAQ;EACR,QAAQ;EACR,QAAQ;CACV,EACA,MAAM;CACR,KAAK,MAAM,MAAM,CAAC,MAAM,QAAQ,MAAM,MAAM,GAAG;EAC7C,IAAI,CAAC,IAAI;EACT,IAAI;GACF,MAAM,GAAG,KAAK;EAChB,SAAS,OAAO;GAGd,QAAQ,OAAO,4BAA4B;IACzC,MAAM,MAAM;IACZ,MAAM,MAAM;IACZ;GACF,CAAC;EACH;CACF;AACF;AAEA,SAAgB,YACd,YACA,SAKA;CACA,OAAO,qBAAqB;EAC1B,MAAM;EACN,UAAU,CAAC,mBAAmB,oBAAoB;EAOlD,kBAAkB,CAAC,gCAAgC,eAAe;EAElE,MAAM,MAAM,KAAK;GACf,MAAM,YAAY,eAAe,KAAK,OAAO;GAS7C,MAAM,aAAa,yBAAkC,OAAO;GAC5D,IAAI,eAAe,KAAA,GAAW;IAC5B,yBAAyB,KAAK,UAAU;IAGxC,qBAAqB,KAAK,IAAI;GAChC;GAKA,MAAM,UAAU,kBAAkB,KAAK,EAAE,UAAU,KAAK,CAAC;GACzD,MAAM,SAAS,SAAS;GAsBxB,MAAM,QAAyB;IAC7B;IACA,cAAc,CAAC;IACf,aAAa,0BAA0B;IACvC,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC;IAC3B,GAAI,aAAa,EAAE,WAAW,IAAI,CAAC;GACrC;GACA,SAAS,IAAI,KAAK,KAAK;GAkBvB,IAAI,eAAe,KAAA,GAAW;IAC5B,IAAI;KACF,MAAM,WAAW,KAAK,eAAe;MACnC,OAAO,IAAI;MACX,UAAU,IAAI;MACd,WAAW,KAAK,IAAI;KACtB,CAAC;IACH,SAAS,OAAO;KAGd,QAAQ,KAAK,wCAAwC;MACnD,OAAO,IAAI;MACX;KACF,CAAC;IACH;IAyBA,IAAI;KACF,MAAM,eAAe,KAAK,EAAE,UAAU,KAAK,CAAC,CAAC,EAAE,SAAS;IAC1D,SAAS,OAAO;KAGd,QAAQ,KAAK,wCAAwC;MACnD,OAAO,IAAI;MACX;KACF,CAAC;IACH;GACF;GAcA,IAAI,eAAe,KAAA,KAAa,WAAW,oBACzC,iBAAiB,KAAK,EAAE,UAAU,KAAK,CAAC,CAAC,EAAE,UAAU,YAAY;IAU/D,IAAI,MAAM,aAAa,YAAY,IAAI,OAAO,KAAK,GAAG;IACtD,IACE,MAAM,aAAa,YAAY,OAAO,YAAY,KAAK,YAAY,GAEnE,MAAM,QAAQ,QACZ,yDACA,EAAE,OAAO,IAAI,MAAM,CACrB;GAEJ,CAAC;GAGH,MAAM,SAAS,MAAM,WAAW,OAAO,SAAS;GAGhD,MAAM,SAAS;GACf,eAAe,KAAK,MAAM;GAC1B,IAAI,WAAW,QAAQ,qBAAqB,KAAK,WAAW,MAAM;GAUlE,IACE,eAAe,KAAA,MACd,UAAU,UAAU,KAAA,KACnB,UAAU,iBAAiB,oBAE7B,QAAQ,KACN,mRAIA,EAAE,OAAO,IAAI,MAAM,CACrB;GAGF,MAAM,YAAY,WAAW,WAAW,QAAA;GACxC,IAAI,UAAU;GACd,IAAI;IACF,MAAM,SAAS,MAAM,OAAO,QAAQ,KAAK,sBAAsB,EAC7D,KAAK,UACP,CAAC;IACD,IAAI,OAAO,aAAa,GAAG;KACzB,UAAU,OAAO,OAAO,KAAK;KAC7B,QAAQ,QAAQ,iCAAiC;MAC/C,MAAM;MACN;KACF,CAAC;IACH,OAKE,QAAQ,QAAQ,oDAAoD;KAClE,MAAM;KACN,UAAU,OAAO;KACjB,QAAQ,OAAO;IACjB,CAAC;GAEL,SAAS,OAAO;IAId,QAAQ,KAAK,uCAAuC;KAClD,MAAM;KACN;IACF,CAAC;GACH;GAEA,MAAM,YAAY,WAAW;GAC7B,IAAI,cAAc,KAAA,GAAW;IAC3B,MAAM,cAAc,UAAU,QAAA;IAC9B,MAAM,OAAO,kBAAkB,QAAQ,WAAW;IAClD,MAAM,gBAAgB,qBAAqB,SAAS;IACpD,MAAM,UAAU,UAAU;IAC1B,2BAA2B,KAAK;KAC9B,QAAQ,UAAU,UAAU,CAAC;KAC7B,SAAS,UAAU,WAAW,CAAC;KAC/B,gBAAgB,QAAQ;MACtB,IAAI,YAAY,KAAA,GACd,MAAM,IAAI,MACR,8DAA8D,IAAI,aAAa,GACjF;MAEF,OAAO,cAAc,SAAS,GAAG;KACnC;KACA,YAAY,GAAG,KAAK,uBAAuB;KAC3C;KACA,GAAI,UAAU,YAAY,KAAA,IACtB,EAAE,SAAS,UAAU,QAAQ,IAC7B,CAAC;IACP,CAAC;GACH;GAEA,MAAM,QAAQ,WAAW;GACzB,MAAM,OAAO,UAAU,MAAM;GAE7B,MAAM,KAAK,kBAAkB,WAAW,UAAU;GAKlD,MAAM,eAAe,MAAM;GAC3B,IAAI;GACJ,IAAI,GAAG,SAAS;IACd,UAAU,MAAM,eAAe,QAAQ;KACrC,UAAU,UAA4B;MACpC,MAAM,WAAW,mBACf,QACA,WACA,SACA,OACA,MACF;MACA,wBAA6B,OAAO,UAAU,MAAM;MACpD,SAAS,KAAK,QAAQ;MACtB,IAAI,GAAG,MACL,aAAa,KACX,SACG,KAAK,CAAC,CACN,MAAM,SAAS;OACd,SAAS,aAAa;QAAE,MAAM,MAAM;QAAM;OAAK,CAAC;MAClD,CAAC,CAAC,CACD,OAAO,UAAmB;OACzB,QAAQ,KAAK,iCAAiC;QAC5C,MAAM,MAAM;QACZ;OACF,CAAC;MACH,CAAC,CACL;KAEJ;KAMA,MAAM;KACN,GAAI,IAAI,WAAW,KAAA,IAAY,EAAE,QAAQ,IAAI,OAAO,IAAI,CAAC;KACzD,GAAI,WAAW,KAAA,IAAY,EAAE,OAAO,IAAI,CAAC;IAC3C,CAAC;IACD,QAAQ,QAAQ,2BAA2B;KACzC,MAAM;KACN,MAAM,GAAG;IACX,CAAC;GACH;GAOA,IAAI,SAAS,MAAM,UAAU;EAC/B;EAOA,SAAS,MAAM,QAAQ;GACrB,MAAM,WAAW,uBAAuB,OAAO,QAAQ;GACvD,IAAI,SAAS,WAAW,OAAO,SAAS,QAAQ;GAChD,OAAO,EAAE,SAAS;EACpB;EAOA,YAAY,KAAK;GACf,SAAS,IAAI,GAAG,CAAC,EAAE,YAAY,UAAU,GAAG;EAC9C;EAIA,QAAQ,KAAK,OAAO;GAClB,SAAS,IAAI,GAAG,CAAC,EAAE,YAAY,QAAQ,OAAO,GAAG;EACnD;EAEA,MAAM,SAAS,KAAK;GAClB,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GACZ,MAAM,EAAE,QAAQ,cAAc;GAK9B,MAAM,YAAY,UAAU,GAAG;GAE/B,MAAM,aAAa,OAAO,QAAQ;GAElC,MAAM,YAAY,WAAW;GAI7B,IACE,WAAW,aAAa,eACxB,QAAQ,aAAa,aACrB,OAAO,UACP;IACA,MAAM,WAAW,MAAM,OAAO,SAAS,aAAa,IAAI,OAAO;IAC/D,MAAM,QAAQ,UAAU;IACxB,IAAI,OAAO;KACT,MAAM,MAAM,WAAW,IAAI,SAAS;KACpC,MAAM,WAAW,MAAM,MAAM,IAAI,GAAG;KACpC,IAAI,UACF,MAAM,MAAM,OAAO;MACjB,GAAG;MACH,kBAAkB,SAAS;MAC3B,WAAW,KAAK,IAAI;KACtB,CAAC;IAEL;GACF;GAEA,IAAI,WAAW,mBAAmB;IAChC,MAAM,WAAW,QAAQ,SAAS;IAClC,MAAM,WAAW,OAAO,YAAY;GACtC;EACF;EAEA,MAAM,QAAQ,KAAK,MAAiB;GAClC,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GAKZ,MAAM,aAAa,OAAO,OAAO;GAEjC,MAAM,aAAa,MAAM;GACzB,MAAM,YAAY,MAAM,aACtB,YACA,IAAI,OACJ,KAAK,oBAAoB,IAC3B;GAEA,IACE,eAAe,KAAA,KACf,CAAC,aACD,WAAW,oBACX;IAeA,IAAI,MAAM,aAAa,YAAY,OAAO,YAAY,KAAK,OAAO,GAChE;IAEF,MAAM,WAAW,QAAQ,MAAM,SAAS;IACxC,MAAM,WAAW,OAAO,YAAY;IACpC;GACF;GAQA,MAAM,WAAW,QAAQ,MAAM,SAAS;GACxC,MAAM,WAAW,OAAO,YAAY;EACtC;EAEA,MAAM,QAAQ,KAAK,MAAM;GACvB,MAAM,QAAQ,SAAS,IAAI,GAAG;GAC9B,IAAI,CAAC,OAAO;GAEZ,MAAM,aAAa,OAAO,OAAO;GACjC,MAAM,WAAW,OAAO,UAAU,KAAK,KAAK;GAI5C,IAAI,WAAW,WAAW,mBAAmB;IAC3C,MAAM,WAAW,QAAQ,MAAM,SAAS;IACxC,MAAM,WAAW,OAAO,YAAY;GACtC;EACF;CACF,CAAC;AACH"}
package/dist/esm/reap.js CHANGED
@@ -118,7 +118,8 @@ async function decodeFrame(stdout) {
118
118
  async function probeRunExit(input) {
119
119
  try {
120
120
  const paths = journalPaths(input.runId, input.dir);
121
- const exitCode = parseJournalExit(await decodeFrame((await input.handle.process.exec(journalExitProbeCommand(paths, input.maxBytes ?? 4096))).stdout), paths);
121
+ const result = await input.handle.process.exec(journalExitProbeCommand(paths, input.maxBytes ?? 4096));
122
+ const exitCode = parseJournalExit(await decodeFrame(result.stdout), paths);
122
123
  return exitCode === null ? { state: "producing" } : {
123
124
  state: "finished",
124
125
  exitCode