cookbook-bridge 0.1.13 → 0.1.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bridge.mjs CHANGED
@@ -29,12 +29,21 @@
29
29
  * Node built-ins only. No dependencies.
30
30
  */
31
31
  import fs from "node:fs";
32
+ /** This Bridge's own version, filed with every trace; "unknown" on a hand-copied install. */
33
+ const BRIDGE_VERSION = (() => {
34
+ // npm installs ship package.json next to this file; a service install (the
35
+ // manifest updater) ships version.json instead; a hand-copied dir has neither.
36
+ for (const name of ["./package.json", "./version.json"]) {
37
+ try { const v = JSON.parse(fs.readFileSync(new URL(name, import.meta.url), "utf8")).version; if (v) return String(v); } catch { /* next */ }
38
+ }
39
+ return "unknown";
40
+ })();
32
41
  import os from "node:os";
33
42
  import path from "node:path";
34
43
  import { fileURLToPath } from "node:url";
35
44
  import { spawn } from "node:child_process";
36
45
  import { callsFromStreamLine, foldCallEvent, wireCalls, shortTool, argFor } from "./live.mjs";
37
- import { planFromStreamLine, notePlan, planLine } from "./plan.mjs";
46
+ import { planFromStreamLine, notePlan, planLine, noteAuth, isSignedOutError } from "./plan.mjs";
38
47
 
39
48
  // Node version guard: below 18 there is no global fetch and none of this runs. One
40
49
  // plain line beats a stack trace from the first `fetch(` call.
@@ -51,18 +60,21 @@ import { planFromStreamLine, notePlan, planLine } from "./plan.mjs";
51
60
  // repair itself — the update path depends ONLY on update.mjs (node built-ins only).
52
61
  // The e2e that forced this: a stale install missing volunteer.mjs couldn't even reach
53
62
  // the updater when these were static imports.
54
- import { deriveWakeTopic, wakeSocketSupported, connectWakeSocket } from "./realtime.mjs";
63
+ import { deriveWakeTopic, wakeSocketSupported, connectWakeSocket, createLivePublisher } from "./realtime.mjs";
55
64
  import { createSessionReporter } from "./sessions.mjs";
56
65
  import { resolveAgentForTask } from "./chef.mjs";
57
66
  // The config home + command phrasing live in update.mjs (node built-ins only), so the
58
67
  // broken-install `update` path and every other command agree on both.
59
68
  import { locateConfig, cli, updateLine, configHome } from "./update.mjs";
60
- let listWorkspaces, listTasks, listOpenWork, getTask, threadResumeContext, completeTaskApi, resolveDelegation, reportTaskUsage, reportTaskProgress, volunteerClaim, dispatchClaim, abandonTask, recallMemories, recallAcrossWorkspaces, creditRecall, getVolunteerSettings, agentsQuery;
61
- let agentEnv, checkGeminiVersion, isGeminiCommand, GEMINI_MIN_VERSION, checkAgyVersion, isAgyCommand, AGY_MIN_VERSION, withCookbookMcp, isClaudeCommand, withApprovalRelay, materializeMcpConfig;
69
+ let buildChatPrompt, buildChatFollowUpPrompt;
70
+ let classifyFailure, failurePayload, stagesPayload, failureLine;
71
+ let createTrace, traceEnvelope; // the Record (bridge/trace.mjs); null on an install without it
72
+ let listWorkspaces, listTasks, listOpenWork, getTask, threadResumeContext, completeTaskApi, resolveDelegation, reportTaskUsage, reportTaskProgress, volunteerClaim, dispatchClaim, abandonTask, recallMemories, recallAcrossWorkspaces, creditRecall, getVolunteerSettings, agentsQuery, releaseTask, postTrace;
73
+ let agentEnv, checkGeminiVersion, isGeminiCommand, GEMINI_MIN_VERSION, checkAgyVersion, isAgyCommand, AGY_MIN_VERSION, withCookbookMcp, isClaudeCommand, withApprovalRelay, materializeMcpConfig, cookbookServerFromClaudeConfig, withCookbookMcpServer, sweepStaleMcpDirs;
62
74
  let isKimiCommand, kimiFromLine, kimiResultEnvelope, kimiCommand, kimiLoginState, kimiMcpState, checkKimiVersion;
63
75
  let extractUsage, displayText;
64
76
  let volunteeringEnabled, volunteerCandidates, decisionPrompt, parseDecision, MAX_DECISIONS_PER_POLL, mergeVolunteerSettings, effectiveCapabilities;
65
- let buildPrompt, buildThreadFollowUpPrompt;
77
+ let buildPrompt, buildThreadFollowUpPrompt, PROMPT_VERSION;
66
78
  let runnerFor, hasRunner, warmUp, adoptRunner, reapIdleRunners, killAllRunners;
67
79
  let hasCodexThread, reapCodexServer, killCodexServer;
68
80
  let checkForUpdate, applyUpdate;
@@ -74,13 +86,18 @@ let fetchHands, claimHandsCall, reportHandsResult;
74
86
  async function loadRuntime() {
75
87
  ({ createLocalServer, toolsForMode, modeForTools, vendorOf } = await import("./local.mjs"));
76
88
  ({ connectAgentsProgrammatic, detectClis } = await import("./device.mjs"));
77
- ({ listWorkspaces, listTasks, listOpenWork, getTask, threadResumeContext, completeTaskApi, resolveDelegation, reportTaskUsage, reportTaskProgress, volunteerClaim, dispatchClaim, abandonTask, recallMemories, recallAcrossWorkspaces, creditRecall, getVolunteerSettings, fetchHands, claimHandsCall, reportHandsResult, agentsQuery } = await import("./cookbook.mjs"));
89
+ ({ listWorkspaces, listTasks, listOpenWork, getTask, threadResumeContext, completeTaskApi, resolveDelegation, reportTaskUsage, reportTaskProgress, volunteerClaim, dispatchClaim, abandonTask, recallMemories, recallAcrossWorkspaces, creditRecall, getVolunteerSettings, fetchHands, claimHandsCall, reportHandsResult, agentsQuery, releaseTask, postTrace } = await import("./cookbook.mjs"));
78
90
  ({ serveCalls, describeCall, hostingMode, which: whichExec, argvForSpawn, redact: redactText, resolveCmdShim, killTree, runPreflight, grantsNeedingPreflight } = await import("./hands.mjs"));
79
91
  ({ agentEnv, checkGeminiVersion, isGeminiCommand, GEMINI_MIN_VERSION, checkAgyVersion, isAgyCommand, AGY_MIN_VERSION, withCookbookMcp, isClaudeCommand, withApprovalRelay, materializeMcpConfig,
80
- isKimiCommand, kimiFromLine, kimiResultEnvelope, kimiCommand, kimiLoginState, kimiMcpState, checkKimiVersion } = await import("./harden.mjs"));
92
+ isKimiCommand, kimiFromLine, kimiResultEnvelope, kimiCommand, kimiLoginState, kimiMcpState, checkKimiVersion, cookbookServerFromClaudeConfig, withCookbookMcpServer, sweepStaleMcpDirs } = await import("./harden.mjs"));
93
+ // A Bridge that died hard (kill -9, a force-quit shell) leaves its 0600 mcp.json
94
+ // temp files behind; every start sweeps the stale ones from earlier processes.
95
+ try { sweepStaleMcpDirs?.(); } catch { /* best effort */ }
81
96
  ({ extractUsage, displayText } = await import("./usage.mjs"));
82
97
  ({ volunteeringEnabled, volunteerCandidates, decisionPrompt, parseDecision, MAX_DECISIONS_PER_POLL, mergeVolunteerSettings, effectiveCapabilities } = await import("./volunteer.mjs"));
83
- ({ buildPrompt, buildThreadFollowUpPrompt } = await import("./prompt.mjs"));
98
+ ({ buildPrompt, buildThreadFollowUpPrompt, buildChatPrompt, buildChatFollowUpPrompt, PROMPT_VERSION } = await import("./prompt.mjs"));
99
+ ({ classifyFailure, failurePayload, stagesPayload, failureLine } = await import("./failures.mjs"));
100
+ try { ({ createTrace, traceEnvelope } = await import("./trace.mjs")); } catch { /* a hand-copied install without trace.mjs: no traces, everything else runs */ }
84
101
  ({ runnerFor, hasRunner, warmUp, adoptRunner, reapIdleRunners, killAllRunners } = await import("./thread-runner.mjs"));
85
102
  ({ hasCodexThread, reapCodexServer, killCodexServer } = await import("./codex-runner.mjs"));
86
103
  ({ checkForUpdate, applyUpdate } = await import("./update.mjs"));
@@ -183,9 +200,24 @@ function loadConfig() {
183
200
  process.exit(1);
184
201
  }
185
202
  cfg.pollSeconds = cfg.pollSeconds ?? 15;
186
- // Persistent per-thread agent processes (terminal-feel replies). Opt-in while it
187
- // proves itself; the one-shot spawn path remains the fallback either way.
188
- cfg.persistentThreads = cfg.persistentThreads ?? false;
203
+ // Persistent per-thread agent processes (terminal-feel replies). ON by default
204
+ // since 2026-09-14: a cold `claude -p` per message was the single largest reason a
205
+ // chat turn took 38 s in production while the same login answered in 4 s inside
206
+ // Buzz. The one-shot spawn path remains the fallback either way.
207
+ const explicitPersistent = cfg.persistentThreads === true;
208
+ cfg.persistentThreads = cfg.persistentThreads ?? true;
209
+ // How long an idle runner stays alive (minutes) and how many may live at once.
210
+ const positive = (v, d) => (Number.isFinite(Number(v)) && Number(v) > 0 ? Number(v) : d);
211
+ cfg.runnerIdleMinutes = positive(cfg.runnerIdleMinutes, 180);
212
+ cfg.maxRunners = positive(cfg.maxRunners, 8);
213
+ // "The final message IS the result" (bridgeFiles) is the CHAT lane's contract.
214
+ // Board tasks keep calling complete_task unless a config explicitly asked for the
215
+ // old persistentThreads:true behavior (Chef's config did), or sets bridgeFilesAll.
216
+ cfg.bridgeFilesAll = cfg.bridgeFilesAll ?? explicitPersistent;
217
+ // Pin claude runs to the CLI's own Cookbook connection when the agent has no
218
+ // token of its own (harden.mjs cookbookServerFromClaudeConfig). `pinMcp: false`
219
+ // keeps the old behavior (every user-scope MCP server loads into every run).
220
+ cfg.pinMcp = cfg.pinMcp ?? true;
189
221
  // LOCAL WORKSPACE ACCESS (2026-08-21): map a workspace to a folder on THIS machine.
190
222
  // localWorkspaces: { "<workspaceId>": { "cwd": "/abs/path", "allowedTools": "…" } }
191
223
  // SELF-ASSIGNED tasks in a mapped workspace run IN that folder with real tools
@@ -269,10 +301,34 @@ export function allowedByPolicy(cfg, agent, task) {
269
301
  * own token (Chef) ran as whoever the machine's Claude was logged in as — seen
270
302
  * 2026-08-28: Chef saw diego's workspaces and "No such grant". Same rewrite, once.
271
303
  */
304
+ let pinnedServerCache = { at: 0, url: null, server: null };
305
+ let pinnedNoteShown = false;
306
+ /** The CLI's own Cookbook server entry (cached 60 s per Cookbook origin) for agents
307
+ * without a token. Only header-authenticated entries qualify (an OAuth entry's
308
+ * credentials live under its server name and would not survive the rename). */
309
+ function cookbookPinServer(cfg) {
310
+ if (!cfg || cfg.pinMcp === false || !cfg.cookbookUrl || !cookbookServerFromClaudeConfig) return null;
311
+ if (pinnedServerCache.url !== cfg.cookbookUrl || Date.now() - pinnedServerCache.at > 60_000) {
312
+ let server = null;
313
+ try { server = cookbookServerFromClaudeConfig({ cookbookUrl: cfg.cookbookUrl }); } catch { server = null; }
314
+ pinnedServerCache = { at: Date.now(), url: cfg.cookbookUrl, server };
315
+ if (server && !pinnedNoteShown) { pinnedNoteShown = true; log(` ↳ claude runs pinned to this machine's Cookbook connection (one MCP server, not every server on the machine); set "pinMcp": false to opt out`); }
316
+ }
317
+ return pinnedServerCache.server;
318
+ }
319
+ /** One place for "make this claude command carry only Cookbook": the agent's own
320
+ * token wins; otherwise the CLI's own Cookbook entry (same identity the member
321
+ * connected with). Anything else is returned untouched. */
322
+ function pinCommand(cfg, agent, command) {
323
+ if (!Array.isArray(command)) return command;
324
+ if (agent.token) return withCookbookMcp(command, { token: agent.token, cookbookUrl: agent.cookbookUrl ?? cfg?.cookbookUrl }).command;
325
+ const server = cookbookPinServer(cfg);
326
+ if (server && withCookbookMcpServer) return withCookbookMcpServer(command, server).command;
327
+ return command;
328
+ }
272
329
  function pinnedAgent(cfg, agent) {
273
330
  if (!agent || !Array.isArray(agent.command)) return agent;
274
- let command = agent.command;
275
- if (agent.token) command = withCookbookMcp(command, { token: agent.token, cookbookUrl: agent.cookbookUrl ?? cfg.cookbookUrl }).command;
331
+ let command = pinCommand(cfg, agent, agent.command);
276
332
  // Relay wiring never depends on a per-agent token (the old guard skipped BOTH).
277
333
  if (agent.approvalRelay && withApprovalRelay) command = withApprovalRelay(command, agent.approvalRelay).command;
278
334
  return command === agent.command ? agent : { ...agent, command };
@@ -297,11 +353,11 @@ export function streamingCommand(command) {
297
353
  const rewritten = [...command];
298
354
  rewritten[i + 1] = "stream-json";
299
355
  if (!rewritten.includes("--verbose")) rewritten.push("--verbose");
300
- // Live words stream PER COMPLETED TURN (assistant events), deliberately NOT
301
- // --include-partial-messages: that flag stores partial-generation artifacts in the
302
- // session file, and RESUMING such a session trips the API's reasoning-extraction
303
- // safeguard (observed live 2026-08-20: resumed thread runs refused with
304
- // `[reasoning_extraction]`). Resume is the flagship; per-turn streaming is plenty.
356
+ // One-shot runs stream PER COMPLETED TURN (assistant events). The persistent
357
+ // runner (thread-runner.mjs) streams token deltas with --include-partial-messages;
358
+ // a session recorded that way was verified to resume cleanly on claude 2.1.272
359
+ // (2026-09-14), so the 2026-08-20 [reasoning_extraction] refusal no longer
360
+ // gates this flag. The one-shot path simply has no reason to pay for deltas.
305
361
  return { command: rewritten, streaming: true };
306
362
  }
307
363
 
@@ -387,7 +443,7 @@ export function sessionIdFrom(line) {
387
443
  * only; buffered/non-streaming runs emit nothing until the end) and keep a
388
444
  * generous absolute ceiling purely as a cost backstop. Pure for tests. */
389
445
  export function shouldKill({ streaming, startedAt, lastActivityAt, now, livenessMs, ceilingMs }) {
390
- if (now - startedAt >= ceilingMs) return { kill: true, why: `hit the ${Math.round(ceilingMs / 60000)}min absolute ceiling` };
446
+ if (now - startedAt >= ceilingMs) return { kill: true, why: `hit the ${ceilingMs >= 120_000 ? `${Math.round(ceilingMs / 60000)}min` : `${Math.round(ceilingMs / 1000)}s`} absolute ceiling` };
391
447
  if (streaming && livenessMs > 0 && now - lastActivityAt >= livenessMs) {
392
448
  return { kill: true, why: `no output for ${Math.round(livenessMs / 1000)}s (stalled — likely a hung prompt or dead CLI)` };
393
449
  }
@@ -444,7 +500,7 @@ function spawnAgent(agent, prompt, timeoutSeconds, env, onProgress, opts = {}) {
444
500
  // agent's name — never as whatever the CLI is logged in as, and blind to stale
445
501
  // claude.ai connectors that poison headless runs (2026-08-25).
446
502
  let baseCommand = withCookbookMcp
447
- ? withCookbookMcp(opts.command ?? agent.command, { token: agent.token, cookbookUrl: agent.cookbookUrl }).command
503
+ ? pinCommand(opts.cfg ?? null, agent, opts.command ?? agent.command)
448
504
  : (opts.command ?? agent.command);
449
505
  if (agent.approvalRelay && withApprovalRelay) baseCommand = withApprovalRelay(baseCommand, agent.approvalRelay).command;
450
506
  const { command, streaming } = onProgress ? streamingCommand(baseCommand) : { command: baseCommand, streaming: false };
@@ -483,7 +539,10 @@ function spawnAgent(agent, prompt, timeoutSeconds, env, onProgress, opts = {}) {
483
539
  reject(new Error(`refusing to pass a prompt through the ${path.basename(bare)} shell shim on Windows (cmd.exe quoting is not safe for workspace text). Point this agent's command at the CLI's .js entry or its real executable instead.`));
484
540
  return;
485
541
  }
486
- wrapped = [process.execPath, script, ...args];
542
+ // Under the compiled launcher process.execPath is the Bridge binary, not node;
543
+ // the shim's own script wants the node that npm installed it with.
544
+ const nodeExe = process.versions?.bun ? (whichExec?.("node") ?? process.execPath) : process.execPath;
545
+ wrapped = [nodeExe, script, ...args];
487
546
  } else {
488
547
  wrapped = argvForSpawn ? argvForSpawn([bare, ...args]) : [bare, ...args];
489
548
  }
@@ -567,11 +626,13 @@ function spawnAgent(agent, prompt, timeoutSeconds, env, onProgress, opts = {}) {
567
626
  // text throttle (still ≥300ms apart so a burst of reads is one tick).
568
627
  let touched = false;
569
628
  for (const ev of callsFromStreamLine(line)) { calls = foldCallEvent(calls, ev); touched = true; }
629
+ if (opts.trace) { try { opts.trace.onStreamLine(line); } catch { /* evidence is best-effort */ } }
570
630
  if (kimi) {
571
631
  // Kimi's lines are keyed by `role` (claude's by `type`), so the claude
572
632
  // parsers above ignore them and this is the only reader.
573
633
  const kev = kimiFromLine(line);
574
634
  if (kev) {
635
+ if (opts.trace) { try { opts.trace.onKimiEvent(kev); } catch { /* evidence is best-effort */ } }
575
636
  if (kev.sessionId && !sessionId) sessionId = kev.sessionId;
576
637
  if (kev.text) {
577
638
  turnsText += (turnsText ? "\n\n" : "") + kev.text;
@@ -663,7 +724,7 @@ async function runAgent(cfg, agent, prompt, onProgress, retry = null, taskCtx =
663
724
  warnedCodexToken = true;
664
725
  log(`! Codex has no agent token, so it runs under your Bridge token. Run \`${cli("connect")}\` so its work reads "Codex · via you".`);
665
726
  }
666
- return runCodexTask(agent, prompt, cfg.taskTimeoutSeconds, codexToken, env, onProgress, { threadKey, log, model });
727
+ return runCodexTask(agent, prompt, cfg.taskTimeoutSeconds, codexToken, env, onProgress, { threadKey, log, model, trace: taskCtx.trace ?? null });
667
728
  }
668
729
  if (agent.runner === "openclaw") {
669
730
  // The visiting-agent lane: one Gateway-backed turn, resumed by session id so a
@@ -688,8 +749,10 @@ async function runAgent(cfg, agent, prompt, onProgress, retry = null, taskCtx =
688
749
  // the failure back. Other CLIs get the failure fed forward in a fresh prompt.
689
750
  const { command, resumed } = resumeCommand(agent.command, retry?.sessionId ?? null);
690
751
  return spawnAgent(agent, prompt, cfg.taskTimeoutSeconds, env, live ? onProgress : undefined, {
752
+ cfg,
691
753
  command,
692
754
  resumed,
755
+ trace: taskCtx.trace ?? null,
693
756
  livenessSeconds: cfg.livenessTimeoutSeconds,
694
757
  onChild: taskCtx.onChild,
695
758
  });
@@ -711,9 +774,22 @@ function capHoldLine(who, res) {
711
774
  return `${who} hit their ${capStr}-token daily cap on your subscription${spentStr} — held until it resets (or you raise it in Account → Agents).`;
712
775
  }
713
776
 
714
- function failureHint(result) {
777
+ function failureHint(result, agent = null) {
715
778
  if (!result || typeof result !== "object") return "";
716
779
  const { code, err, out } = result;
780
+ // Codex app-server turns carry their reason in `err` (codex-runner.mjs): judged
781
+ // first, because the transcript is empty and the generic rules below would
782
+ // read "ran without completing". Both cases are terminal for this attempt.
783
+ if (agent && agent.runner === "app-server") {
784
+ const e = String(err ?? "").toLowerCase();
785
+ if (/could not be refreshed|log out and sign in|unauthorized|token_expired|invalid refresh token/.test(e))
786
+ return "Codex on this machine is signed out (its login could not be refreshed) → run `CODEX_HOME=~/.codex-bridge codex login` in a terminal; this Bridge picks it up on its own";
787
+ if (/requires a newer version of codex|upgrade to the latest/.test(e))
788
+ return "the Codex CLI this Bridge runs is too old for your account's model → update it (`npm i -g @openai/codex`) or point the Codex agent's command at a current `codex`";
789
+ if (/failed to connect|network|dns|connection error/.test(e))
790
+ return `Codex can't reach ChatGPT (${String(err).split("\n")[0].slice(0, 160)}) → check the connection`;
791
+ if (e) return `Codex said: ${String(err).split("\n")[0].slice(0, 200)}`;
792
+ }
717
793
  // JSON-mode envelopes ALWAYS contain the literal substring "permission_denials",
718
794
  // so judging on raw output misdiagnosed every failed claude+json run as "a tool
719
795
  // was blocked" (live, 2026-07-03). Parse the envelope and judge the REAL fields:
@@ -739,8 +815,8 @@ function failureHint(result) {
739
815
  const infra = String(err || "").toLowerCase();
740
816
  const tailSrc = `${err || display || (brace < 0 ? rawOut : "")}`.trim();
741
817
  const tail = tailSrc.split("\n").slice(-2).join(" ").slice(0, 240);
742
- if (infra.includes("not logged in") || infra.includes("please log in"))
743
- return "the agent CLI isn't logged in → run `claude auth login`";
818
+ if (isSignedOutError(infra))
819
+ return "Claude on this machine is signed out → open a terminal, run `claude`, and sign in; this Bridge picks it up on its own";
744
820
  // Kimi's own error strings (stderr: "error: failed to run prompt: provider.connection_error: …").
745
821
  if (infra.includes("provider.connection_error") || infra.includes("connection error"))
746
822
  return "the agent CLI can't reach its API (a network or DNS block on the vendor's hosts) → check the connection, then try `kimi -p hi` by hand";
@@ -845,6 +921,7 @@ let warnedCodexToken = false;
845
921
  const BOOT_MS = Date.now();
846
922
  const inFlight = new Set();
847
923
  const inFlightWs = new Map(); // task id → workspace id, for release on shutdown
924
+ const inFlightThreadKeys = new Map(); // taskId -> thread key (one run per conversation at a time)
848
925
 
849
926
  /**
850
927
  * GRACEFUL STOP (2026-08-29): hand every in-flight task back to the board before
@@ -862,7 +939,7 @@ async function releaseInFlight(cfg, why) {
862
939
  Promise.allSettled(ids.map(async (id) => {
863
940
  try {
864
941
  const { callTool } = await import("./cookbook.mjs");
865
- await callTool(cfg, "release_task", { workspace_id: inFlightWs.get(id), task_id: id });
942
+ await callTool(cfg, "release_task", { workspace_id: inFlightWs.get(id), task_id: id, reason: "Bridge stopped; resumes when a Bridge is up" });
866
943
  log(` ↳ released ${id.slice(0, 8)}`);
867
944
  } catch (e) { log(` ↳ couldn't release ${id.slice(0, 8)}: ${e.message}`); }
868
945
  })),
@@ -930,6 +1007,9 @@ let lastContactAt = 0;
930
1007
  * stay alive on a revoked token (so the user can fix it from the app's Connect UI). A
931
1008
  * headless/terminal Bridge still exits loudly with the fix — never a silent zombie. */
932
1009
  const IS_DESKTOP = process.env.COOKBOOK_DESKTOP === "1";
1010
+ /** CLOUD ONE-SHOT (0102): run what is claimable now, then exit. cloud-entry.mjs sets these. */
1011
+ const ONCE = process.env.COOKBOOK_ONCE === "1";
1012
+ const ONLY_TASK = (process.env.COOKBOOK_ONLY_TASK || "").trim() || null;
933
1013
 
934
1014
  /** Run lifecycle → Bridge Local subscribers (the desktop app's notifications). */
935
1015
  function emitRun(state, ws, task, agent, localMeta) {
@@ -1090,7 +1170,7 @@ async function considerVolunteering(cfg, ws, task, budget) {
1090
1170
  budget.used++;
1091
1171
  let answered = true;
1092
1172
  try {
1093
- const r = await spawnAgent(agent, decisionPrompt(task, effectiveCapabilities(agent, merged) ?? agent.capabilities), cfg.decisionTimeoutSeconds ?? 90, agentEnv(cfg).env);
1173
+ const r = await spawnAgent(agent, decisionPrompt(task, effectiveCapabilities(agent, merged) ?? agent.capabilities), cfg.decisionTimeoutSeconds ?? 90, agentEnv(cfg).env, undefined, { cfg });
1094
1174
  decision = parseDecision(displayText(r.out));
1095
1175
  } catch {
1096
1176
  // Conservative THIS poll — but a timeout/hiccup is not the model's answer,
@@ -1149,8 +1229,10 @@ async function considerVolunteering(cfg, ws, task, budget) {
1149
1229
  }
1150
1230
 
1151
1231
  async function processTask(cfg, ws, task, agent) {
1232
+ if (inFlight.has(task.id)) return; // the janitor's stale snapshot vs the pull: one launch only
1152
1233
  inFlight.add(task.id);
1153
1234
  inFlightWs.set(task.id, ws.id);
1235
+ inFlightThreadKeys.set(task.id, task.thread_root_id ?? task.id);
1154
1236
  // PRE-CLAIM (Phase 0, audit #1): a dispatched task must be OURS before we spend
1155
1237
  // quota on it. Without this, a to:'any' task — or the same member's Bridge on a
1156
1238
  // second machine — ran N times and the losers found out at the 409 after paying
@@ -1182,7 +1264,43 @@ async function processTask(cfg, ws, task, agent) {
1182
1264
  attempts.set(task.id, (attempts.get(task.id) ?? 0) + 1);
1183
1265
  saveRunState();
1184
1266
  const n = attempts.get(task.id);
1185
- log(`→ waking ${agent.name} for "${task.title}" in ${ws.name} (attempt ${n}/${cfg.maxAttempts})`);
1267
+ // THE LIVE LANE (realtime.mjs createLivePublisher): the server handed this run a
1268
+ // per-task topic; the browser is subscribed to it. First publish = the
1269
+ // acknowledgement ("your Claude picked this up"), before any process spawns.
1270
+ if (task.server_claimed === true) log(` ↳ pre-claimed by the server in the pull (one hop)`);
1271
+ const liveTopic = typeof task.live_topic === "string" && /^blive:[0-9a-f]{64}$/.test(task.live_topic) ? task.live_topic : null;
1272
+ const livePub = liveTopic && livePublisher ? livePublisher : null;
1273
+ const tClaim = Date.now();
1274
+ const sinceClaim = () => `${((Date.now() - tClaim) / 1000).toFixed(1)}s`;
1275
+ // STAGES (honest failures): claimed, booted, first words, done or failed, as
1276
+ // milliseconds after the claim. They ride progress ticks live and the usage
1277
+ // report (or the failure) at the end, so the card can draw where a run was.
1278
+ const stages = { claimed: tClaim };
1279
+ // THE RECORD (bridge/trace.mjs): every tool call with its result, filed with the
1280
+ // run's conditions when it ends, done or abandoned. Evidence, never a dependency.
1281
+ const traceCtx = { trace: null, meta: null, recalled: [], cross: [], resume: null, chat: false, model: null };
1282
+ const fileTrace = (outcome) => {
1283
+ if (!traceCtx.trace || !traceEnvelope || !postTrace) return;
1284
+ try {
1285
+ const body = traceEnvelope({
1286
+ workspaceId: ws.id, taskId: task.id, outcome, agent, cfg, model: traceCtx.model,
1287
+ promptVersion: PROMPT_VERSION ?? "", bridgeVersion: BRIDGE_VERSION,
1288
+ recalled: traceCtx.recalled, crossRecalled: traceCtx.cross, resume: traceCtx.resume, chat: traceCtx.chat, stages,
1289
+ localMeta: traceCtx.meta ? { cwd: traceCtx.meta.local_cwd ?? null, mode: traceCtx.meta.local_mode ?? null } : null,
1290
+ trace: traceCtx.trace.finish(),
1291
+ });
1292
+ postTrace(cfg, body).then((ok) => {
1293
+ if (ok) log(` ↳ trace filed: ${body.trace.calls.length} call${body.trace.calls.length === 1 ? "" : "s"}${body.trace.truncated.complete ? "" : " (truncated)"}`);
1294
+ }).catch(() => {});
1295
+ } catch { /* the trace is evidence, never a reason a run fails */ }
1296
+ };
1297
+ const stageStamps = () => (stagesPayload ? stagesPayload(stages) : {});
1298
+ const livePublish = (stage, extra = {}) => {
1299
+ if (!livePub) return;
1300
+ try { livePub.publish(liveTopic, { v: 1, task_id: task.id, stage, runner: agent.name, at: new Date().toISOString(), ...extra }); } catch { /* fast lane only */ }
1301
+ };
1302
+ livePublish("claimed");
1303
+ log(`→ waking ${agent.name} for "${task.title}" in ${ws.name} (attempt ${n}/${cfg.maxAttempts})${task.chat ? " · chat" : ""}${livePub ? " · live lane" : ""}`);
1186
1304
  let resume = null; // hoisted: the catch path needs to know if the run RESUMED a session
1187
1305
  // PLAN HOLD: a run that discovered a closed plan window (rate_limit_event) is not
1188
1306
  // a failed attempt. Give the attempt back, shelve the task, and let agentHeld()
@@ -1190,6 +1308,10 @@ async function processTask(cfg, ws, task, agent) {
1190
1308
  const runStartedAt = Date.now();
1191
1309
  const holdDuringRun = () => { const h = planHoldFor(agentVendor(agent)); return !!h && h.at >= runStartedAt; };
1192
1310
  const shelveForHold = (reason, sessionId) => {
1311
+ const h0 = planHoldFor(agentVendor(agent));
1312
+ const when = h0 ? "at " + new Date(h0.until).toLocaleTimeString("en-US", { hour: "numeric", minute: "2-digit", timeZoneName: "short" }) : "soon";
1313
+ const keep = sessionId ?? retryCtx.get(task.id)?.sessionId ?? null;
1314
+ reportTaskProgress(cfg, ws.id, task.id, { stage: `held: ${agent.name}'s plan window resets ${when}`, ...(keep ? { session_ref: keep } : {}), ...stageStamps() }).catch(() => {});
1193
1315
  attempts.set(task.id, Math.max(0, (attempts.get(task.id) ?? 1) - 1));
1194
1316
  retryCtx.set(task.id, { sessionId: sessionId ?? retryCtx.get(task.id)?.sessionId ?? null, reason: reason || "plan limit reached" });
1195
1317
  saveRunState();
@@ -1203,28 +1325,50 @@ async function processTask(cfg, ws, task, agent) {
1203
1325
  // Composer follow-ups (0064) skip recall entirely: a resumed session already
1204
1326
  // carries what rode into the original run, and crediting notes that never rode
1205
1327
  // into THIS prompt would corrupt the outcome-weighted signal.
1206
- const { memories, conventions } = task.thread_root_id
1207
- ? { memories: [], conventions: [] }
1208
- : await recallMemories(cfg, ws.id, task.title);
1328
+ // Both recall calls are independent round trips: run them together (a chat
1329
+ // root paid for them back to back before the runner could even be fed).
1330
+ const recallP = task.thread_root_id ? Promise.resolve({ memories: [], conventions: [] }) : recallMemories(cfg, ws.id, task.title);
1331
+ const crossP = task.thread_root_id ? Promise.resolve([]) : recallAcrossWorkspaces(cfg, task.title, ws.id, 3);
1332
+ const { memories, conventions } = await recallP;
1209
1333
  // Credit both classes on verified completion — conventions earn helpful_count too
1210
1334
  // (the outcome signal that ranks proven rules first).
1211
1335
  const recalledIds = [...memories, ...conventions].map((m) => m && m.id).filter(Boolean);
1336
+ traceCtx.recalled = recalledIds;
1212
1337
  if (memories.length) log(` ↳ injecting ${memories.length} team-memory note${memories.length === 1 ? "" : "s"}`);
1213
1338
  if (conventions.length) log(` ↳ + ${conventions.length} team convention${conventions.length === 1 ? "" : "s"} (verbatim)`);
1214
1339
  // Proactive cross-workspace recall: proven knowledge from the member's OTHER projects
1215
1340
  // (a playbook, a gotcha) surfaces here without being pointed at it. Best-effort.
1216
- const crossWorkspace = task.thread_root_id ? [] : await recallAcrossWorkspaces(cfg, task.title, ws.id, 3);
1341
+ const crossWorkspace = await crossP;
1342
+ traceCtx.cross = crossWorkspace.map((m) => m && m.id).filter(Boolean);
1217
1343
  if (crossWorkspace.length) log(` ↳ + ${crossWorkspace.length} proven note${crossWorkspace.length === 1 ? "" : "s"} from your other projects`);
1218
1344
  const startedAt = Date.now();
1219
1345
  // Live ticker: stream in-flight token counts to the board so the assigner watches
1220
1346
  // the cost accrue. Fire-and-forget + swallow errors — a progress hiccup must never
1221
1347
  // touch the run. (report_task_progress is a no-op once the task leaves 'claimed'.)
1222
1348
  let lastProgressPost = 0;
1349
+ let lastLivePost = 0;
1350
+ let firstWordsAt = 0;
1223
1351
  let localMeta = null; // { local_cwd, local_mode } once local access is decided below
1224
1352
  const onProgress = (p) => {
1225
- if (Date.now() - lastProgressPost < 1000) return;
1226
- lastProgressPost = Date.now();
1227
- reportTaskProgress(cfg, ws.id, task.id, localMeta ? { ...p, ...localMeta } : p).catch(() => {});
1353
+ const now = Date.now();
1354
+ if (!firstWordsAt && (p.live_text || (Array.isArray(p.live_calls) && p.live_calls.length))) {
1355
+ firstWordsAt = now;
1356
+ stages.first_words = now;
1357
+ log(` ⏱ first words ${sinceClaim()} after claim`);
1358
+ }
1359
+ // Fast lane: the live tail straight to the thread view, every 200 ms at most.
1360
+ if (livePub && now - lastLivePost >= 200) {
1361
+ lastLivePost = now;
1362
+ livePublish("working", {
1363
+ ...(p.live_text ? { live_text: p.live_text } : {}),
1364
+ ...(Array.isArray(p.live_calls) && p.live_calls.length ? { live_calls: p.live_calls } : {}),
1365
+ output_tokens: p.output_tokens ?? 0,
1366
+ });
1367
+ }
1368
+ // Record lane: the database (receipt, late joiners), every 1.5 s.
1369
+ if (now - lastProgressPost < 1500) return;
1370
+ lastProgressPost = now;
1371
+ reportTaskProgress(cfg, ws.id, task.id, { ...p, ...(localMeta ?? {}), ...stageStamps() }).catch(() => {});
1228
1372
  };
1229
1373
  // Retry attempts CONTINUE, not redo (Phase 1): with a saved claude session the
1230
1374
  // prompt is just the next message in the resumed conversation; without one, the
@@ -1296,15 +1440,21 @@ async function processTask(cfg, ws, task, agent) {
1296
1440
  let warmRunner = runnerEligible ? hasRunner(threadKey) : null;
1297
1441
  // ADOPT a pre-warmed runner (0065) for a NEW conversation: the process booted
1298
1442
  // while the member was still typing, so their first words hit a live agent.
1299
- if (!warmRunner && runnerEligible && !task.thread_root_id && !local) {
1300
- // (Local runs never adopt from the warm pool — pooled processes are jailed
1301
- // to workspace tools and the wrong cwd.)
1302
- warmRunner = adoptRunner(`warm::${ws.id}::${agent.name}`, threadKey);
1303
- if (warmRunner) log(` ↳ adopted a pre-warmed ${agent.name} — first message hits a live process`);
1443
+ if (!warmRunner && runnerEligible && !task.thread_root_id && !task.model) {
1444
+ // Local runs adopt only their LOCAL twin (warmed with the folder's cwd and
1445
+ // tools); jailed spares never serve a local turn, nor vice versa. A task
1446
+ // pinned to a model never adopts: the spare runs the default model and
1447
+ // would only be killed and rebooted a few lines down.
1448
+ warmRunner = adoptRunner(`warm::${ws.id}::${agent.name}${local ? "::local" : ""}`, threadKey);
1449
+ if (warmRunner) log(` ↳ adopted a pre-warmed ${agent.name}${local ? " (local)" : ""} — first message hits a live process`);
1304
1450
  }
1305
1451
  let thread = null;
1306
1452
  if (task.thread_root_id && !warmRunner) {
1307
- thread = await threadResumeContext(cfg, ws.id, task.thread_root_id, myProfileId).catch(() => null);
1453
+ // The pull may have carried the thread's resume handle (preclaim.ts); the
1454
+ // listing round trip is only for tasks that arrived without it.
1455
+ thread = task.resume && typeof task.resume === "object"
1456
+ ? { sessionRef: task.resume.session_ref ?? null, root: task.resume.root ?? null }
1457
+ : await threadResumeContext(cfg, ws.id, task.thread_root_id, myProfileId).catch(() => null);
1308
1458
  if (thread?.root?.status === "cancelled") {
1309
1459
  inFlight.delete(task.id);
1310
1460
  givenUp.add(task.id);
@@ -1325,14 +1475,26 @@ async function processTask(cfg, ws, task, agent) {
1325
1475
  // Bridge files it — saves a whole model round-trip (the complete_task tool
1326
1476
  // call) plus the verify fetch, every single turn. Applies to claude runners
1327
1477
  // AND the persistent codex server (its final agent message is the answer).
1328
- const bridgeFiles = cfg.persistentThreads && agent.runner !== "robot";
1478
+ // CHAT: the final message IS the answer and the Bridge files it. Board tasks
1479
+ // keep the complete_task contract unless the config asked for bridgeFilesAll.
1480
+ const bridgeFiles = agent.runner !== "robot" && (task.chat === true || cfg.bridgeFilesAll === true);
1481
+ // CHAT TURNS (server-flagged: the task carries a transcript row) get the chat
1482
+ // prompt: answer first, tools only when they help. Board tasks keep the
1483
+ // autonomous task prompt with its decomposition and capture contract.
1484
+ const chat = task.chat === true && !!buildChatPrompt;
1485
+ traceCtx.chat = chat;
1486
+ traceCtx.model = task.model || null;
1329
1487
  let basePrompt = task.thread_root_id
1330
- ? buildThreadFollowUpPrompt(ws, task, { resumed: conversationWarm, root: thread?.root ?? null, bridgeFiles })
1331
- : buildPrompt(ws, task, { memories, conventions, crossWorkspace, volunteered: task.claimed_via === "volunteered", bridgeFiles });
1488
+ ? (chat
1489
+ ? buildChatFollowUpPrompt(ws, task, { resumed: conversationWarm, root: thread?.root ?? null, bridgeFiles })
1490
+ : buildThreadFollowUpPrompt(ws, task, { resumed: conversationWarm, root: thread?.root ?? null, bridgeFiles }))
1491
+ : (chat
1492
+ ? buildChatPrompt(ws, task, { memories, conventions, crossWorkspace, bridgeFiles, agentName: agent.name })
1493
+ : buildPrompt(ws, task, { memories, conventions, crossWorkspace, volunteered: task.claimed_via === "volunteered", bridgeFiles }));
1332
1494
  if (local) {
1333
1495
  basePrompt += `\n\nLOCAL ACCESS: you are running ON the member's machine in ${local.cwd} — this folder is the workspace's local project. You have real file and shell tools; use them for the actual work (build artifacts, code, sites live HERE). Mirror durable outcomes into the Cookbook workspace (files / remember) so the team side stays true.`;
1334
1496
  }
1335
- const prompt = retry?.sessionId
1497
+ let prompt = retry?.sessionId
1336
1498
  ? `Your previous attempt on this task was interrupted: ${String(retry.reason ?? "unknown failure").slice(0, 300)}. ` +
1337
1499
  `Continue EXACTLY where you left off — do not redo completed work. If you are close, finish and call complete_task; ` +
1338
1500
  `if the task is impossible from this environment, call abandon_task with the reason.`
@@ -1345,6 +1507,9 @@ async function processTask(cfg, ws, task, agent) {
1345
1507
  else if (canResumeThread) log(` ↳ resuming the thread's conversation (Composer follow-up)`);
1346
1508
  else if (task.thread_root_id) log(` ↳ thread follow-up, no resumable session — running with the cold baton`);
1347
1509
  resume = retry ?? (canResumeThread ? { sessionId: threadSession } : null);
1510
+ traceCtx.resume = resume;
1511
+ traceCtx.meta = localMeta;
1512
+ traceCtx.trace = createTrace ? createTrace() : null;
1348
1513
  let result = null;
1349
1514
  // approvalRelay runs (0086) NEVER take the runner: stream-json input mode
1350
1515
  // denies non-allowlisted tools without consulting --permission-prompt-tool.
@@ -1368,11 +1533,14 @@ async function processTask(cfg, ws, task, agent) {
1368
1533
  model: taskModel,
1369
1534
  });
1370
1535
  if (taskModel) log(` ↳ model: ${taskModel}`);
1536
+ if (!usable) stages.booted = Date.now(); // a warm runner was ready before the claim
1537
+ log(` ⏱ runner ${usable ? "warm" : r.sends === 0 ? "booted" : "reused"} ${sinceClaim()} after claim`);
1371
1538
  killers.set(task.id, () => r.kill());
1372
1539
  result = await r.send(prompt, {
1373
1540
  onProgress: live ? onProgress : undefined,
1374
1541
  timeoutMs: cfg.taskTimeoutSeconds * 1000,
1375
1542
  livenessMs: (cfg.livenessTimeoutSeconds ?? 0) * 1000,
1543
+ trace: traceCtx.trace,
1376
1544
  });
1377
1545
  } catch (e) {
1378
1546
  // busy / not claude / a runner that died or never launched: all fallback
@@ -1380,18 +1548,31 @@ async function processTask(cfg, ws, task, agent) {
1380
1548
  if (/busy|not a claude-shaped|runner is dead|failed to launch/.test(e.message)) {
1381
1549
  log(` ↳ runner unavailable (${e.message}) — one-shot fallback`);
1382
1550
  result = null;
1551
+ // A thread turn that assumed a warm process must NOT go cold with the
1552
+ // "resumed" prompt (no history in it). Rebuild it as the cold baton.
1553
+ if (task.thread_root_id && warmRunner && !retry) {
1554
+ const ctx = await threadResumeContext(cfg, ws.id, task.thread_root_id, myProfileId).catch(() => null);
1555
+ const cold = { resumed: false, root: ctx?.root ?? null, bridgeFiles };
1556
+ prompt = chat ? buildChatFollowUpPrompt(ws, task, cold) : buildThreadFollowUpPrompt(ws, task, cold);
1557
+ const sess = ctx?.sessionRef && resumeCommand(agent.command, ctx.sessionRef).resumed ? ctx.sessionRef : null;
1558
+ resume = sess ? { sessionId: sess } : null;
1559
+ }
1383
1560
  } else {
1384
1561
  throw e; // real failure: ride the existing retry/abandon machinery
1385
1562
  }
1386
1563
  }
1387
1564
  }
1388
- if (!result) result = await runAgent(cfg, agent, prompt, onProgress, resume, { ws, task, onChild: (child) => killers.set(task.id, () => killChild(child, "SIGTERM")) });
1565
+ if (!result) {
1566
+ stages.booted = stages.booted ?? Date.now(); // one-shot and app-server paths start here
1567
+ result = await runAgent(cfg, agent, prompt, onProgress, resume, { ws, task, trace: traceCtx.trace, onChild: (child) => killers.set(task.id, () => killChild(child, "SIGTERM")) });
1568
+ }
1389
1569
  if (result && result.sessionId) retryCtx.set(task.id, { sessionId: result.sessionId, reason: retryCtx.get(task.id)?.reason ?? null });
1390
1570
  let after;
1391
1571
  if (bridgeFiles) {
1392
1572
  // File the final message as the result (agent may still have abandoned or
1393
1573
  // self-completed via tools — any conflict just falls back to reading state).
1394
1574
  const finalText = scrub(displayText(result?.out)).trim().slice(0, 20_000);
1575
+ if (finalText) livePublish("working", { live_text: finalText.length > 1800 ? "…" + finalText.slice(-1800) : finalText });
1395
1576
  // Success shape differs by runner: claude one-shots exit 0; the codex server
1396
1577
  // resolves with a turn status (no exit code) — failed statuses fall through.
1397
1578
  const cleanExit = result?.code === 0 || (result?.code === undefined && !/failed/i.test(String(result?.status ?? "")));
@@ -1415,18 +1596,23 @@ async function processTask(cfg, ws, task, agent) {
1415
1596
  }
1416
1597
  if (after?.status === "done") {
1417
1598
  retryCtx.delete(task.id);
1418
- log(`✓ ${agent.name} completed "${task.title}"`);
1599
+ // A finished Codex turn proves the login works: clear a signed-out flag.
1600
+ if (agent.runner === "app-server" && noteAuth("codex", true)) log(" ↳ Codex is signed in again");
1601
+ livePublish("done");
1602
+ log(`✓ ${agent.name} completed "${task.title}" · ${sinceClaim()} after claim${firstWordsAt ? ` (first words at ${((firstWordsAt - tClaim) / 1000).toFixed(1)}s)` : ""}`);
1419
1603
  emitRun("done", ws, task, agent, localMeta);
1420
1604
  // Outcome signal: credit the memory notes that rode into this SUCCESSFUL run,
1421
1605
  // so proven notes surface first next time (outcome-weighted recall). Best-effort.
1422
- if (recalledIds.length) creditRecall(cfg, ws.id, recalledIds).catch(() => {});
1606
+ if (recalledIds.length) creditRecall(cfg, ws.id, recalledIds, task.id).catch(() => {});
1607
+ stages.done = Date.now();
1608
+ fileTrace("done");
1423
1609
  // Quota visibility: tell Cookbook what this run cost (tokens/cost from the CLI's
1424
1610
  // own report when available, wall time always) so the assigner sees the price of
1425
1611
  // the delegation. Fire-and-forget — a usage hiccup must never fail a done task.
1426
1612
  try {
1427
1613
  const usage = extractUsage(result, agent.name, Date.now() - startedAt);
1428
1614
  if (usage) {
1429
- await reportTaskUsage(cfg, ws.id, task.id, usage);
1615
+ await reportTaskUsage(cfg, ws.id, task.id, { ...usage, ...stageStamps() });
1430
1616
  const tok = (usage.input_tokens ?? 0) + (usage.output_tokens ?? 0);
1431
1617
  log(` ↳ usage reported${tok ? `: ${tok.toLocaleString()} tokens` : ""}${usage.cost_usd ? ` · ~$${usage.cost_usd.toFixed(2)}` : ""}`);
1432
1618
  }
@@ -1437,11 +1623,28 @@ async function processTask(cfg, ws, task, agent) {
1437
1623
  // The agent CLI exited but the task isn't done — surface WHY (login/MCP/tools),
1438
1624
  // instead of the old silent "ran but isn't marked done". This is the line that
1439
1625
  // turns a multi-hour debug into a one-glance fix.
1440
- const hint = failureHint(result);
1626
+ // HONEST FAILURES: one structured cause for the log, the card and the receipt.
1627
+ const failure = classifyFailure ? classifyFailure({ result, vendor: agentVendor(agent), agent }) : null;
1628
+ if (failure && failure.kind === "plan_window" && !holdDuringRun()) failure.fix = "Retry when the window resets, or pick another agent for this turn.";
1629
+ const hint = failure ? failureLine(failure) : failureHint(result, agent);
1441
1630
  const why = hint ? ` — ${hint}` : "";
1631
+ stages.failed = Date.now();
1632
+ if (/signed out/.test(hint) && isClaudeCommand && isClaudeCommand(agent.command)) {
1633
+ if (noteAuth("claude", false)) log("! Claude is signed out on this machine → run `claude` in a terminal and sign in. Until then Claude work here falls back or waits.");
1634
+ lastAuthCheck = Date.now() - AUTH_CHECK_MS + 60_000; // re-check in a minute
1635
+ }
1636
+ if (/^Codex on this machine is signed out/.test(hint)) {
1637
+ // Ride the heartbeat like Claude does (0100): the room shows "signed out"
1638
+ // and readers stop routing to this Codex until a run succeeds again.
1639
+ if (noteAuth("codex", false)) log("! Codex is signed out on this machine → run `CODEX_HOME=~/.codex-bridge codex login` and sign in. Until then Codex work here waits.");
1640
+ }
1641
+ // A terminal cause (signed out, CLI too old, a blocked tool) will not fix itself
1642
+ // between attempts: hand the task back NOW with the reason instead of burning
1643
+ // the retry. Transient causes (network, a quiet process) get the second try.
1644
+ const terminal = failure ? failure.terminal === true : /signed out|too old for your account/.test(hint);
1442
1645
  if (holdDuringRun()) {
1443
1646
  shelveForHold(hint || "plan limit reached", result?.sessionId ?? null);
1444
- } else if (n >= cfg.maxAttempts) {
1647
+ } else if (n >= cfg.maxAttempts || terminal) {
1445
1648
  givenUp.add(task.id);
1446
1649
  saveRunState();
1447
1650
  // Final attempt: account the burn (see catch path). extractUsage reads the
@@ -1454,7 +1657,8 @@ async function processTask(cfg, ws, task, agent) {
1454
1657
  // so the ASSIGNER sees "tried, gave up, here's why" on the board — instead
1455
1658
  // of a task that silently rots open (or strands claimed, invisible to all).
1456
1659
  retryCtx.delete(task.id);
1457
- await abandonTask(cfg, ws.id, task.id, hint || `ran ${n} attempt(s) without completing`);
1660
+ await abandonTask(cfg, ws.id, task.id, hint || `ran ${n} attempt(s) without completing`, failure && failurePayload ? failurePayload(failure, stages) : null);
1661
+ fileTrace("abandoned");
1458
1662
  emitRun("failed", ws, task, agent, localMeta);
1459
1663
  log(`✗ ${agent.name} didn't complete "${task.title}" after ${n} attempts${why} — handed back as abandoned.`);
1460
1664
  } else {
@@ -1466,7 +1670,8 @@ async function processTask(cfg, ws, task, agent) {
1466
1670
  const resumedAndFailed = task.thread_root_id && resume?.sessionId;
1467
1671
  retryCtx.set(task.id, {
1468
1672
  sessionId: resumedAndFailed ? null : result?.sessionId ?? retryCtx.get(task.id)?.sessionId ?? null,
1469
- reason: hint || "ran but did not complete the task",
1673
+ // The model gets the cause, never the human-facing fix text.
1674
+ reason: (failure ? failure.message : hint) || "ran but did not complete the task",
1470
1675
  });
1471
1676
  saveRunState();
1472
1677
  // EVERY failed run is a CLAIMED task now (Phase 0 pre-claims dispatch too),
@@ -1478,13 +1683,22 @@ async function processTask(cfg, ws, task, agent) {
1478
1683
  }
1479
1684
  }
1480
1685
  } catch (e) {
1481
- if (holdDuringRun() && !stoppedRuns.has(task.id)) {
1686
+ stages.failed = Date.now();
1687
+ const thrownFailure = classifyFailure ? classifyFailure({ error: e, vendor: agentVendor(agent), agent }) : null;
1688
+ if (stoppedRuns.has(task.id)) {
1689
+ // A human pressed Stop: the task is already cancelled server-side; nothing to
1690
+ // retry, nothing to abandon. Checked first so a kill message that happens to
1691
+ // read like a terminal cause cannot route a Stop into abandon_task.
1692
+ stoppedRuns.delete(task.id);
1693
+ givenUp.add(task.id);
1694
+ } else if (holdDuringRun()) {
1482
1695
  shelveForHold(e.message, e.sessionId ?? null);
1483
- } else if (n >= cfg.maxAttempts) {
1696
+ } else if (n >= cfg.maxAttempts || (thrownFailure && thrownFailure.terminal === true)) {
1484
1697
  givenUp.add(task.id);
1485
1698
  saveRunState();
1486
1699
  retryCtx.delete(task.id);
1487
- await abandonTask(cfg, ws.id, task.id, e.message || "failed after final attempt");
1700
+ await abandonTask(cfg, ws.id, task.id, thrownFailure ? failureLine(thrownFailure) : (e.message || "failed after final attempt"), thrownFailure && failurePayload ? failurePayload(thrownFailure, stages) : null);
1701
+ fileTrace("abandoned");
1488
1702
  emitRun("failed", ws, task, agent, null);
1489
1703
  // FINAL attempt failed: report what the failed runs actually burned, so the
1490
1704
  // chain token budget sees it. Only on the LAST attempt — usage is first-
@@ -1504,12 +1718,6 @@ async function processTask(cfg, ws, task, agent) {
1504
1718
  // A volunteered task is CLAIMED — invisible to the open scan — so the throw
1505
1719
  // branch (timeouts are the COMMON failure here) must feed the retry shelf
1506
1720
  // exactly like the ran-but-not-done branch, or the claim strands.
1507
- else if (stoppedRuns.has(task.id)) {
1508
- // A human pressed Stop: the task is already cancelled server-side; nothing to
1509
- // retry, nothing to abandon.
1510
- stoppedRuns.delete(task.id);
1511
- givenUp.add(task.id);
1512
- }
1513
1721
  else {
1514
1722
  // Same thread self-heal as the ran-not-done branch: a failed RESUMED thread
1515
1723
  // attempt retries cold rather than back into the same session.
@@ -1522,7 +1730,9 @@ async function processTask(cfg, ws, task, agent) {
1522
1730
  lastRunError = `${agent.name}: ${String(e.message).slice(0, 300)}`;
1523
1731
  log(`✗ ${agent.name} error on "${task.title}": ${e.message}${slow ? ` — the run hit taskTimeoutSeconds (${slow[1]}s); raise it in config.json if this task is just slow` : ""}`);
1524
1732
  } finally {
1733
+ if (livePub) { try { livePub.leave(liveTopic); } catch { /* fast lane only */ } }
1525
1734
  inFlight.delete(task.id);
1735
+ inFlightThreadKeys.delete(task.id);
1526
1736
  inFlightWs.delete(task.id);
1527
1737
  killers.delete(task.id);
1528
1738
  markHot(ws.id); // the reply usually lands right after a run finishes — stay fast for it
@@ -1727,21 +1937,43 @@ const sse = { connected: false, supported: true, aliveAt: 0 };
1727
1937
  // "ring"; /api/bridge/pull carries everything the stream used to push (work,
1728
1938
  // warm hints, stops, hands) and stamps liveness. SSE remains the fallback lane.
1729
1939
  const door = { active: false, connected: false, lastPullOkAt: 0 };
1940
+ // The live lane's publisher (realtime.mjs). Created with the doorbell (same
1941
+ // Realtime creds from the manifest); null = the database path alone paints.
1942
+ let livePublisher = null;
1730
1943
 
1731
1944
  /** One authenticated pull = one former SSE frame. Returns false only when the
1732
1945
  * server predates /api/bridge/pull (route missing → use the SSE lane). */
1733
1946
  async function pullOnce(cfg, { boot = false } = {}) {
1734
1947
  const aq = agentsQuery(cfg);
1735
1948
  const bootQ = boot ? `${aq ? "&" : "?"}boot=${BOOT_MS}` : "";
1736
- const res = await fetch(`${cfg.cookbookUrl}/api/bridge/pull${aq}${bootQ}`, {
1949
+ // ONE HOP TO CLAIM: with free slots, the server claims our next tasks inside the
1950
+ // pull and returns them ready to run (delegation resolved, resume handle
1951
+ // attached). We advertise only what we can claim RIGHT NOW (enabled, not in a
1952
+ // plan hold, and no local acceptFrom allowlist the server cannot evaluate), and
1953
+ // we ask for nothing for a minute after handing anything back, so a mismatch can
1954
+ // never loop faster than the beat. Zero slots = plain open work, as before.
1955
+ for (const [id, until] of releasedRecently) if (until <= Date.now()) releasedRecently.delete(id);
1956
+ const slots = Math.max(0, (cfg.maxConcurrentRuns ?? 1) - inFlight.size);
1957
+ const claimable = (cfg.agents ?? []).filter((a) => a && a.enabled !== false && a.name && !agentHeld(a) && !(a.acceptFrom ?? cfg.acceptFrom));
1958
+ const wantClaim = slots > 0 && claimable.length > 0 && releasedRecently.size === 0;
1959
+ const sep = () => (aq || bootQ ? "&" : "?");
1960
+ const claimQ = wantClaim ? `${sep()}claim=${slots}&claim_agents=${encodeURIComponent(claimable.map((a) => String(a.name)).join(","))}` : "";
1961
+ retryPendingReleases(cfg);
1962
+ const res = await fetch(`${cfg.cookbookUrl}/api/bridge/pull${aq}${bootQ}${claimQ}`, {
1737
1963
  headers: { Authorization: `Bearer ${cfg.token}` },
1738
1964
  });
1739
1965
  if (res.status === 404 || res.status === 405) return false;
1740
1966
  if (!res.ok) throw new Error(`pull ${res.status}`);
1741
1967
  const j = await res.json();
1742
1968
  door.lastPullOkAt = Date.now();
1743
- if (Array.isArray(j.stops) && j.stops.length) stopRuns(j.stops);
1744
- await dispatchWork(cfg, j.work ?? [], j.warm_hints ?? []);
1969
+ try {
1970
+ if (Array.isArray(j.stops) && j.stops.length) stopRuns(j.stops);
1971
+ await dispatchWork(cfg, j.work ?? [], j.warm_hints ?? []);
1972
+ } catch (e) {
1973
+ // Whatever the server claimed for this pull and we did not start goes back.
1974
+ for (const t of j.work ?? []) if (isPreclaimed(t) && !inFlight.has(t.id)) giveBack(cfg, t, `dispatch failed: ${e.message}`);
1975
+ throw e;
1976
+ }
1745
1977
  if (j.hands && typeof j.hands === "object") {
1746
1978
  hands.activeGrants = j.hands.grants ?? hands.activeGrants;
1747
1979
  noteAwaiting(j.hands.awaiting ?? []);
@@ -1768,6 +2000,12 @@ async function doorbellLoop(cfg) {
1768
2000
  if (res.ok) rt = (await res.json()).realtime ?? null;
1769
2001
  } catch { /* manifest unreachable — the SSE lane copes */ }
1770
2002
  if (!rt?.url || !rt?.anonKey) return false;
2003
+ // The live publisher must exist BEFORE the boot pull: that pull can pre-claim
2004
+ // and start a run, and a run started without a publisher streams nothing
2005
+ // (seen 2026-09-15: a task resumed after a Bridge restart had no live lane).
2006
+ if (!livePublisher) {
2007
+ try { livePublisher = createLivePublisher({ url: rt.url, anonKey: rt.anonKey, log }); } catch { livePublisher = null; }
2008
+ }
1771
2009
  let first;
1772
2010
  try { first = await pullOnce(cfg, { boot: true }); } catch { return false; }
1773
2011
  if (first === false) return false; // no pull route on this server
@@ -1888,16 +2126,52 @@ let sseStopped = false;
1888
2126
  * Re-entrancy-guarded: overlapping snapshots are redundant (each is a full picture),
1889
2127
  * and claim CAS + inFlight make any stragglers harmless anyway. */
1890
2128
  let dispatchBusy = false;
2129
+ // A snapshot that lands mid-pass is MERGED and run right after, never dropped: a
2130
+ // dropped snapshot could carry tasks the server had just pre-claimed for us.
2131
+ let pendingDispatch = null;
2132
+ function mergeSnapshot(a, b) {
2133
+ const byId = new Map();
2134
+ for (const t of [...(a?.work ?? []), ...(b?.work ?? [])]) if (t && t.id && !(byId.has(t.id) && isPreclaimed(byId.get(t.id)))) byId.set(t.id, t);
2135
+ return { work: [...byId.values()], warmHints: [...(a?.warmHints ?? []), ...(b?.warmHints ?? [])] };
2136
+ }
1891
2137
  async function dispatchWork(cfg, work, warmHints) {
1892
- if (dispatchBusy) return;
2138
+ if (dispatchBusy) { pendingDispatch = mergeSnapshot(pendingDispatch, { work, warmHints }); return; }
1893
2139
  dispatchBusy = true;
1894
2140
  try {
1895
2141
  await dispatchWorkInner(cfg, work, warmHints);
2142
+ while (pendingDispatch) {
2143
+ const p = pendingDispatch;
2144
+ pendingDispatch = null;
2145
+ await dispatchWorkInner(cfg, p.work, p.warmHints);
2146
+ }
1896
2147
  } finally {
1897
2148
  dispatchBusy = false;
1898
2149
  }
1899
2150
  }
1900
2151
 
2152
+ // ── PRE-CLAIM BOOKKEEPING (one hop to claim) ────────────────────────────────
2153
+ // Once a task is claimed nothing lists it again, so "pre-claimed and not started"
2154
+ // must always end in a release. These helpers make that an invariant.
2155
+ const releasedRecently = new Map(); // task id -> until (ms): no claim requests while non-empty
2156
+ const pendingReleases = new Map(); // task id -> workspace id: a release that failed, retried on the next pull
2157
+ function isPreclaimed(t) {
2158
+ return !!t && t.server_claimed === true && t.status === "claimed" && (!myProfileId || t.claimed_by_profile === myProfileId);
2159
+ }
2160
+ function giveBack(cfg, t, why) {
2161
+ if (releasedRecently.has(t.id) && !pendingReleases.has(t.id)) return; // already handed back this minute
2162
+ log(` ↳ releasing "${t.title}" (${why})`);
2163
+ releasedRecently.set(t.id, Date.now() + 60_000);
2164
+ releaseTask(cfg, t.workspace_id, t.id).then((ok) => {
2165
+ if (ok) pendingReleases.delete(t.id);
2166
+ else pendingReleases.set(t.id, t.workspace_id);
2167
+ }).catch(() => pendingReleases.set(t.id, t.workspace_id));
2168
+ }
2169
+ function retryPendingReleases(cfg) {
2170
+ for (const [id, wsId] of pendingReleases) {
2171
+ releaseTask(cfg, wsId, id).then((ok) => { if (ok) pendingReleases.delete(id); }).catch(() => {});
2172
+ }
2173
+ }
2174
+
1901
2175
  async function dispatchWorkInner(cfg, work, warmHints) {
1902
2176
  // PRE-WARM (0065): the chat surface hinted a conversation is imminent — boot the
1903
2177
  // runner NOW so the first message hits a live process instead of a cold spawn.
@@ -1905,13 +2179,24 @@ async function dispatchWorkInner(cfg, work, warmHints) {
1905
2179
  for (const h of warmHints ?? []) {
1906
2180
  const agent = agentFor(cfg, h.agent);
1907
2181
  if (!agent || agent.runner === "app-server" || agent.runner === "robot") continue;
1908
- warmUp({
1909
- poolKey: `warm::${h.workspace_id}::${agent.name}`,
1910
- agent: pinnedAgent(cfg, agent),
1911
- env: agentEnv(cfg).env,
1912
- helpers: { fold: foldStreamLine, textFrom: textFromStreamLine, sessionFrom: sessionIdFrom, onPlan: notePlanHold },
1913
- log,
1914
- });
2182
+ const helpers = { fold: foldStreamLine, textFrom: textFromStreamLine, sessionFrom: sessionIdFrom, onPlan: notePlanHold };
2183
+ warmUp({ poolKey: `warm::${h.workspace_id}::${agent.name}`, agent: pinnedAgent(cfg, agent), env: agentEnv(cfg).env, helpers, log, cap: cfg.maxRunners });
2184
+ // A locally-mapped workspace's chat runs IN the folder with real tools, which
2185
+ // the jailed spare above can't serve. Warm a local twin too (same shape
2186
+ // processTask builds), so the first message in Diego's dev workspace hits a
2187
+ // live process instead of a cold spawn. Ask mode never uses the runner.
2188
+ const local = cfg.localWorkspaces?.[h.workspace_id];
2189
+ const localMode = local?.cwd ? (local.mode ?? modeForTools?.(local.allowedTools) ?? "run") : null;
2190
+ if (local?.cwd && localMode !== "ask" && localizeCommand) {
2191
+ warmUp({
2192
+ poolKey: `warm::${h.workspace_id}::${agent.name}::local`,
2193
+ agent: pinnedAgent(cfg, { ...agent, command: localizeCommand(agent.command, local.allowedTools), cwd: local.cwd }),
2194
+ env: agentEnv(cfg).env,
2195
+ helpers,
2196
+ log,
2197
+ cap: cfg.maxRunners,
2198
+ });
2199
+ }
1915
2200
  }
1916
2201
  }
1917
2202
  const queue = [];
@@ -1934,30 +2219,56 @@ async function dispatchWorkInner(cfg, work, warmHints) {
1934
2219
  }
1935
2220
  }
1936
2221
  volunteeredRetries.push(...held);
2222
+ // ONE RUN PER CONVERSATION AT A TIME: a follow-up typed while the previous turn
2223
+ // is still running waits for it (the UI promises "runs after the current one
2224
+ // finishes"). Without this, the second task dispatched on the next pull, found
2225
+ // the thread's runner busy, and fell to a cold one-shot with no history.
2226
+ const busyThreads = new Set(inFlightThreadKeys.values());
2227
+ const threadKeyOf = (t) => t.thread_root_id ?? t.id;
1937
2228
  for (const t of work) {
1938
2229
  if ((t.assigned_to || "").toLowerCase() === "goal") continue;
1939
2230
  if (inFlight.has(t.id) || givenUp.has(t.id)) continue;
2231
+ // A task the server pre-claimed FOR US in this pull (preclaim.ts). Anything we
2232
+ // cannot run right now goes back at once (and the sweep below catches every
2233
+ // other way of not starting it), so it never sits claimed and idle.
2234
+ const preclaimed = isPreclaimed(t);
2235
+ if (!preclaimed && t.status !== "open") continue;
2236
+ if (busyThreads.has(threadKeyOf(t))) { if (preclaimed) giveBack(cfg, t, "its conversation already has a run in flight"); continue; }
1940
2237
  if ((attempts.get(t.id) ?? 0) >= cfg.maxAttempts) continue;
1941
2238
  // Round two: no configured agent for "Chef" + a support task addressed to Chef +
1942
2239
  // chef-persona.md shipped next to this file = a Chef synthesized from this
1943
2240
  // Bridge's own Claude (bridge/chef.mjs). Configured agents always win.
1944
2241
  const agent = agentFor(cfg, t.assigned_to) ?? resolveAgentForTask(cfg.agents, t, cfg);
1945
- if (!agent) continue;
1946
- if (agentHeld(agent)) continue;
1947
- if (!allowedByPolicy(cfg, agent, t)) continue;
1948
- let policy;
1949
- try { policy = await resolveDelegation(cfg, t.id); } catch { continue; }
1950
- if (policy.decision !== "run") continue;
2242
+ if (!agent) { if (preclaimed) giveBack(cfg, t, "no agent here runs it"); continue; }
2243
+ if (agentHeld(agent)) { if (preclaimed) giveBack(cfg, t, `${agent.name} is in a plan hold`); continue; }
2244
+ if (!allowedByPolicy(cfg, agent, t)) { if (preclaimed) giveBack(cfg, t, "this Bridge's approval policy"); continue; }
2245
+ // The server resolved the delegation policy inside the pull; only a task it
2246
+ // did not decide (older server, a transient error) costs a round trip.
2247
+ let policy = t.delegation && typeof t.delegation === "object" && t.delegation.decision ? t.delegation : null;
2248
+ if (!policy) { try { policy = await resolveDelegation(cfg, t.id); } catch { continue; } }
2249
+ if (policy.decision !== "run") { if (preclaimed) giveBack(cfg, t, `delegation says ${policy.decision}`); continue; }
1951
2250
  queue.push({ ws: { id: t.workspace_id, name: t.workspace_name ?? "workspace" }, task: t, agent });
1952
2251
  }
1953
2252
  queue.sort((a, b) => new Date(a.task.created_at ?? 0) - new Date(b.task.created_at ?? 0));
2253
+ // Two turns of one conversation in the same snapshot: the older one runs now,
2254
+ // the newer waits for the next pull (it will be re-listed as open).
2255
+ const seenThreads = new Set();
2256
+ const serialized = queue.filter((item) => {
2257
+ const k = threadKeyOf(item.task);
2258
+ if (seenThreads.has(k)) return false;
2259
+ seenThreads.add(k);
2260
+ return true;
2261
+ });
1954
2262
  const slots = Math.max(0, cfg.maxConcurrentRuns - inFlight.size);
1955
- for (const item of queue.slice(0, slots)) {
2263
+ for (const item of serialized.slice(0, slots)) {
1956
2264
  void processTask(cfg, item.ws, item.task, item.agent).catch((e) => log(`✗ run error on "${item.task.title}": ${e.message}`));
1957
2265
  }
2266
+ // THE INVARIANT: every pre-claimed task in this snapshot either started just
2267
+ // now (processTask marks inFlight synchronously) or goes back to open.
2268
+ for (const t of work) if (isPreclaimed(t) && !inFlight.has(t.id)) giveBack(cfg, t, "not started in this pass");
1958
2269
  }
1959
2270
 
1960
- async function pollOnce(cfg, onlyWorkspaceIds = null) {
2271
+ async function pollOnce(cfg, onlyWorkspaceIds = null, onlyTaskId = null) {
1961
2272
  let workspaces = await listWorkspaces(cfg);
1962
2273
  // HOT-SCOPED SWEEP (chat feel): a full sweep across N workspaces costs N HTTP
1963
2274
  // round-trips — 12 workspaces ≈ 15s, which WAS the reply latency users felt.
@@ -2002,6 +2313,7 @@ async function pollOnce(cfg, onlyWorkspaceIds = null) {
2002
2313
  continue;
2003
2314
  }
2004
2315
  for (const t of tasks) {
2316
+ if (onlyTaskId && t.id !== onlyTaskId) continue;
2005
2317
  if (inFlight.has(t.id) || givenUp.has(t.id)) continue;
2006
2318
  if ((attempts.get(t.id) ?? 0) >= cfg.maxAttempts) continue;
2007
2319
 
@@ -2080,8 +2392,82 @@ async function pollOnce(cfg, onlyWorkspaceIds = null) {
2080
2392
  }
2081
2393
  }
2082
2394
 
2395
+ /**
2396
+ * CLOUD ONE-SHOT (0102): a sandbox runs this Bridge for one sweep. Claim what is
2397
+ * claimable (or just ONLY_TASK), wait for the runs to finish, exit. Exit codes:
2398
+ * 0 ran (or nothing to do), 3 the named task was not claimable here, 4 a run
2399
+ * outlived the ceiling and was released back to the board.
2400
+ */
2401
+ async function runOnceAndExit(cfg) {
2402
+ const t0 = Date.now();
2403
+ log(`one-shot: ${ONLY_TASK ? `task ${ONLY_TASK.slice(0, 8)}` : "everything claimable"}, then exit`);
2404
+ await pollOnce(cfg, null, ONLY_TASK);
2405
+ if (inFlight.size === 0) {
2406
+ log(ONLY_TASK ? "one-shot: that task was not claimable here (gone, claimed elsewhere, or not addressed to an agent in this box)" : "one-shot: nothing to run");
2407
+ process.exit(ONLY_TASK ? 3 : 0);
2408
+ }
2409
+ const started = inFlight.size;
2410
+ const deadline = t0 + (cfg.taskTimeoutSeconds + 120) * 1000 * Math.max(1, cfg.maxAttempts);
2411
+ while (inFlight.size > 0 && Date.now() < deadline) await new Promise((r) => setTimeout(r, 500));
2412
+ if (inFlight.size > 0) {
2413
+ await releaseInFlight(cfg, "one-shot deadline").catch(() => {});
2414
+ try { killAllRunners?.(); killCodexServer?.(); } catch { /* exiting */ }
2415
+ log(`one-shot: ${inFlight.size} run(s) outlived the ceiling and were released`);
2416
+ process.exit(4);
2417
+ }
2418
+ try { killAllRunners?.(); killCodexServer?.(); } catch { /* exiting */ }
2419
+ log(`one-shot: ${started} run(s) finished in ${Math.round((Date.now() - t0) / 1000)}s`);
2420
+ process.exit(0);
2421
+ }
2422
+
2083
2423
  /** How often a RUNNING bridge re-checks the deploy manifest ("app updated → I update"). */
2084
2424
  const UPDATE_CHECK_MS = 6 * 60 * 60 * 1000;
2425
+ /** How often to ask the CLI whether it is still signed in (0100). */
2426
+ const AUTH_CHECK_MS = 10 * 60 * 1000;
2427
+ let lastAuthCheck = 0;
2428
+
2429
+ /**
2430
+ * Ask claude whether it is signed in (`claude auth status` prints JSON with
2431
+ * `loggedIn`). Only claude has this today; other vendors are learned from their
2432
+ * failures. Records the state (plan.mjs noteAuth) so the heartbeat advertises it,
2433
+ * and logs once per change. Never throws; an unreadable answer changes nothing.
2434
+ */
2435
+ async function checkClaudeAuth(cfg) {
2436
+ const agent = (cfg?.agents ?? []).find((a) => a && a.enabled !== false && isClaudeCommand && isClaudeCommand(a.command));
2437
+ if (!agent) return null;
2438
+ const cmd = agent.command[0];
2439
+ const env = typeof agentEnv === "function" ? agentEnv(cfg).env : process.env;
2440
+ const out = await new Promise((resolve) => {
2441
+ let text = "";
2442
+ let done = false;
2443
+ const finish = (v) => { if (!done) { done = true; resolve(v); } };
2444
+ try {
2445
+ const child = spawn(cmd, ["auth", "status"], { env, stdio: ["ignore", "pipe", "pipe"], windowsHide: true });
2446
+ child.stdout.on("data", (d) => { text += d; });
2447
+ child.stderr.on("data", (d) => { text += d; });
2448
+ child.on("error", () => finish(null));
2449
+ child.on("close", () => finish(text));
2450
+ setTimeout(() => { try { child.kill(); } catch { /* gone */ } finish(null); }, 15_000).unref();
2451
+ } catch { finish(null); }
2452
+ });
2453
+ if (out === null) return null;
2454
+ let loggedIn = null;
2455
+ const brace = out.indexOf("{");
2456
+ if (brace >= 0) { try { const j = JSON.parse(out.slice(brace)); if (typeof j.loggedIn === "boolean") loggedIn = j.loggedIn; } catch { /* not JSON */ } }
2457
+ if (loggedIn === null && isSignedOutError(out)) loggedIn = false;
2458
+ if (loggedIn === null) return null;
2459
+ const changed = noteAuth("claude", loggedIn);
2460
+ if (changed && !loggedIn) log("! Claude is signed out on this machine → open a terminal, run `claude`, and sign in. This Bridge re-checks every 10 minutes.");
2461
+ if (changed && loggedIn && lastAuthCheck > 0) {
2462
+ log("✓ Claude is signed in again — Claude work resumes here.");
2463
+ // Forget the give-ups: the server re-opens the tasks that died on the sign-out
2464
+ // (heartbeat, reopenAuthAbandoned), and this Bridge must be willing to run them.
2465
+ givenUp.clear();
2466
+ attempts.clear();
2467
+ saveRunState();
2468
+ }
2469
+ return loggedIn;
2470
+ }
2085
2471
 
2086
2472
  /**
2087
2473
  * WHO OWNS THIS INSTALL'S VERSION.
@@ -2101,7 +2487,11 @@ const UPDATE_CHECK_MS = 6 * 60 * 60 * 1000;
2101
2487
  * behavior: verify, back up, replace, re-exec.
2102
2488
  */
2103
2489
  function updateChannel() {
2104
- if (IS_DESKTOP) return "app";
2490
+ // The desktop app runs the SEEDED copy in its data dir (COOKBOOK_RUNTIME_WRITABLE=1,
2491
+ // 2026-09-09): outside the signed bundle, so self-update is safe there and the
2492
+ // supervisor restarts us on the new code (COOKBOOK_SERVICE=1). Only the bundled
2493
+ // fallback copy stays on the "app" channel.
2494
+ if (IS_DESKTOP) return process.env.COOKBOOK_RUNTIME_WRITABLE === "1" ? "self" : "app";
2105
2495
  if (HERE.includes(`${path.sep}node_modules${path.sep}`)) return "npm";
2106
2496
  if (fs.existsSync(path.join(HERE, "package.json"))) return "npm";
2107
2497
  return "self";
@@ -2116,6 +2506,39 @@ let updateNagged = false;
2116
2506
  * Check failures are non-fatal (offline is fine); a FAILED apply never breaks the
2117
2507
  * running code (verification happens before any write; originals in bridge.backup/).
2118
2508
  */
2509
+ /**
2510
+ * Install the runtime + login service and wait for it to come up. Returns true
2511
+ * when a Bridge is running under the service; false (with the reason printed) so
2512
+ * the caller can fall back to a foreground run.
2513
+ */
2514
+ async function installAsService(svc, { cookbookUrl, cfgPath }) {
2515
+ const say = (m) => console.log(` ${m}`);
2516
+ try {
2517
+ console.log("\nInstalling the Bridge as a background service…");
2518
+ await svc.installRuntime({ cookbookUrl, log: say });
2519
+ svc.installService({ config: cfgPath, log: say });
2520
+ process.stdout.write(" Starting");
2521
+ const pid = await svc.waitForBridge(cfgPath, { timeoutMs: 30_000 });
2522
+ console.log("");
2523
+ const st = svc.serviceState({ config: cfgPath });
2524
+ if (pid) {
2525
+ console.log(`\n ✓ The Bridge is running in the background (pid ${pid}). It starts with your computer,`);
2526
+ console.log(" updates itself from cookbook.team, and this window can close.");
2527
+ } else {
2528
+ console.log(`\n ! The service is installed but the Bridge has not reported in yet. Its log: ${st.log}`);
2529
+ }
2530
+ console.log(`\n Config: ${cfgPath}`);
2531
+ console.log(` Log: ${st.log}`);
2532
+ console.log(` Check: ${cli("doctor")}`);
2533
+ console.log(` Restart: ${cli("restart")}`);
2534
+ console.log(` Remove: ${cli("uninstall")}\n`);
2535
+ return true;
2536
+ } catch (e) {
2537
+ console.log(`\n ! Could not install the service: ${e.message}`);
2538
+ return false;
2539
+ }
2540
+ }
2541
+
2119
2542
  async function selfUpdate(cfg, { reexec }) {
2120
2543
  let check;
2121
2544
  try {
@@ -2140,6 +2563,10 @@ async function selfUpdate(cfg, { reexec }) {
2140
2563
  try {
2141
2564
  const replaced = await applyUpdate(cfg, HERE, check);
2142
2565
  log(`⬆ Bridge self-updated to deploy ${check.version} (${replaced.length} file(s), hash-verified; previous in bridge.backup/${check.version}/).`);
2566
+ if (reexec && process.env.COOKBOOK_SERVICE === "1") {
2567
+ log("↻ restarting on the new code (the service brings it back)…");
2568
+ process.exit(0);
2569
+ }
2143
2570
  if (reexec) {
2144
2571
  log("↻ restarting on the new code…");
2145
2572
  const { spawn } = await import("node:child_process");
@@ -2157,10 +2584,14 @@ async function main() {
2157
2584
  const cfg = loadConfig();
2158
2585
  log(`Cookbook Bridge started · ${cfg.cookbookUrl}`);
2159
2586
  log(`Managing: ${cfg.agents.map((a) => a.name).join(", ") || "(no agents enabled!)"} · polling every ${cfg.pollSeconds}s`);
2587
+ // Is the CLI actually able to run? (0100) The answer rides the first heartbeat.
2588
+ if (!ONCE) await checkClaudeAuth(cfg).catch(() => null);
2589
+ lastAuthCheck = Date.now();
2160
2590
 
2161
2591
  // "When the app updates, so does the Bridge": check the deploy manifest now, then
2162
- // every 6h while running. Set "autoUpdate": false in config to pin.
2163
- await selfUpdate(cfg, { reexec: true });
2592
+ // every 6h while running. Set "autoUpdate": false in config to pin. A cloud
2593
+ // one-shot already runs the deploy's own files.
2594
+ if (!ONCE) await selfUpdate(cfg, { reexec: true });
2164
2595
  let lastUpdateCheck = Date.now();
2165
2596
 
2166
2597
  // Loud warning if the default agent (the one that runs "any"-assigned tasks) isn't
@@ -2237,8 +2668,9 @@ async function main() {
2237
2668
  // connect a folder, doctor, restart). Started BEFORE the token check on purpose:
2238
2669
  // the whole point of the connect UI is to FIX a broken/missing connection, so its
2239
2670
  // control plane must be up even when the Cookbook token is bad. Additive — a bind
2240
- // failure never stops the Bridge from doing its real job.
2241
- try {
2671
+ // failure never stops the Bridge from doing its real job. A cloud one-shot has
2672
+ // no app to talk to and no port to offer.
2673
+ if (!ONCE) try {
2242
2674
  let version = "dev";
2243
2675
  try {
2244
2676
  const { createHash } = await import("node:crypto");
@@ -2294,7 +2726,7 @@ async function main() {
2294
2726
  // TEAM CONNECTORS (0068): a tool connected once in a workspace reaches every
2295
2727
  // member's CLIs, minus secrets (members export secret_env vars themselves).
2296
2728
  // Opt out with "syncConnectors": false. Startup-only; every write is .bak'd.
2297
- if (cfg.syncConnectors !== false) {
2729
+ if (!ONCE && cfg.syncConnectors !== false) {
2298
2730
  void (async () => {
2299
2731
  try {
2300
2732
  const { listTeamConnectors } = await import("./cookbook.mjs");
@@ -2317,7 +2749,7 @@ async function main() {
2317
2749
  }
2318
2750
  })();
2319
2751
  }
2320
- {
2752
+ if (!ONCE) {
2321
2753
  const mode = hostingMode(cfg);
2322
2754
  if (mode === "always") log(`⌂ Hosting is ON: an agent you invite can run granted checks on this machine. You'll see every step; \`${cli("host --off")}\` closes the door.`);
2323
2755
  else if (mode === "grants") log(`⌂ Hosting: grants you approve in Cookbook run here (every change still waits for your click). \`${cli("host --off")}\` refuses all.`);
@@ -2333,10 +2765,22 @@ async function main() {
2333
2765
  log(` Fix it in the app (Connect your agents), or run \`${cli("connect")}\`. The control API stays up so you can.`);
2334
2766
  } else {
2335
2767
  console.error(`\nCouldn't connect to Cookbook: ${e.message}`);
2768
+ // Under a login service the supervisor restarts us at once; a dead token
2769
+ // would then hit the server every 15 seconds forever. Say the fix, then
2770
+ // wait before exiting so the loop is gentle (service.mjs, 2026-09-09).
2771
+ if (process.env.COOKBOOK_SERVICE === "1") {
2772
+ console.error(` Fix: ${cli("connect")} (reconnects; the service picks the new token up on its own). Retrying in 5 minutes.`);
2773
+ await new Promise((r) => setTimeout(r, 5 * 60 * 1000));
2774
+ }
2336
2775
  process.exit(1);
2337
2776
  }
2338
2777
  }
2339
2778
 
2779
+ if (ONCE) {
2780
+ await runOnceAndExit(cfg);
2781
+ return;
2782
+ }
2783
+
2340
2784
  let lastFullSweepAt = 0;
2341
2785
  let fastPath = true; // one-call dispatch until the server says it can't
2342
2786
  void (async () => {
@@ -2402,11 +2846,19 @@ async function main() {
2402
2846
  } else {
2403
2847
  log("✗ Cookbook has rejected this token 5 polls in a row — it was likely revoked (a new login replaces old tokens) or expired.");
2404
2848
  log(` Fix: ${cli("connect")} (reconnects and starts the Bridge)`);
2849
+ if (process.env.COOKBOOK_SERVICE === "1") {
2850
+ log(" Running as a service: retrying in 5 minutes.");
2851
+ await new Promise((r) => setTimeout(r, 5 * 60 * 1000));
2852
+ }
2405
2853
  process.exit(1);
2406
2854
  }
2407
2855
  }
2408
2856
  }
2409
2857
  }
2858
+ if (Date.now() - lastAuthCheck > AUTH_CHECK_MS) {
2859
+ lastAuthCheck = Date.now();
2860
+ void checkClaudeAuth(cfg).catch(() => null);
2861
+ }
2410
2862
  if (Date.now() - lastUpdateCheck > UPDATE_CHECK_MS) {
2411
2863
  lastUpdateCheck = Date.now();
2412
2864
  await selfUpdate(cfg, { reexec: true });
@@ -2414,7 +2866,7 @@ async function main() {
2414
2866
  // HOT MODE (chat feel): while a conversation is active (a run started or
2415
2867
  // finished in the last 3 minutes), poll every second so a reply dispatches
2416
2868
  // near-instantly; decay back to the configured cadence when the room quiets.
2417
- if (cfg.persistentThreads) { reapIdleRunners(log); reapCodexServer(log); }
2869
+ if (cfg.persistentThreads) { reapIdleRunners(log, cfg.runnerIdleMinutes * 60_000, cfg.maxRunners); reapCodexServer(log); }
2418
2870
  const hot = Date.now() - lastHotAt < HOT_WINDOW_MS;
2419
2871
  await new Promise((r) => setTimeout(r, fastPath || hot ? 1000 : cfg.pollSeconds * 1000));
2420
2872
  }
@@ -2529,6 +2981,19 @@ async function doctorReport(args) {
2529
2981
  }
2530
2982
  }
2531
2983
 
2984
+ // 2a½. The login service (2026-09-09): installed? running?
2985
+ try {
2986
+ const svc = await import("./service.mjs");
2987
+ const st = svc.serviceState({ config: cfgPath });
2988
+ if (!st.kind) ok("Login service: not available on this platform (run the Bridge in a terminal)");
2989
+ else if (st.installed && st.pid) ok(`Login service: installed (${st.definition}) and running (pid ${st.pid})`);
2990
+ else if (st.installed) warn(`Login service: installed (${st.definition}) but no Bridge is reporting in`, `look at ${st.log}, or \`${cli("restart")}\``);
2991
+ else if (IS_DESKTOP) ok("Login service: Cookbook Desktop supervises this Bridge (starts at login from the tray)");
2992
+ else warn("Login service: not installed, so the Bridge stops when this window closes", `\`${cli("install")}\` installs it and starts it now`);
2993
+ } catch (e) {
2994
+ ok(`Login service: could not check (${e.message})`);
2995
+ }
2996
+
2532
2997
  // 2b. Another Bridge on this machine? Two on one config fight over the same token
2533
2998
  // (a `connect` revokes the other's); one on a different config is the classic
2534
2999
  // "I connected but a stale Bridge is still running" trap.
@@ -2685,6 +3150,9 @@ async function doctorReport(args) {
2685
3150
  }
2686
3151
 
2687
3152
  if (isClaudeCommand && isClaudeCommand(agent.command)) {
3153
+ const signedIn = await checkClaudeAuth({ ...cfg, agents: [agent] }).catch(() => null);
3154
+ if (signedIn === false) bad(`${agent.name}: claude is SIGNED OUT on this machine`, "open a terminal, run `claude`, and sign in (or `claude auth login`)");
3155
+ else if (signedIn === true) ok(`${agent.name}: claude is signed in`);
2688
3156
  if (agent.token) ok(`${agent.name}: runs carry their own Cookbook connection (per-agent token) — identity is this Bridge's member`);
2689
3157
  else warn(`${agent.name}: no per-agent token — runs use the claude CLI's OWN Cookbook login, which may be a different account and inherits stale claude.ai connectors`,
2690
3158
  `run \`${cli("connect")}\` (mints a token for this agent) or add "token" to this agent in ${cfgPath}`);
@@ -2706,14 +3174,14 @@ async function doctorReport(args) {
2706
3174
  continue;
2707
3175
  }
2708
3176
  if (agent.runner === "robot") {
2709
- const r = await spawnAgent(agent, "", 15, agentEnv(cfg).env);
3177
+ const r = await spawnAgent(agent, "", 15, agentEnv(cfg).env, undefined, { cfg });
2710
3178
  if (r.code === 0 && String(r.out).trim().endsWith("ok")) ok(`${agent.name}: robot agent responds (probe ok)`);
2711
3179
  else warn(`${agent.name}: robot agent probe inconclusive`, String(r.err || r.out).slice(0, 160));
2712
3180
  continue;
2713
3181
  }
2714
3182
 
2715
3183
  try {
2716
- const r = await spawnAgent(agent, "Reply with the single word: ok", 30, agentEnv(cfg).env);
3184
+ const r = await spawnAgent(agent, "Reply with the single word: ok", 30, agentEnv(cfg).env, undefined, { cfg });
2717
3185
  const text = `${r.out || ""}\n${r.err || ""}`.toLowerCase();
2718
3186
  if (text.includes("not logged in") || text.includes("/login") || text.includes("please log in")) {
2719
3187
  bad(`${agent.name}: CLI is NOT logged in`, `run \`${cmd} auth login\` (persists; setup-token does not)`);
@@ -2749,7 +3217,13 @@ async function doctorReport(args) {
2749
3217
  // undefined, the guard read false, and EVERY command (connect, doctor, the Bridge
2750
3218
  // itself) exited 0 in silence on Node 18/20/22.x. The argv[1] comparison is the
2751
3219
  // portable fallback. Exported for tests.
2752
- export function isMainModule(meta = import.meta, argv = process.argv) {
3220
+ export function isMainModule(meta = import.meta, argv = process.argv, launched = globalThis.__cookbookLauncher) {
3221
+ // Under the compiled launcher (launcher.mjs) the entry module is the launcher and
3222
+ // this file is imported, so meta.main is false here. The launcher sets argv the way
3223
+ // node hands it over ([exec, script, ...args]) and marks itself; trust argv then.
3224
+ if (launched) {
3225
+ try { return !!argv[1] && path.resolve(argv[1]) === fileURLToPath(meta.url); } catch { return false; }
3226
+ }
2753
3227
  if (meta.main === true) return true;
2754
3228
  if (meta.main === false) return false;
2755
3229
  try {
@@ -2853,13 +3327,24 @@ if (!IS_MAIN) {
2853
3327
  .then(async (m) => {
2854
3328
  const args = process.argv.slice(3);
2855
3329
  const noRun = args.includes("--no-run");
2856
- const r = await m.connectAgents(args.filter((a) => a !== "--no-run"), { willRun: !noRun });
3330
+ // THE SERVICE (2026-09-09): after the approval the Bridge is installed as a
3331
+ // login service and started, so the window can close and it survives reboots.
3332
+ // `--no-service` keeps the foreground run; the desktop app supervises its own.
3333
+ const svc = await import("./service.mjs");
3334
+ const wantService = !noRun && !args.includes("--no-service") && !IS_DESKTOP && !!svc.serviceKind();
3335
+ const passArgs = args.filter((a) => a !== "--no-run" && a !== "--no-service" && a !== "--no-signin").concat(args.includes("--no-signin") ? ["--no-signin"] : []);
3336
+ const r = await m.connectAgents(passArgs, { willRun: !noRun, service: wantService });
2857
3337
  if (!r || !r.ok) {
2858
3338
  // Nothing to run (no agent CLI found): the doctor says what is missing and how to fix it.
2859
3339
  if (r && r.reason === "no-agents") await runDoctor(["--config", r.cfgPath]);
2860
3340
  return;
2861
3341
  }
2862
3342
  if (noRun || !r.startBridge) return;
3343
+ if (wantService) {
3344
+ const done = await installAsService(svc, { cookbookUrl: r.baseUrl, cfgPath: r.cfgPath });
3345
+ if (done) return;
3346
+ console.log("Falling back to running the Bridge here.\n");
3347
+ }
2863
3348
  console.log("Connected. Running the Bridge now; leave this window open. Ctrl-C stops it.\n");
2864
3349
  process.argv = [process.argv[0], process.argv[1], r.cfgPath];
2865
3350
  await main();
@@ -2896,6 +3381,33 @@ if (!IS_MAIN) {
2896
3381
  console.error(e.message);
2897
3382
  process.exit(1);
2898
3383
  });
3384
+ } else if (sub === "install") {
3385
+ // Runtime + login service + start, against an existing config (connect does this
3386
+ // for you; `install` is for a machine that already has a config).
3387
+ (async () => {
3388
+ const svc = await import("./service.mjs");
3389
+ const cfgPath = configPathFromArgs(process.argv.slice(3));
3390
+ let cookbookUrl = "";
3391
+ try { cookbookUrl = String(JSON.parse(fs.readFileSync(cfgPath, "utf8")).cookbookUrl || "").replace(/\/$/, ""); } catch { /* below */ }
3392
+ if (!cookbookUrl) { console.error(`No config at ${cfgPath}. Run \`${cli("connect")}\` first.`); process.exit(1); }
3393
+ const ok = await installAsService(svc, { cookbookUrl, cfgPath });
3394
+ process.exit(ok ? 0 : 1);
3395
+ })().catch((e) => { console.error(e.message); process.exit(1); });
3396
+ } else if (sub === "uninstall") {
3397
+ (async () => {
3398
+ const svc = await import("./service.mjs");
3399
+ const cfgPath = configPathFromArgs(process.argv.slice(3));
3400
+ const r = svc.uninstallService({ config: cfgPath, log: (m) => console.log(` ${m}`) });
3401
+ console.log(r.removed.length ? `Service removed (${r.removed.join(", ")}).` : "No service was installed.");
3402
+ console.log(`Your config and token are untouched at ${cfgPath}. To revoke the Bridge's access, remove it under Account > Connected apps.`);
3403
+ })().catch((e) => { console.error(e.message); process.exit(1); });
3404
+ } else if (sub === "restart") {
3405
+ (async () => {
3406
+ const svc = await import("./service.mjs");
3407
+ const cfgPath = configPathFromArgs(process.argv.slice(3));
3408
+ const ok = svc.restartService({ config: cfgPath });
3409
+ console.log(ok ? "Restarting the Bridge service." : `No running service found for ${cfgPath}. Start one with \`${cli("install")}\`.`);
3410
+ })().catch((e) => { console.error(e.message); process.exit(1); });
2899
3411
  } else if (sub === "status") {
2900
3412
  import("./device.mjs")
2901
3413
  .then((m) => m.status(process.argv.slice(3)))