@cohortapp/agent-sdk 2.18.8 → 2.18.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -527,11 +527,19 @@ const ACTIVE_PATH = join(AGENT_REPO_DIR, "state", "sessions", "active.json");
527
527
  // A marker is written to state/sessions/resume-pending/<sessionId>.json right
528
528
  // after spawn and DELETED on a clean close. Its presence after a crash/reboot
529
529
  // means "this session was mid-flight" — resetActiveSessions() reconciles them
530
- // (re-dispatching `claude --print --session-id <claudeSessionId> <prompt>`,
531
- // the SAME continuation mechanism responder.mjs uses — NOT `--resume`) within a
532
- // freshness window, instead of blindly wiping the slate. 3 strikes → the queue
533
- // item is marked status:blocked. This is the "never brick / never drop work"
534
- // recovery path for the dispatcher.
530
+ // (re-dispatching `claude --print --resume <claudeSessionId> <prompt>`, the
531
+ // same continuation mechanism responder.mjs uses) within a freshness window,
532
+ // instead of blindly wiping the slate. 3 strikes → the queue item is marked
533
+ // status:blocked. This is the "never brick / never drop work" recovery path
534
+ // for the dispatcher.
535
+ //
536
+ // ~~"re-dispatching `claude --print --session-id <claudeSessionId> <prompt>`
537
+ // … NOT `--resume`"~~ — struck 2026-09-25 with the flag itself. A marker's
538
+ // claudeSessionId by construction HAS a transcript, and `--session-id` is
539
+ // REFUSED for such an id (`Error: Session ID <uuid> is already in use.`,
540
+ // exit 1, no model call), so every reconcile exited 1 rather than recovering
541
+ // anything. `--resume` is the continuation. See lib/runtime/adapter.mjs
542
+ // #sessionArgs for the measurement.
535
543
  const RESUME_PENDING_DIR = join(AGENT_REPO_DIR, "state", "sessions", "resume-pending");
536
544
  // Sessions whose marker is older than this are too stale to resume meaningfully
537
545
  // (the work context has moved on); reconcile blocks them rather than re-running.
@@ -634,10 +642,14 @@ function markItemBlocked(marker, reason) {
634
642
  * WS4 — instead of *also* blindly discarding any work that was mid-flight when
635
643
  * the box rebooted/lost power, when `opts.reconcile` is set we walk
636
644
  * state/sessions/resume-pending/ and, for each marker still inside the
637
- * freshness window, re-spawn `claude --print --session-id <claudeSessionId>
638
- * <prompt>` (the responder's continuation pattern — NOT `--resume`) so the work
639
- * continues where it left off. After RESUME_MAX_ATTEMPTS (3) the item is marked
640
- * status:blocked with a reason rather than retried forever.
645
+ * freshness window, re-spawn `claude --print --resume <claudeSessionId>
646
+ * <prompt>` (the responder's continuation pattern) so the work continues where
647
+ * it left off. After RESUME_MAX_ATTEMPTS (3) the item is marked status:blocked
648
+ * with a reason rather than retried forever.
649
+ *
650
+ * ~~"`--session-id <claudeSessionId> <prompt>` … NOT `--resume`"~~ — struck
651
+ * 2026-09-25: the CLI refuses `--session-id` for an id that already has a
652
+ * transcript, which every marker's id does.
641
653
  *
642
654
  * Graceful shutdown passes no opts (just clears active.json); only startup
643
655
  * reconciles, so we never re-spawn work we're deliberately stopping.
@@ -736,8 +748,11 @@ export function reconcileResumePending(opts = {}) {
736
748
  }
737
749
 
738
750
  // No original prompt persisted → there's nothing to re-spawn with (the
739
- // resume re-issues `--session-id <id> <prompt>`, NOT `--resume`). A marker
740
- // without a prompt is either a legacy marker (pre-H1) or one whose write
751
+ // resume re-issues `--resume <id> <prompt>`: the flag continues the
752
+ // transcript, the prompt is what the continuation is FOR, and a bare
753
+ // `--resume` with no prompt is the do-nothing exit 0 that H1 guards
754
+ // against). A marker without a prompt is either a legacy marker (pre-H1)
755
+ // or one whose write
741
756
  // dropped the field; we can't truly resume it, so retire it rather than
742
757
  // spawn a no-op that the close handler would mistake for "done". (H1)
743
758
  if (typeof marker.prompt !== "string" || !marker.prompt) {
@@ -894,6 +909,11 @@ export function buildResumeSpawn(o = {}) {
894
909
  model: flag,
895
910
  prompt: marker.prompt,
896
911
  sessionId: marker.claudeSessionId,
912
+ // Crash recovery by definition continues a transcript the dead session
913
+ // already wrote, so this is `--resume`. It was `--session-id`, which the
914
+ // CLI refuses for an id it has seen — so recovery has been exiting 1 in a
915
+ // tenth of a second rather than recovering anything.
916
+ resumeSession: true,
897
917
  permissions: Array.isArray(o.permissionArgs) ? o.permissionArgs : [],
898
918
  mcp: {},
899
919
  knobs: t,
@@ -922,19 +942,30 @@ export function buildResumeSpawn(o = {}) {
922
942
  }
923
943
 
924
944
  /**
925
- * Real resume spawner. Mirrors responder.mjs's blessed session-continuation
926
- * pattern (see runClaudeCLI ~L116-122 + generateResponse ~L461-465): a
927
- * continuation re-spawns `claude --print --session-id <sessionId> <prompt>`
928
- * with the SAME model/flags the original used — it does NOT use `--resume`.
945
+ * Real resume spawner. Mirrors responder.mjs's session-continuation pattern: a
946
+ * continuation re-spawns `claude --print --resume <sessionId> <prompt>` with
947
+ * the SAME model/flags the original used. The flag rehydrates the transcript;
948
+ * the prompt is what the rehydrated session is asked to carry on with, which
949
+ * is why a resume here is never the bare `--resume` with no prompt.
950
+ *
951
+ * ~~"a continuation re-spawns `claude --print --session-id <sessionId>
952
+ * <prompt>` … it does NOT use `--resume`", and the "Why NOT `--resume` (the
953
+ * previous bug)" paragraph under it~~ — STRUCK 2026-09-25, in the same change
954
+ * that made `resumeSession: true` the flag this function emits. Both halves
955
+ * were false and the file still said them:
929
956
  *
930
- * Why NOT `--resume` (the previous bug): `--resume <id>` with no prompt and
931
- * stdin ignored either (a) exits 0 having done nothing — the close handler
932
- * then clears the marker and the interrupted work is silently lost — or (b)
933
- * errors, force-`blocked`ing every in-flight item after a reboot. Pre-minting
934
- * a stable session id and RE-SPAWNING against it with the original prompt is
935
- * the actual resume mechanism in `--print` mode (per the b1 flag-verification
936
- * report cited in responder.mjs). The CLI rehydrates the session-id's history
937
- * and the fresh prompt continues the work.
957
+ * · `--session-id <id>` STARTS a session under a chosen id. The CLI refuses
958
+ * it for an id that already has a transcript — `Error: Session ID <uuid>
959
+ * is already in use.`, exit 1, ~0.1s, no model call — and a resume marker's
960
+ * id always has one. So crash recovery had never recovered anything.
961
+ * · The bare-`--resume` failure the struck paragraph described (exit 0 having
962
+ * done nothing) is a resume with NO PROMPT. This lane always passes the
963
+ * marker's prompt, so it was never the shape being warned about.
964
+ *
965
+ * Left standing, that paragraph is how the next author reinstates the flag:
966
+ * it reads as a rationale rather than as the belief that cost four replies on
967
+ * this seat, 09-22..24. lib/runtime/adapter.mjs#sessionArgs carries the
968
+ * reproduction.
938
969
  *
939
970
  * Audit F11: the argv and env are built through the SAME resolveSpawnTarget /
940
971
  * runtime-adapter path as spawnSession (buildResumeSpawn), so a retargeted
@@ -1652,9 +1683,13 @@ function spawnSession(entry) {
1652
1683
  // session in a live thread started COLD — the agent re-read the room, re-did
1653
1684
  // the orientation work, and answered a follow-up as if it were an opening
1654
1685
  // (design §3 R10). The router the responder already consults keys the
1655
- // conversation; a follow-up inside the TTL resumes the same session id, which
1656
- // in this daemon IS continuation (`--session-id <id> <prompt>`, never
1657
- // `--resume` — see the resume-pending notes above).
1686
+ // conversation; a follow-up inside the TTL resumes the same session id.
1687
+ //
1688
+ // ~~"which in this daemon IS continuation (`--session-id <id> <prompt>`,
1689
+ // never `--resume`)"~~ — struck: it is not, and never was. `--session-id`
1690
+ // STARTS a session under a chosen id and the CLI refuses it for an id that
1691
+ // already has a transcript. Continuation is `--resume`, which is what
1692
+ // `resumeSession` below now selects.
1658
1693
  //
1659
1694
  // Inbox only. Backlog work is not a conversation, and keying it would put
1660
1695
  // unrelated items in one session.
@@ -1725,6 +1760,13 @@ function spawnSession(entry) {
1725
1760
  model: engineShape ? engineShape.model : effectiveModelFlag,
1726
1761
  prompt,
1727
1762
  sessionId: claudeSessionId,
1763
+ // `--resume <id>` when this id names a transcript that already exists (the
1764
+ // router's RESUME decision), `--session-id <id>` when we just minted it.
1765
+ // Passing a used id to `--session-id` is refused outright by the CLI
1766
+ // ("Session ID … is already in use", exit 1 before any model call), which
1767
+ // is what made every routed continuation die — see sessionArgs() in
1768
+ // lib/runtime/adapter.mjs.
1769
+ resumeSession: resumedSessionId != null,
1728
1770
  permissions: sessionPermissionArgs({ source: "dispatcher", priority: classResult?.priority }),
1729
1771
  mcp: {},
1730
1772
  knobs: engineShape ? engineShape.knobs : target,
@@ -85,6 +85,27 @@ export function _resetInFlightForTests() {
85
85
  inFlightKeys.clear();
86
86
  }
87
87
 
88
+ /**
89
+ * Does this CLI failure mean "the id you asked to START is already taken"?
90
+ *
91
+ * The CLI's exact text is `Error: Session ID <uuid> is already in use.` on
92
+ * stderr with exit 1, emitted in ~0.1s before any model call. It is a PURE
93
+ * addressing fault: the same prompt with a fresh id succeeds. So it is the one
94
+ * spawn failure a reply path may retry blind, and the caller that catches it
95
+ * owes exactly one retry with a new id rather than a dropped reply.
96
+ *
97
+ * Matched loosely on purpose — the caller sees this wrapped as
98
+ * `claude CLI exited 1: Error: Session ID … is already in use.` — but anchored
99
+ * on both halves so an unrelated "in use" message cannot trigger a retry.
100
+ *
101
+ * @param {unknown} err an Error, or the message/stderr text
102
+ * @returns {boolean}
103
+ */
104
+ export function isSessionIdCollision(err) {
105
+ const text = err instanceof Error ? err.message : typeof err === "string" ? err : "";
106
+ return /session id\b/i.test(text) && /\bis already in use/i.test(text);
107
+ }
108
+
88
109
  /**
89
110
  * Services whose routing key is `<source>:<channel>[:<thread>]`.
90
111
  *
@@ -34,7 +34,12 @@ import {
34
34
  routerItemFromDaemonItem,
35
35
  claimSession,
36
36
  releaseSession,
37
+ isSessionIdCollision,
37
38
  } from "./lib/session-router.mjs";
39
+ // The seat's count of replies it owed and did not send (lib/daemon/reply-debt).
40
+ // Every silent end of this path bumps one of these names so the number rides
41
+ // the presence beat instead of dying in a log line nobody tails.
42
+ import { REPLY_DEBT_COUNTERS } from "../../lib/daemon/reply-debt.mjs";
38
43
  // Permission scoping (security CRITICAL / audit H1). sessionPermissionArgs()
39
44
  // preserves the historical "--dangerously-skip-permissions" by default and only
40
45
  // scopes tools when the operator opts in (MAESTRO_SCOPED_PERMISSIONS=1), so the
@@ -284,12 +289,24 @@ function recordResponderCost({ json, model, durationMs, exitCode, unmeasuredReas
284
289
  // so the caller can surface the failure.
285
290
  //
286
291
  // Session-router wire-up (b2-b4, cycle 474): when the caller supplies a
287
- // `sessionId` (pre-minted UUID) and a `router` + `routingKey`, the spawn
288
- // adds `--session-id <uuid> --output-format json`, parses the one-line JSON
289
- // stdout into {session_id, result, is_error}, calls router.touch on success
290
- // and router.recordExit on close. Per b1 flag-verification report, NEVER
291
- // combine `--resume` with `--session-id` (not needed: pre-minting + reusing
292
- // the same UUID across spawns is the resume mechanism).
292
+ // `sessionId` (pre-minted UUID) and a `router` + `routingKey`, the spawn adds
293
+ // the session flag for what it HOLDS — `--session-id <uuid>` for an id the
294
+ // router just minted, `--resume <uuid>` for one it decided to continue — plus
295
+ // `--output-format json`, parses the one-line JSON stdout into
296
+ // {session_id, result, is_error}, calls router.touch on success and
297
+ // router.recordExit on close. The two flags are never combined, and
298
+ // `sessionArgs()` in lib/runtime/adapter.mjs is the one place that chooses.
299
+ //
300
+ // ~~"Per b1 flag-verification report, NEVER combine `--resume` with
301
+ // `--session-id` (not needed: pre-minting + reusing the same UUID across
302
+ // spawns is the resume mechanism)"~~ — STRUCK 2026-09-25. The parenthetical
303
+ // is the false half and it is the half that costs replies: reusing a UUID is
304
+ // NOT the resume mechanism, because `--session-id` is refused once that id has
305
+ // a transcript (`Error: Session ID <uuid> is already in use.`, exit 1, before
306
+ // any model call). The router's RESUME decision returns exactly such an id, so
307
+ // this lane dropped every continued reply — four measured on this seat,
308
+ // 09-22..24. Only the first half survives: the flags remain mutually
309
+ // exclusive, which is what `resumeSession` below expresses.
293
310
  //
294
311
  // @returns {Promise<{ text: string, jsonResult: object|null, exitCode: number }>}
295
312
  /**
@@ -305,9 +322,13 @@ function recordResponderCost({ json, model, durationMs, exitCode, unmeasuredReas
305
322
  * W4-E1 (CF-44): `seat` is the seat's engine resolved once for this spawn
306
323
  * (lib/runtime/seat-engine.mjs). A claude seat adds no fields, so its spawn is
307
324
  * unchanged; an engine-cohort seat runs `cli.mjs run`, continuing --session-id.
308
- * @param {{systemPrompt:string, userPrompt:string, model:string, sessionId?:string|null, env?:object, bin?:string, seat?:{fields:object}}} o
325
+ * `resumeSession` distinguishes CONTINUING a transcript (`--resume <id>`) from
326
+ * starting one under a chosen id (`--session-id <id>`). Passing a used id to
327
+ * `--session-id` is what the CLI refuses with "Session ID … is already in use",
328
+ * and it was the session router's RESUME decision doing exactly that.
329
+ * @param {{systemPrompt:string, userPrompt:string, model:string, sessionId?:string|null, resumeSession?:boolean, env?:object, bin?:string, seat?:{fields:object}}} o
309
330
  */
310
- export function responderSpawn({ systemPrompt, userPrompt, model, sessionId = null, env = process.env, bin = CLAUDE_BIN, seat = { fields: {} } }) {
331
+ export function responderSpawn({ systemPrompt, userPrompt, model, sessionId = null, resumeSession = false, env = process.env, bin = CLAUDE_BIN, seat = { fields: {} } }) {
311
332
  return buildSpawn({
312
333
  lane: "responder",
313
334
  bin,
@@ -315,6 +336,7 @@ export function responderSpawn({ systemPrompt, userPrompt, model, sessionId = nu
315
336
  systemPrompt,
316
337
  prompt: userPrompt,
317
338
  sessionId,
339
+ resumeSession,
318
340
  permissions: sessionPermissionArgs({ source: "responder" }),
319
341
  mcp: { source: "responder" },
320
342
  env,
@@ -328,11 +350,11 @@ function runClaudeCLI(systemPrompt, userPrompt, model, opts = {}) {
328
350
  }
329
351
 
330
352
  function runSeatCLI(seat, systemPrompt, userPrompt, model, opts = {}) {
331
- const { sessionId = null, router = null, routingKey = null } = opts;
353
+ const { sessionId = null, resumeSession = false, router = null, routingKey = null } = opts;
332
354
  const startedAt = Date.now();
333
355
 
334
356
  return new Promise((resolvePromise, rejectPromise) => {
335
- const spawnSpec = responderSpawn({ systemPrompt, userPrompt, model, sessionId, seat });
357
+ const spawnSpec = responderSpawn({ systemPrompt, userPrompt, model, sessionId, resumeSession, seat });
336
358
  if (!spawnSpec.ok) {
337
359
  rejectPromise(new Error(`claude CLI unavailable: ${spawnSpec.error.message}`));
338
360
  return;
@@ -884,6 +906,7 @@ ${MESSAGE_CRAFT}`;
884
906
  const routerItem = deriveRouterItem(item);
885
907
  let key = null;
886
908
  let sessionId = null;
909
+ let resumeSession = false;
887
910
  if (routerItem) {
888
911
  try {
889
912
  const candidateKey = deriveRoutingKey(routerItem);
@@ -896,6 +919,11 @@ ${MESSAGE_CRAFT}`;
896
919
  key = candidateKey;
897
920
  if (decision.decision === "RESUME" && decision.resumeId) {
898
921
  sessionId = decision.resumeId;
922
+ // CONTINUING a transcript is `--resume`. Handing this id to
923
+ // `--session-id` is what the CLI refuses outright, and it refused
924
+ // EVERY resume this router ever decided — see sessionArgs() in
925
+ // lib/runtime/adapter.mjs.
926
+ resumeSession = true;
899
927
  } else {
900
928
  // EPHEMERAL or EPHEMERAL_REPLACE — pre-mint a fresh UUID. Reusing
901
929
  // the same key on next call (with a different sessionId) is fine;
@@ -920,17 +948,64 @@ ${MESSAGE_CRAFT}`;
920
948
  // and released in a `finally`: a throw anywhere in between must not leave the
921
949
  // room's key claimed forever, which would silently disable continuity for it.
922
950
  try {
923
- const cliResult = await (deps.runCLI || runClaudeCLI)(systemPrompt, userContent, model, {
924
- sessionId,
925
- router: key ? router : null,
926
- routingKey: key,
927
- });
951
+ const runCLI = deps.runCLI || runClaudeCLI;
952
+ let cliResult;
953
+ try {
954
+ cliResult = await runCLI(systemPrompt, userContent, model, {
955
+ sessionId,
956
+ resumeSession,
957
+ router: key ? router : null,
958
+ routingKey: key,
959
+ });
960
+ } catch (err) {
961
+ // "Session ID <uuid> is already in use" is an ADDRESSING fault, not a
962
+ // model failure: the CLI exits 1 in ~0.1s having called nothing, and the
963
+ // same prompt with a fresh id succeeds. Before `sessionArgs` this was
964
+ // every RESUME; it can still happen for an id another process on this box
965
+ // claimed between route() and spawn. Either way, dropping a person's
966
+ // reply over a name collision is indefensible — mint a new id and go
967
+ // once more. ONE retry, and it never resumes: the second attempt is
968
+ // deliberately cold, because whatever owns that transcript is not us.
969
+ if (!isSessionIdCollision(err)) throw err;
970
+ counters.bump(REPLY_DEBT_COUNTERS.sessionCollision, { key: key || "none", model: String(model || "unknown") });
971
+ const freshId = randomUUID();
972
+ console.warn(`[responder] session id ${sessionId} already in use — retrying once cold as ${freshId}`);
973
+ cliResult = await runCLI(systemPrompt, userContent, model, {
974
+ sessionId: freshId,
975
+ resumeSession: false,
976
+ router: key ? router : null,
977
+ routingKey: key,
978
+ });
979
+ // Recovered: a person got their answer. Counted separately from the
980
+ // collision so `machine.replyDebt.withheld` does not accuse this seat of
981
+ // a silence it did not commit.
982
+ counters.bump(REPLY_DEBT_COUNTERS.sessionCollisionRecovered, { key: key || "none" });
983
+ sessionId = freshId;
984
+ }
928
985
 
929
986
  const text = (cliResult.text || "").trim();
930
987
  if (!text) {
988
+ // COUNT THE SILENCE BEFORE THROWING IT.
989
+ //
931
990
  // Empty text is "nothing to send" — including the fail-closed non-JSON
932
- // path, where the raw model turn was withheld. Throwing here is what
933
- // keeps it off the wire; the reason rides along for the log.
991
+ // path, where the raw model turn (tool chatter, thinking, an un-enveloped
992
+ // draft) was withheld because it has reached the wire before. Failing
993
+ // closed is right. But from anywhere except this log file, a withheld
994
+ // reply and a quiet hour are the same picture, and that is how a seat
995
+ // stops answering people without anything saying so.
996
+ //
997
+ // MEASURED FIRST, THEN SIZED. This path fired ZERO times in this seat's
998
+ // daemon logs between 2026-09-21 (when the fail-closed change landed) and
999
+ // 2026-09-24 — every observed quick-reply failure in that window was a
1000
+ // session-id collision instead. So the fix is a counter and a number on
1001
+ // the beat, not a retry machine for a failure nobody has seen yet.
1002
+ if (cliResult.withheldChars !== undefined || cliResult.error) {
1003
+ counters.bump(REPLY_DEBT_COUNTERS.unparseable, {
1004
+ model: String(model || "unknown"),
1005
+ chars: Number(cliResult.withheldChars) || 0,
1006
+ service: String((item && item.service) || "unknown"),
1007
+ });
1008
+ }
934
1009
  throw new Error(`claude CLI returned empty result text in generateResponse${cliResult.error ? ` (${cliResult.error}; ${cliResult.withheldChars ?? 0} chars withheld)` : ""}`);
935
1010
  }
936
1011
 
@@ -52,7 +52,7 @@ import { pathToFileURL } from "node:url";
52
52
  import { resolveAgentRoot } from "../../lib/agent-root.mjs";
53
53
  import { writeJsonAtomic as fsWriteJsonAtomic, writeFileAtomic as fsWriteFileAtomic } from "../../lib/fs-atomic.mjs";
54
54
  import { parseHeartbeat } from "../../lib/session/liveness.mjs";
55
- import { ensureClaudeConfig, heartbeatSilence, attentionRecord, HEARTBEAT_GRACE_MS } from "../../lib/session/first-run.mjs";
55
+ import { ensureClaudeConfig, heartbeatSilence, attentionRecord, carriedRestarts, HEARTBEAT_GRACE_MS } from "../../lib/session/first-run.mjs";
56
56
  import { paneTail, classifyPane, watchdogAction, captureCommand, sendKeyCommand } from "../../lib/session/pane.mjs";
57
57
  import { acquireLock as singletonAcquireLock } from "../../lib/singleton.js";
58
58
  import { buildSpawn } from "../../lib/runtime/adapter.mjs";
@@ -96,6 +96,25 @@ function defaultAfter(ms, fn) {
96
96
  return () => clearTimeout(t);
97
97
  }
98
98
 
99
+ /**
100
+ * The installed Claude Code version, for the version-gated first-run screens.
101
+ *
102
+ * Fail-open to `null`: a version we could not read is never GUESSED, and
103
+ * `seedClaudeConfig` leaves the version-gated keys alone when it gets none.
104
+ * Writing an invented version would re-open the very screen the seeding exists
105
+ * to close, which is strictly worse than leaving it to the watchdog.
106
+ *
107
+ * @returns {Promise<string|null>}
108
+ */
109
+ async function readCliVersion(bin, d) {
110
+ try {
111
+ const r = await d.execFile(bin, ["--version"]);
112
+ if (!r || r.code !== 0) return null;
113
+ const m = String(r.stdout || "").trim().match(/(\d+\.\d+\.\d+)/);
114
+ return m ? m[1] : null;
115
+ } catch { return null; }
116
+ }
117
+
99
118
  /**
100
119
  * Real `execFile`: spawn without a shell, capture stdout, resolve with the
101
120
  * exit code (never reject on a non-zero code — that is a result, not a defect).
@@ -162,6 +181,7 @@ export async function runSupervisor(deps = {}) {
162
181
  homeDir: deps.homeDir === undefined ? homedir() : deps.homeDir,
163
182
  after: deps.after || defaultAfter,
164
183
  heartbeatGraceMs: deps.heartbeatGraceMs ?? HEARTBEAT_GRACE_MS,
184
+ cliVersion: deps.cliVersion === undefined ? undefined : deps.cliVersion,
165
185
  mkdirSync: deps.mkdirSync || fsMkdirSync,
166
186
  unlinkSync: deps.unlinkSync || fsUnlinkSync,
167
187
  now: deps.now || Date.now,
@@ -286,20 +306,48 @@ export async function runSupervisor(deps = {}) {
286
306
  const envFile = engineCohort ? join(paths.stateDir, `engine-env-${d.envFileId()}.env`) : null;
287
307
  const cmds = buildMuxCommands({ mux, muxName, cwd: agentRoot, claudeBin: spec.bin, claudeArgs, exitFile: paths.lastExitFile, envFile });
288
308
 
289
- // 4b. First-run dialogs. Nobody is attached to press a key, so the two
290
- // one-time acknowledgements are recorded up front. Fail-open: a seat
291
- // still launches when this cannot be done — the watchdog below then
292
- // reports the blocked session instead of nothing at all. The engine has
293
- // no first-run dialogs, so an engine-cohort seat's ~/.claude.json is left alone.
309
+ // 4b. First-run dialogs. Nobody is attached to press a key, so every one-time
310
+ // acknowledgement is recorded up front. Fail-open: a seat still launches
311
+ // when this cannot be done — the watchdog below then reads the screen and
312
+ // acts on it instead of nothing at all. The engine has no first-run
313
+ // dialogs, so an engine-cohort seat's ~/.claude.json is left alone.
314
+ //
315
+ // Two of those screens are VERSION-GATED (`lastOnboardingVersion`,
316
+ // `lastReleaseNotesSeen`): the CLI compares them against the version it is
317
+ // RUNNING, so an upgrade re-opens a screen a seeded seat had already
318
+ // passed. That is the difference between "the seat was set up wrong" and
319
+ // "the seat was fine until Tuesday" — and it was the second on the seats
320
+ // that went silent inside an upgrade window. So the seeder has to be told
321
+ // which version is installed; an unreadable one is `null` and those keys
322
+ // are left alone rather than guessed.
323
+ const cliVersion = d.cliVersion !== undefined
324
+ ? d.cliVersion
325
+ : (engineCohort ? null : await readCliVersion(spec.bin, d));
294
326
  if (!engineCohort) {
295
327
  const seeded = ensureClaudeConfig({
296
328
  homeDir: d.homeDir, agentRoot, bypass: permissionArgs.includes("--dangerously-skip-permissions"),
329
+ cliVersion,
297
330
  readFileSync: d.readFileSync, writeFileAtomic: d.writeFileAtomic,
298
331
  });
299
- if (seeded.changed) d.log(`${seeded.path} seeded for an unattended launch: ${seeded.applied.join(", ")}`);
300
- else if (!seeded.ok) d.log(`${seeded.path || "~/.claude.json"} not seeded (${seeded.error}) — a first-run dialog may block the session; the watchdog will report it`);
332
+ if (seeded.changed) d.log(`${seeded.path} seeded for an unattended launch${cliVersion ? ` (cli ${cliVersion})` : " (cli version unreadable — the version-gated screens were left alone)"}: ${seeded.applied.join(", ")}`);
333
+ else if (!seeded.ok) d.log(`${seeded.path || "~/.claude.json"} not seeded (${seeded.error}) — a first-run dialog may block the session; the watchdog reads the screen and acts on it`);
301
334
  }
302
335
 
336
+ // THE RESTART BUDGET IS CARRIED, NOT RESET — and it has to be read HERE,
337
+ // before the unlink below destroys the record that holds it. Every restart
338
+ // the watchdog performs ENDS this process and launchd starts a fresh one, so
339
+ // `restarts` held in memory is zero on every launch and `watchdogAction`'s
340
+ // bound never binds: a seat blocked on something a relaunch cannot fix would
341
+ // restart every two minutes forever. `carriedRestarts` resets it on evidence
342
+ // of work — a beat stamped after the record — so a seat that recovered once
343
+ // gets a full budget for its next fault.
344
+ let priorAttention = null;
345
+ try { priorAttention = JSON.parse(d.readFileSync(paths.attentionFile, "utf8")); } catch { priorAttention = null; }
346
+ let priorHeartbeat = null;
347
+ try { priorHeartbeat = parseHeartbeat(d.readFileSync(paths.heartbeatFile, "utf8")); } catch { priorHeartbeat = null; }
348
+ const restartsSoFar = carriedRestarts({ attention: priorAttention, heartbeat: priorHeartbeat });
349
+ if (restartsSoFar > 0) d.log(`this launch carries ${restartsSoFar} silent restart(s) from the previous run`);
350
+
303
351
  try { d.unlinkSync(paths.lastExitFile); } catch { /* none from a previous run */ }
304
352
  try { d.unlinkSync(paths.attentionFile); } catch { /* none outstanding */ }
305
353
  // 4c. An upgrade notice asking for a restart onto the version THIS launch
@@ -350,7 +398,7 @@ export async function runSupervisor(deps = {}) {
350
398
  // worse failure than the one it replaces. Every pass records what it saw, so
351
399
  // the note stops being a guess and becomes evidence.
352
400
  let clears = 0;
353
- let restarts = 0;
401
+ let restarts = restartsSoFar;
354
402
  let watchdogStopped = false;
355
403
  let cancelPass = () => {};
356
404
  const paneFile = join(agentRoot, "state", "session", "pane-capture.txt");
@@ -388,20 +436,29 @@ export async function runSupervisor(deps = {}) {
388
436
  const seen = classifyPane(pane);
389
437
  const act = watchdogAction({ silent: true, kind: seen.kind, keys: seen.keys, clears, restarts });
390
438
 
391
- const rec = attentionRecord({ reason: silence.reason, mux, muxName, since: startedAt, runMs: silence.runMs });
439
+ // THE RECORD COUNTS THE ACTION IT IS ABOUT TO TAKE, not the ones before it.
440
+ // A restart ENDS this process, so this write is the last chance to tell the
441
+ // next launch what was spent; recording the pre-action count would hand
442
+ // every relaunch a budget one short of the truth and the bound would never
443
+ // reach its limit.
444
+ const nextClears = act.act === "clear" ? clears + 1 : clears;
445
+ const nextRestarts = act.act === "restart" ? restarts + 1 : restarts;
446
+ const rec = attentionRecord({ reason: silence.reason, mux, muxName, since: startedAt, runMs: silence.runMs, restarts: nextRestarts, action: act.act });
392
447
  // The record now carries WHAT THE SCREEN SAYS and what was done about it.
393
448
  // `hint` keeps the attach line for a human who is at the machine, but it is
394
449
  // no longer the only remedy on offer.
395
- const record = { ...rec, modal: seen.kind, modalWhy: seen.why, pane, action: act.act, actionReason: act.reason, clears, restarts };
450
+ const record = { ...rec, modal: seen.kind, modalWhy: seen.why, pane, action: act.act, actionReason: act.reason, clears: nextClears, restarts: nextRestarts };
451
+ // Written BEFORE the stop, because the stop ends this process's ability to
452
+ // write anything at all.
396
453
  try { d.writeJsonAtomic(paths.attentionFile, record); } catch { /* the log line still says it */ }
397
454
  d.log(`session ${muxName} silent ${Math.round(silence.runMs / 1000)} s — screen shows: ${seen.kind} (${seen.why}); ${act.reason}`);
398
455
 
399
456
  if (act.act === "clear") {
400
- clears += 1;
457
+ clears = nextClears;
401
458
  d.log(`session ${muxName}: answering ${seen.kind} with ${JSON.stringify(seen.choice || seen.keys)}`);
402
459
  await sendKeys(seen.keys);
403
460
  } else if (act.act === "restart") {
404
- restarts += 1;
461
+ restarts = nextRestarts;
405
462
  d.log(`session ${muxName}: restarting (${restarts}) — the session is not answering and the screen cannot be cleared safely`);
406
463
  try { await d.execFile(cmds.stop.file, cmds.stop.args, { cwd: agentRoot, env: d.env }); } catch { /* the relaunch loop handles a dead mux */ }
407
464
  watchdogStopped = true; // the supervisor's own loop relaunches; do not fight it