@cohortapp/agent-sdk 2.18.13 → 2.18.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -42,7 +42,7 @@
42
42
  * @module scripts/session/supervisor
43
43
  */
44
44
 
45
- import { existsSync as fsExistsSync, readFileSync as fsReadFileSync, mkdirSync as fsMkdirSync, unlinkSync as fsUnlinkSync, writeFileSync as fsWriteFileSync, chmodSync as fsChmodSync } from "node:fs";
45
+ import { existsSync as fsExistsSync, readFileSync as fsReadFileSync, mkdirSync as fsMkdirSync, unlinkSync as fsUnlinkSync, writeFileSync as fsWriteFileSync, chmodSync as fsChmodSync, readdirSync as fsReaddirSync } from "node:fs";
46
46
  import { randomUUID } from "node:crypto";
47
47
  import { join } from "node:path";
48
48
  import { spawn } from "node:child_process";
@@ -66,6 +66,11 @@ import {
66
66
  launchMode, recordLaunch, rotationDecision, rotateMainSession, BACKOFF_MS,
67
67
  } from "../../lib/session/identity.mjs";
68
68
  import { detectMux, chooseMux, buildMuxCommands, parseScreenList, parseExitFile, renderEnvFile } from "../../lib/session/launch-args.mjs";
69
+ import { resumeTargetState } from "../../lib/session/resume-target.mjs";
70
+ import {
71
+ classifyLaunchFailure, launchFailureSignature, parseLaunchFailures,
72
+ launchFailureStreak, escalationDecision, escalationHint, IDENTICAL_FAILURE_LIMIT,
73
+ } from "../../lib/session/launch-failure.mjs";
69
74
 
70
75
  /** Write a private (0600) file that must not already exist. The engine env file carries the seat token. */
71
76
  export function writePrivateFile(path, text) {
@@ -83,6 +88,17 @@ export const LOCK_NAME = "session";
83
88
  export const DEFAULT_POLL_MS = 2_000;
84
89
  /** The signals the supervisor owns (launchd sends SIGTERM to stop a job). */
85
90
  export const OWNED_SIGNALS = Object.freeze(["SIGTERM", "SIGINT"]);
91
+ /**
92
+ * How long after start the pane is captured ONCE, to keep whatever the runtime
93
+ * printed before it died. A resume against a missing conversation is over in
94
+ * about two seconds, taking its mux session (and its pane) with it — so the
95
+ * capture has to happen while it is still there. Best-effort by design: when
96
+ * it comes back empty, `resumeTargetState`'s preflight is what carries the
97
+ * verdict, and the post-mortem heuristics below carry the rest.
98
+ */
99
+ export const EARLY_PANE_MS = 1_500;
100
+ /** Consecutive identical launch failures before the seat escalates to the org. */
101
+ export const LAUNCH_FAILURE_LIMIT = IDENTICAL_FAILURE_LIMIT;
86
102
 
87
103
  /** Real signal wiring: `fn(signal)` on each owned signal. */
88
104
  function installSignalHandlers(fn) {
@@ -198,6 +214,15 @@ export async function runSupervisor(deps = {}) {
198
214
  seatSpawn: deps.seatSpawn || ((root) => resolveSeatSpawn(root)),
199
215
  writePrivateFile: deps.writePrivateFile || writePrivateFile,
200
216
  envFileId: deps.envFileId || randomUUID,
217
+ readdirSync: deps.readdirSync || fsReaddirSync,
218
+ launchFailureLimit: deps.launchFailureLimit ?? LAUNCH_FAILURE_LIMIT,
219
+ earlyPaneMs: deps.earlyPaneMs ?? EARLY_PANE_MS,
220
+ // The one-shot pane probe has its OWN timer seam, separate from the
221
+ // watchdog's `after`. They are different clocks doing different jobs — one
222
+ // fires once at a second and a half, the other recurs at the grace period —
223
+ // and collapsing them would make every watchdog test reason about a timer
224
+ // that has nothing to do with the watchdog.
225
+ afterPaneProbe: deps.afterPaneProbe || deps.after || defaultAfter,
201
226
  };
202
227
  const fsDeps = { readFileSync: d.readFileSync, writeJsonAtomic: d.writeJsonAtomic };
203
228
  const agentRoot = d.agentRoot;
@@ -271,7 +296,7 @@ export async function runSupervisor(deps = {}) {
271
296
  // 4. Identity + argv.
272
297
  let record = loadMainSession(agentRoot, fsDeps);
273
298
  if (!record || cfg.resume === false) record = newMainSession({ now: d.now, uuid: d.uuid });
274
- const mode = launchMode(record, cfg);
299
+ let mode = launchMode(record, cfg);
275
300
  let permissionArgs = d.permissionArgs;
276
301
  if (!permissionArgs) {
277
302
  permissionArgs = sessionPermissionArgs({ source: MAIN_SESSION_SOURCE, allowedTools: cfg.allowedTools });
@@ -288,6 +313,38 @@ export async function runSupervisor(deps = {}) {
288
313
  let seat = { engine: "claude", fields: {} };
289
314
  try { seat = (await d.seatSpawn(agentRoot)) || seat; } catch { seat = { engine: "claude", fields: {} }; }
290
315
  const engineCohort = seat.engine === "cohort";
316
+
317
+ // ── (a) A DEAD RESUME TARGET IS CAUGHT BEFORE THE LAUNCH, NOT AFTER IT ────
318
+ //
319
+ // James Kirkland's seat, measured 2026-09-25: `claude --resume <id>` for a
320
+ // conversation that no longer existed on that machine, exit 1 after 2 s,
321
+ // relaunched every ten minutes for DAYS. state/session/main-session.json held
322
+ // the dead id and nothing ever invalidated it.
323
+ //
324
+ // THE CHOICE, MADE EXPLICIT: invalidate the record AND fall back to a fresh
325
+ // id — both, in one write, here, before anything is spawned. Invalidating
326
+ // alone would leave the supervisor with no id (and `newMainSession` would
327
+ // mint one anyway, unrecorded); falling back alone would leave the dead id on
328
+ // disk for the next reader to resume again. `rotateMainSession` does both:
329
+ // the dead id survives as `rotatedFrom` — provenance, never a target.
330
+ //
331
+ // It runs on POSITIVE EVIDENCE ONLY (`resume-target.mjs` fails open): a
332
+ // wrong "missing" throws away a live transcript, a wrong "unknown" costs one
333
+ // launch and lands in the post-mortem that was already there.
334
+ if (mode === "resume") {
335
+ const target = resumeTargetState(
336
+ { homeDir: d.homeDir, projectDir: agentRoot, sessionId: record.sessionId, engine: seat.engine },
337
+ { readdirSync: d.readdirSync },
338
+ );
339
+ if (target.state === "missing") {
340
+ const dead = record.sessionId;
341
+ record = rotateMainSession(record, { now: d.now, uuid: d.uuid, reason: `resume target ${dead} is not on this machine` });
342
+ mode = launchMode(record, cfg);
343
+ const res = saveMainSession(agentRoot, record, fsDeps);
344
+ d.log(`resume target ${dead} is gone (${target.reason}) — invalidated it and starting a fresh session ${record.sessionId}${res.ok ? "" : ` (not saved: ${res.error})`}`);
345
+ }
346
+ }
347
+
291
348
  const spec = buildSpawn({ lane: "main-session", bin: d.claudeBin || undefined, first, sessionId: record.sessionId, resumeMode: mode, permissions: permissionArgs, env: d.env, ...seat.fields });
292
349
  if (!spec.ok && engineCohort) {
293
350
  // An engine-cohort seat never falls back to claude (it would spend the seat's
@@ -348,8 +405,34 @@ export async function runSupervisor(deps = {}) {
348
405
  const restartsSoFar = carriedRestarts({ attention: priorAttention, heartbeat: priorHeartbeat });
349
406
  if (restartsSoFar > 0) d.log(`this launch carries ${restartsSoFar} silent restart(s) from the previous run`);
350
407
 
408
+ // ── (b) THE CARRIED LAUNCH-FAILURE STREAK ────────────────────────────────
409
+ //
410
+ // Read BEFORE the unlink below, for the same reason `carriedRestarts` is: a
411
+ // count held in memory is zero on every launch, because every launch is a
412
+ // fresh process. This one lives in its own file rather than on the attention
413
+ // record, because it has to survive the unlink that clears that record.
414
+ let priorFailures = null;
415
+ try { priorFailures = parseLaunchFailures(d.readFileSync(paths.launchFailuresFile, "utf8")); } catch { priorFailures = null; }
416
+
351
417
  try { d.unlinkSync(paths.lastExitFile); } catch { /* none from a previous run */ }
352
418
  try { d.unlinkSync(paths.attentionFile); } catch { /* none outstanding */ }
419
+
420
+ // A seat already past the limit re-states the escalation on EVERY launch.
421
+ // The note rides `machine.sessionNote` on the presence beat, and the unlink
422
+ // above just cleared it: without this the org would see the alert blink out
423
+ // each time the seat tried again, which reads as a seat that healed.
424
+ if (priorFailures) {
425
+ const carried = escalationDecision({ streak: priorFailures.streak, limit: d.launchFailureLimit, escalatedAt: priorFailures.escalatedAt });
426
+ if (carried.escalate) {
427
+ writeLaunchAttention({
428
+ d, paths, mux, muxName, now: d.now(),
429
+ streak: carried.streak, limit: carried.limit, signature: priorFailures.signature,
430
+ detail: `${priorFailures.signature} since ${priorFailures.firstAt || "an earlier launch"}`,
431
+ sessionId: record.sessionId, phase: "retrying",
432
+ });
433
+ d.log(`launch-failure escalation still standing: ${carried.reason}`);
434
+ }
435
+ }
353
436
  // 4c. An upgrade notice asking for a restart onto the version THIS launch
354
437
  // runs is honoured by this launch: retire it, so `maestro session
355
438
  // status` stops flagging a hop the session has completed (and a notice
@@ -380,6 +463,9 @@ export async function runSupervisor(deps = {}) {
380
463
  // almost certainly waiting on a dialog or a prompt nobody can see.
381
464
  const startedAt = Number(d.now());
382
465
  let launchFailed = false;
466
+ /** Whatever the mux client itself printed — `screen -D -m` reports a missing
467
+ * command here, and it is the one piece of evidence no pane capture can race. */
468
+ let startStdout = "";
383
469
  stopCmd = cmds.stop;
384
470
  // ── THE WATCHDOG ────────────────────────────────────────────────────────
385
471
  //
@@ -400,6 +486,10 @@ export async function runSupervisor(deps = {}) {
400
486
  let clears = 0;
401
487
  let restarts = restartsSoFar;
402
488
  let watchdogStopped = false;
489
+ /** The last non-empty pane the watchdog saw — evidence for the post-mortem. */
490
+ let lastPane = "";
491
+ /** Did the WATCHDOG end this session? A death we caused is never rotated on. */
492
+ let watchdogRestarted = false;
403
493
  let cancelPass = () => {};
404
494
  const paneFile = join(agentRoot, "state", "session", "pane-capture.txt");
405
495
 
@@ -412,6 +502,17 @@ export async function runSupervisor(deps = {}) {
412
502
  } catch { return ""; }
413
503
  };
414
504
 
505
+ // The pane is captured ONCE, early, and kept. A resume whose conversation is
506
+ // gone prints "No conversation found with session ID: <id>" and exits inside
507
+ // a couple of seconds, taking the mux session and its pane with it — so the
508
+ // only moment this evidence exists is while the doomed session is still up.
509
+ // Best-effort: an empty capture is not an absence of fault, it is an absence
510
+ // of evidence, and `classifyLaunchFailure` is written to say so.
511
+ let earlyPane = "";
512
+ d.afterPaneProbe(d.earlyPaneMs, () => {
513
+ capturePane().then((text) => { if (!earlyPane) earlyPane = paneTail(text); }).catch(() => {});
514
+ });
515
+
415
516
  const sendKeys = async (keys) => {
416
517
  for (const k of keys) {
417
518
  const cmd = sendKeyCommand(mux, muxName, k);
@@ -433,6 +534,7 @@ export async function runSupervisor(deps = {}) {
433
534
  if (!silence.silent) { schedulePass(); return; }
434
535
 
435
536
  const pane = paneTail(await capturePane());
537
+ if (pane) lastPane = pane;
436
538
  const seen = classifyPane(pane);
437
539
  const act = watchdogAction({ silent: true, kind: seen.kind, keys: seen.keys, clears, restarts });
438
540
 
@@ -459,6 +561,7 @@ export async function runSupervisor(deps = {}) {
459
561
  await sendKeys(seen.keys);
460
562
  } else if (act.act === "restart") {
461
563
  restarts = nextRestarts;
564
+ watchdogRestarted = true;
462
565
  d.log(`session ${muxName}: restarting (${restarts}) — the session is not answering and the screen cannot be cleared safely`);
463
566
  try { await d.execFile(cmds.stop.file, cmds.stop.args, { cwd: agentRoot, env: d.env }); } catch { /* the relaunch loop handles a dead mux */ }
464
567
  watchdogStopped = true; // the supervisor's own loop relaunches; do not fight it
@@ -485,6 +588,7 @@ export async function runSupervisor(deps = {}) {
485
588
  // env file; the mux client itself gets the supervisor's env, as for claude.
486
589
  if (envFile) d.writePrivateFile(envFile, renderEnvFile(spec.env));
487
590
  const r = await d.execFile(cmds.start.file, cmds.start.args, { cwd: agentRoot, env: d.env });
591
+ startStdout = `${String((r && r.stdout) || "")}${String((r && r.stderr) || "")}`.trim();
488
592
  if (r && r.code !== 0) { launchFailed = true; d.log(`${mux.kind} start exited ${r.code}${r.stderr ? `: ${String(r.stderr).trim()}` : ""}`); }
489
593
  else if (!cmds.blocking) await waitForProbe(cmds.waitProbe, d);
490
594
  } catch (err) {
@@ -506,11 +610,69 @@ export async function runSupervisor(deps = {}) {
506
610
  d.log(`session ${muxName} stopped on ${stopping} — id kept, exit ${EXIT_RELAUNCH}`);
507
611
  return EXIT_RELAUNCH;
508
612
  }
509
- const decision = rotationDecision({ record: launched, mode, exitCode, startedAt, endedAt, now: d.now });
613
+
614
+ // Did THIS launch ever report in? The single most useful fact about a run,
615
+ // and the one the ten-minute relaunch loop never asked: James's seat looked
616
+ // up for days and had not beaten once.
617
+ let endHeartbeat = null;
618
+ try { endHeartbeat = parseHeartbeat(d.readFileSync(paths.heartbeatFile, "utf8")); } catch { endHeartbeat = null; }
619
+ const endBeatMs = endHeartbeat && typeof endHeartbeat.ts === "string" ? Date.parse(endHeartbeat.ts) : NaN;
620
+ const beatSeen = Number.isFinite(endBeatMs) && endBeatMs >= startedAt;
621
+
622
+ // What the runtime itself said, if we caught it. `earlyPane` is the capture
623
+ // taken while a fast-failing session was still up; `lastPane` is the
624
+ // watchdog's. `startStdout` is whatever the mux client printed to us.
625
+ const evidence = [startStdout, earlyPane, lastPane].filter(Boolean).join("\n");
626
+ const failure = classifyLaunchFailure({ mode, exitCode, output: evidence, beatSeen });
627
+ if (failure.kind !== "none") d.log(`launch verdict: ${failure.kind}${failure.proven ? " (proven)" : ""} — ${failure.detail}`);
628
+
629
+ // ── (b) N IDENTICAL FAILURES ARE A FAULT, NOT N RETRIES ──────────────────
630
+ //
631
+ // The ten-minute backoff was the right instinct and the wrong outcome: every
632
+ // attempt failed identically and the seat stayed quiet about it for days,
633
+ // because the only record was a log on a machine nobody can reach. So the
634
+ // streak is counted across supervisor lifetimes and, at the limit, written
635
+ // where it LEAVES the machine — `state/session/attention.json` rides the
636
+ // presence beat as `machine.sessionNote`, which is the org's view of this
637
+ // seat. A watchdog restart is not counted here: that path has its own
638
+ // bounded budget (`carriedRestarts`) and counting it twice would escalate a
639
+ // fault the seat is already handling.
640
+ let escalation = { escalate: false, fresh: false, streak: 0, limit: d.launchFailureLimit, reason: "" };
641
+ if (!watchdogRestarted) {
642
+ const signature = launchFailureSignature({ mode, kind: failure.kind, exitCode });
643
+ const next = launchFailureStreak(priorFailures, { signature, at: new Date(Number(d.now())).toISOString() });
644
+ if (!next) {
645
+ // The launch worked. Evidence of work clears the record — the same reset
646
+ // rule `carriedRestarts` uses, and for the same reason.
647
+ try { d.unlinkSync(paths.launchFailuresFile); } catch { /* nothing carried */ }
648
+ } else {
649
+ escalation = escalationDecision({ streak: next.streak, limit: d.launchFailureLimit, escalatedAt: next.escalatedAt });
650
+ if (escalation.escalate && !next.escalatedAt) next.escalatedAt = new Date(Number(d.now())).toISOString();
651
+ try { d.writeJsonAtomic(paths.launchFailuresFile, next); } catch { /* the log line still says it */ }
652
+ d.log(`launch failure ${next.streak}× in a row (${signature}) — ${escalation.reason}`);
653
+ if (escalation.escalate) {
654
+ writeLaunchAttention({
655
+ d, paths, mux, muxName, now: d.now(),
656
+ streak: next.streak, limit: escalation.limit, signature,
657
+ detail: failure.detail, sessionId: launched.sessionId, phase: "failed",
658
+ });
659
+ if (escalation.fresh) d.log(`escalating to the org: ${escalationHint({ signature, streak: next.streak, limit: escalation.limit, detail: failure.detail, sessionId: launched.sessionId })}`);
660
+ }
661
+ }
662
+ }
663
+
664
+ const decision = rotationDecision({
665
+ record: launched, mode, exitCode, startedAt, endedAt, now: d.now,
666
+ resumeTargetMissing: failure.kind === "resume-target-missing",
667
+ beatSeen,
668
+ streakAtLimit: escalation.escalate,
669
+ selfStopped: watchdogRestarted,
670
+ });
510
671
  if (decision.action === "rotate") {
511
- const rotated = rotateMainSession(launched, { now: d.now, uuid: d.uuid });
672
+ const why = decision.why || "the resume failed";
673
+ const rotated = rotateMainSession(launched, { now: d.now, uuid: d.uuid, reason: why });
512
674
  const res = saveMainSession(agentRoot, rotated, fsDeps);
513
- d.log(`resume of ${launched.sessionId} failed inside the failure window — rotated to ${rotated.sessionId}${res.ok ? "" : ` (not saved: ${res.error})`}`);
675
+ d.log(`resume of ${launched.sessionId}: ${why}${decision.proven ? " (proven, so not rationed by the rotation budget)" : ""} — invalidated it and rotated to ${rotated.sessionId}${res.ok ? "" : ` (not saved: ${res.error})`}`);
514
676
  } else if (decision.action === "backoff") {
515
677
  d.log(`rotation budget spent for this hour — sleeping ${decision.sleepMs / 60000} min before relaunch`);
516
678
  await d.sleep(decision.sleepMs);
@@ -518,6 +680,37 @@ export async function runSupervisor(deps = {}) {
518
680
  return EXIT_RELAUNCH;
519
681
  }
520
682
 
683
+ /**
684
+ * Write the escalation where it LEAVES THE MACHINE.
685
+ *
686
+ * `state/session/attention.json` is read by `lib/telemetry/collect#sessionNote`
687
+ * and rides the presence beat to hq as `machine.sessionNote` — the only channel
688
+ * out of a seat that accepts no ssh. The record is shaped exactly like the
689
+ * watchdog's (`first-run#attentionRecord` fields) so the existing reader needs
690
+ * no special case; `restarts: 0` is honest and, per `carriedRestarts`, cannot
691
+ * steal budget from the watchdog.
692
+ *
693
+ * @param {object} a
694
+ */
695
+ export function writeLaunchAttention(a) {
696
+ const { d, paths, mux, muxName } = a;
697
+ const attach = mux && mux.kind === "tmux" ? `tmux attach -t =${muxName}` : `screen -r ${muxName}`;
698
+ const record = {
699
+ reason: "launch-failing",
700
+ since: new Date(Number(a.now)).toISOString(),
701
+ runMs: 0,
702
+ restarts: 0,
703
+ action: a.phase === "retrying" ? "wait" : "give-up",
704
+ attach,
705
+ signature: a.signature,
706
+ streak: a.streak,
707
+ limit: a.limit,
708
+ hint: escalationHint({ signature: a.signature, streak: a.streak, limit: a.limit, detail: a.detail, sessionId: a.sessionId }),
709
+ };
710
+ try { d.writeJsonAtomic(paths.attentionFile, record); } catch { /* the log line still says it */ }
711
+ return record;
712
+ }
713
+
521
714
  const isMain = (() => {
522
715
  try { return process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href; } catch { return false; }
523
716
  })();