@cohortapp/agent-sdk 2.18.13 → 2.18.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/maestro.mjs +38 -1
- package/docs/runbooks/fleet-rollout.md +14 -7
- package/docs/runbooks/recovery-and-failover.md +18 -0
- package/lib/cadence-failure-class.mjs +245 -0
- package/lib/claude-bin.mjs +26 -7
- package/lib/cli/doctor-checks.mjs +149 -1
- package/lib/diagnostics/alerts.mjs +33 -0
- package/lib/engine/agents/usage.mjs +45 -0
- package/lib/engine/budget.mjs +293 -29
- package/lib/engine/cli.mjs +54 -5
- package/lib/engine/loop.mjs +30 -0
- package/lib/engine/output/json.mjs +26 -0
- package/lib/engine/wire/errors.mjs +179 -0
- package/lib/engine/wire/search.mjs +44 -8
- package/lib/org/quota.mjs +27 -0
- package/lib/session/config.mjs +4 -0
- package/lib/session/identity.mjs +71 -7
- package/lib/session/launch-failure.mjs +251 -0
- package/lib/session/resume-target.mjs +86 -0
- package/lib/telemetry/collect.mjs +129 -0
- package/lib/upgrade/pinned-drift.mjs +467 -0
- package/package.json +1 -1
- package/scaffold/config/alerts.yaml +7 -0
- package/scripts/ci/check-cadence-prompts-exist.mjs +96 -0
- package/scripts/ci/check.mjs +3 -0
- package/scripts/daemon/cadence-consumer.mjs +281 -34
- package/scripts/emergency-stop.sh +114 -13
- package/scripts/fleet/rollout.mjs +10 -3
- package/scripts/healthcheck.sh +131 -33
- package/scripts/resume-operations.sh +101 -6
- package/scripts/session/supervisor.mjs +198 -5
|
@@ -42,7 +42,7 @@
|
|
|
42
42
|
* @module scripts/session/supervisor
|
|
43
43
|
*/
|
|
44
44
|
|
|
45
|
-
import { existsSync as fsExistsSync, readFileSync as fsReadFileSync, mkdirSync as fsMkdirSync, unlinkSync as fsUnlinkSync, writeFileSync as fsWriteFileSync, chmodSync as fsChmodSync } from "node:fs";
|
|
45
|
+
import { existsSync as fsExistsSync, readFileSync as fsReadFileSync, mkdirSync as fsMkdirSync, unlinkSync as fsUnlinkSync, writeFileSync as fsWriteFileSync, chmodSync as fsChmodSync, readdirSync as fsReaddirSync } from "node:fs";
|
|
46
46
|
import { randomUUID } from "node:crypto";
|
|
47
47
|
import { join } from "node:path";
|
|
48
48
|
import { spawn } from "node:child_process";
|
|
@@ -66,6 +66,11 @@ import {
|
|
|
66
66
|
launchMode, recordLaunch, rotationDecision, rotateMainSession, BACKOFF_MS,
|
|
67
67
|
} from "../../lib/session/identity.mjs";
|
|
68
68
|
import { detectMux, chooseMux, buildMuxCommands, parseScreenList, parseExitFile, renderEnvFile } from "../../lib/session/launch-args.mjs";
|
|
69
|
+
import { resumeTargetState } from "../../lib/session/resume-target.mjs";
|
|
70
|
+
import {
|
|
71
|
+
classifyLaunchFailure, launchFailureSignature, parseLaunchFailures,
|
|
72
|
+
launchFailureStreak, escalationDecision, escalationHint, IDENTICAL_FAILURE_LIMIT,
|
|
73
|
+
} from "../../lib/session/launch-failure.mjs";
|
|
69
74
|
|
|
70
75
|
/** Write a private (0600) file that must not already exist. The engine env file carries the seat token. */
|
|
71
76
|
export function writePrivateFile(path, text) {
|
|
@@ -83,6 +88,17 @@ export const LOCK_NAME = "session";
|
|
|
83
88
|
export const DEFAULT_POLL_MS = 2_000;
|
|
84
89
|
/** The signals the supervisor owns (launchd sends SIGTERM to stop a job). */
|
|
85
90
|
export const OWNED_SIGNALS = Object.freeze(["SIGTERM", "SIGINT"]);
|
|
91
|
+
/**
|
|
92
|
+
* How long after start the pane is captured ONCE, to keep whatever the runtime
|
|
93
|
+
* printed before it died. A resume against a missing conversation is over in
|
|
94
|
+
* about two seconds, taking its mux session (and its pane) with it — so the
|
|
95
|
+
* capture has to happen while it is still there. Best-effort by design: when
|
|
96
|
+
* it comes back empty, `resumeTargetState`'s preflight is what carries the
|
|
97
|
+
* verdict, and the post-mortem heuristics below carry the rest.
|
|
98
|
+
*/
|
|
99
|
+
export const EARLY_PANE_MS = 1_500;
|
|
100
|
+
/** Consecutive identical launch failures before the seat escalates to the org. */
|
|
101
|
+
export const LAUNCH_FAILURE_LIMIT = IDENTICAL_FAILURE_LIMIT;
|
|
86
102
|
|
|
87
103
|
/** Real signal wiring: `fn(signal)` on each owned signal. */
|
|
88
104
|
function installSignalHandlers(fn) {
|
|
@@ -198,6 +214,15 @@ export async function runSupervisor(deps = {}) {
|
|
|
198
214
|
seatSpawn: deps.seatSpawn || ((root) => resolveSeatSpawn(root)),
|
|
199
215
|
writePrivateFile: deps.writePrivateFile || writePrivateFile,
|
|
200
216
|
envFileId: deps.envFileId || randomUUID,
|
|
217
|
+
readdirSync: deps.readdirSync || fsReaddirSync,
|
|
218
|
+
launchFailureLimit: deps.launchFailureLimit ?? LAUNCH_FAILURE_LIMIT,
|
|
219
|
+
earlyPaneMs: deps.earlyPaneMs ?? EARLY_PANE_MS,
|
|
220
|
+
// The one-shot pane probe has its OWN timer seam, separate from the
|
|
221
|
+
// watchdog's `after`. They are different clocks doing different jobs — one
|
|
222
|
+
// fires once at a second and a half, the other recurs at the grace period —
|
|
223
|
+
// and collapsing them would make every watchdog test reason about a timer
|
|
224
|
+
// that has nothing to do with the watchdog.
|
|
225
|
+
afterPaneProbe: deps.afterPaneProbe || deps.after || defaultAfter,
|
|
201
226
|
};
|
|
202
227
|
const fsDeps = { readFileSync: d.readFileSync, writeJsonAtomic: d.writeJsonAtomic };
|
|
203
228
|
const agentRoot = d.agentRoot;
|
|
@@ -271,7 +296,7 @@ export async function runSupervisor(deps = {}) {
|
|
|
271
296
|
// 4. Identity + argv.
|
|
272
297
|
let record = loadMainSession(agentRoot, fsDeps);
|
|
273
298
|
if (!record || cfg.resume === false) record = newMainSession({ now: d.now, uuid: d.uuid });
|
|
274
|
-
|
|
299
|
+
let mode = launchMode(record, cfg);
|
|
275
300
|
let permissionArgs = d.permissionArgs;
|
|
276
301
|
if (!permissionArgs) {
|
|
277
302
|
permissionArgs = sessionPermissionArgs({ source: MAIN_SESSION_SOURCE, allowedTools: cfg.allowedTools });
|
|
@@ -288,6 +313,38 @@ export async function runSupervisor(deps = {}) {
|
|
|
288
313
|
let seat = { engine: "claude", fields: {} };
|
|
289
314
|
try { seat = (await d.seatSpawn(agentRoot)) || seat; } catch { seat = { engine: "claude", fields: {} }; }
|
|
290
315
|
const engineCohort = seat.engine === "cohort";
|
|
316
|
+
|
|
317
|
+
// ── (a) A DEAD RESUME TARGET IS CAUGHT BEFORE THE LAUNCH, NOT AFTER IT ────
|
|
318
|
+
//
|
|
319
|
+
// James Kirkland's seat, measured 2026-09-25: `claude --resume <id>` for a
|
|
320
|
+
// conversation that no longer existed on that machine, exit 1 after 2 s,
|
|
321
|
+
// relaunched every ten minutes for DAYS. state/session/main-session.json held
|
|
322
|
+
// the dead id and nothing ever invalidated it.
|
|
323
|
+
//
|
|
324
|
+
// THE CHOICE, MADE EXPLICIT: invalidate the record AND fall back to a fresh
|
|
325
|
+
// id — both, in one write, here, before anything is spawned. Invalidating
|
|
326
|
+
// alone would leave the supervisor with no id (and `newMainSession` would
|
|
327
|
+
// mint one anyway, unrecorded); falling back alone would leave the dead id on
|
|
328
|
+
// disk for the next reader to resume again. `rotateMainSession` does both:
|
|
329
|
+
// the dead id survives as `rotatedFrom` — provenance, never a target.
|
|
330
|
+
//
|
|
331
|
+
// It runs on POSITIVE EVIDENCE ONLY (`resume-target.mjs` fails open): a
|
|
332
|
+
// wrong "missing" throws away a live transcript, a wrong "unknown" costs one
|
|
333
|
+
// launch and lands in the post-mortem that was already there.
|
|
334
|
+
if (mode === "resume") {
|
|
335
|
+
const target = resumeTargetState(
|
|
336
|
+
{ homeDir: d.homeDir, projectDir: agentRoot, sessionId: record.sessionId, engine: seat.engine },
|
|
337
|
+
{ readdirSync: d.readdirSync },
|
|
338
|
+
);
|
|
339
|
+
if (target.state === "missing") {
|
|
340
|
+
const dead = record.sessionId;
|
|
341
|
+
record = rotateMainSession(record, { now: d.now, uuid: d.uuid, reason: `resume target ${dead} is not on this machine` });
|
|
342
|
+
mode = launchMode(record, cfg);
|
|
343
|
+
const res = saveMainSession(agentRoot, record, fsDeps);
|
|
344
|
+
d.log(`resume target ${dead} is gone (${target.reason}) — invalidated it and starting a fresh session ${record.sessionId}${res.ok ? "" : ` (not saved: ${res.error})`}`);
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
|
|
291
348
|
const spec = buildSpawn({ lane: "main-session", bin: d.claudeBin || undefined, first, sessionId: record.sessionId, resumeMode: mode, permissions: permissionArgs, env: d.env, ...seat.fields });
|
|
292
349
|
if (!spec.ok && engineCohort) {
|
|
293
350
|
// An engine-cohort seat never falls back to claude (it would spend the seat's
|
|
@@ -348,8 +405,34 @@ export async function runSupervisor(deps = {}) {
|
|
|
348
405
|
const restartsSoFar = carriedRestarts({ attention: priorAttention, heartbeat: priorHeartbeat });
|
|
349
406
|
if (restartsSoFar > 0) d.log(`this launch carries ${restartsSoFar} silent restart(s) from the previous run`);
|
|
350
407
|
|
|
408
|
+
// ── (b) THE CARRIED LAUNCH-FAILURE STREAK ────────────────────────────────
|
|
409
|
+
//
|
|
410
|
+
// Read BEFORE the unlink below, for the same reason `carriedRestarts` is: a
|
|
411
|
+
// count held in memory is zero on every launch, because every launch is a
|
|
412
|
+
// fresh process. This one lives in its own file rather than on the attention
|
|
413
|
+
// record, because it has to survive the unlink that clears that record.
|
|
414
|
+
let priorFailures = null;
|
|
415
|
+
try { priorFailures = parseLaunchFailures(d.readFileSync(paths.launchFailuresFile, "utf8")); } catch { priorFailures = null; }
|
|
416
|
+
|
|
351
417
|
try { d.unlinkSync(paths.lastExitFile); } catch { /* none from a previous run */ }
|
|
352
418
|
try { d.unlinkSync(paths.attentionFile); } catch { /* none outstanding */ }
|
|
419
|
+
|
|
420
|
+
// A seat already past the limit re-states the escalation on EVERY launch.
|
|
421
|
+
// The note rides `machine.sessionNote` on the presence beat, and the unlink
|
|
422
|
+
// above just cleared it: without this the org would see the alert blink out
|
|
423
|
+
// each time the seat tried again, which reads as a seat that healed.
|
|
424
|
+
if (priorFailures) {
|
|
425
|
+
const carried = escalationDecision({ streak: priorFailures.streak, limit: d.launchFailureLimit, escalatedAt: priorFailures.escalatedAt });
|
|
426
|
+
if (carried.escalate) {
|
|
427
|
+
writeLaunchAttention({
|
|
428
|
+
d, paths, mux, muxName, now: d.now(),
|
|
429
|
+
streak: carried.streak, limit: carried.limit, signature: priorFailures.signature,
|
|
430
|
+
detail: `${priorFailures.signature} since ${priorFailures.firstAt || "an earlier launch"}`,
|
|
431
|
+
sessionId: record.sessionId, phase: "retrying",
|
|
432
|
+
});
|
|
433
|
+
d.log(`launch-failure escalation still standing: ${carried.reason}`);
|
|
434
|
+
}
|
|
435
|
+
}
|
|
353
436
|
// 4c. An upgrade notice asking for a restart onto the version THIS launch
|
|
354
437
|
// runs is honoured by this launch: retire it, so `maestro session
|
|
355
438
|
// status` stops flagging a hop the session has completed (and a notice
|
|
@@ -380,6 +463,9 @@ export async function runSupervisor(deps = {}) {
|
|
|
380
463
|
// almost certainly waiting on a dialog or a prompt nobody can see.
|
|
381
464
|
const startedAt = Number(d.now());
|
|
382
465
|
let launchFailed = false;
|
|
466
|
+
/** Whatever the mux client itself printed — `screen -D -m` reports a missing
|
|
467
|
+
* command here, and it is the one piece of evidence no pane capture can race. */
|
|
468
|
+
let startStdout = "";
|
|
383
469
|
stopCmd = cmds.stop;
|
|
384
470
|
// ── THE WATCHDOG ────────────────────────────────────────────────────────
|
|
385
471
|
//
|
|
@@ -400,6 +486,10 @@ export async function runSupervisor(deps = {}) {
|
|
|
400
486
|
let clears = 0;
|
|
401
487
|
let restarts = restartsSoFar;
|
|
402
488
|
let watchdogStopped = false;
|
|
489
|
+
/** The last non-empty pane the watchdog saw — evidence for the post-mortem. */
|
|
490
|
+
let lastPane = "";
|
|
491
|
+
/** Did the WATCHDOG end this session? A death we caused is never rotated on. */
|
|
492
|
+
let watchdogRestarted = false;
|
|
403
493
|
let cancelPass = () => {};
|
|
404
494
|
const paneFile = join(agentRoot, "state", "session", "pane-capture.txt");
|
|
405
495
|
|
|
@@ -412,6 +502,17 @@ export async function runSupervisor(deps = {}) {
|
|
|
412
502
|
} catch { return ""; }
|
|
413
503
|
};
|
|
414
504
|
|
|
505
|
+
// The pane is captured ONCE, early, and kept. A resume whose conversation is
|
|
506
|
+
// gone prints "No conversation found with session ID: <id>" and exits inside
|
|
507
|
+
// a couple of seconds, taking the mux session and its pane with it — so the
|
|
508
|
+
// only moment this evidence exists is while the doomed session is still up.
|
|
509
|
+
// Best-effort: an empty capture is not an absence of fault, it is an absence
|
|
510
|
+
// of evidence, and `classifyLaunchFailure` is written to say so.
|
|
511
|
+
let earlyPane = "";
|
|
512
|
+
d.afterPaneProbe(d.earlyPaneMs, () => {
|
|
513
|
+
capturePane().then((text) => { if (!earlyPane) earlyPane = paneTail(text); }).catch(() => {});
|
|
514
|
+
});
|
|
515
|
+
|
|
415
516
|
const sendKeys = async (keys) => {
|
|
416
517
|
for (const k of keys) {
|
|
417
518
|
const cmd = sendKeyCommand(mux, muxName, k);
|
|
@@ -433,6 +534,7 @@ export async function runSupervisor(deps = {}) {
|
|
|
433
534
|
if (!silence.silent) { schedulePass(); return; }
|
|
434
535
|
|
|
435
536
|
const pane = paneTail(await capturePane());
|
|
537
|
+
if (pane) lastPane = pane;
|
|
436
538
|
const seen = classifyPane(pane);
|
|
437
539
|
const act = watchdogAction({ silent: true, kind: seen.kind, keys: seen.keys, clears, restarts });
|
|
438
540
|
|
|
@@ -459,6 +561,7 @@ export async function runSupervisor(deps = {}) {
|
|
|
459
561
|
await sendKeys(seen.keys);
|
|
460
562
|
} else if (act.act === "restart") {
|
|
461
563
|
restarts = nextRestarts;
|
|
564
|
+
watchdogRestarted = true;
|
|
462
565
|
d.log(`session ${muxName}: restarting (${restarts}) — the session is not answering and the screen cannot be cleared safely`);
|
|
463
566
|
try { await d.execFile(cmds.stop.file, cmds.stop.args, { cwd: agentRoot, env: d.env }); } catch { /* the relaunch loop handles a dead mux */ }
|
|
464
567
|
watchdogStopped = true; // the supervisor's own loop relaunches; do not fight it
|
|
@@ -485,6 +588,7 @@ export async function runSupervisor(deps = {}) {
|
|
|
485
588
|
// env file; the mux client itself gets the supervisor's env, as for claude.
|
|
486
589
|
if (envFile) d.writePrivateFile(envFile, renderEnvFile(spec.env));
|
|
487
590
|
const r = await d.execFile(cmds.start.file, cmds.start.args, { cwd: agentRoot, env: d.env });
|
|
591
|
+
startStdout = `${String((r && r.stdout) || "")}${String((r && r.stderr) || "")}`.trim();
|
|
488
592
|
if (r && r.code !== 0) { launchFailed = true; d.log(`${mux.kind} start exited ${r.code}${r.stderr ? `: ${String(r.stderr).trim()}` : ""}`); }
|
|
489
593
|
else if (!cmds.blocking) await waitForProbe(cmds.waitProbe, d);
|
|
490
594
|
} catch (err) {
|
|
@@ -506,11 +610,69 @@ export async function runSupervisor(deps = {}) {
|
|
|
506
610
|
d.log(`session ${muxName} stopped on ${stopping} — id kept, exit ${EXIT_RELAUNCH}`);
|
|
507
611
|
return EXIT_RELAUNCH;
|
|
508
612
|
}
|
|
509
|
-
|
|
613
|
+
|
|
614
|
+
// Did THIS launch ever report in? The single most useful fact about a run,
|
|
615
|
+
// and the one the ten-minute relaunch loop never asked: James's seat looked
|
|
616
|
+
// up for days and had not beaten once.
|
|
617
|
+
let endHeartbeat = null;
|
|
618
|
+
try { endHeartbeat = parseHeartbeat(d.readFileSync(paths.heartbeatFile, "utf8")); } catch { endHeartbeat = null; }
|
|
619
|
+
const endBeatMs = endHeartbeat && typeof endHeartbeat.ts === "string" ? Date.parse(endHeartbeat.ts) : NaN;
|
|
620
|
+
const beatSeen = Number.isFinite(endBeatMs) && endBeatMs >= startedAt;
|
|
621
|
+
|
|
622
|
+
// What the runtime itself said, if we caught it. `earlyPane` is the capture
|
|
623
|
+
// taken while a fast-failing session was still up; `lastPane` is the
|
|
624
|
+
// watchdog's. `startStdout` is whatever the mux client printed to us.
|
|
625
|
+
const evidence = [startStdout, earlyPane, lastPane].filter(Boolean).join("\n");
|
|
626
|
+
const failure = classifyLaunchFailure({ mode, exitCode, output: evidence, beatSeen });
|
|
627
|
+
if (failure.kind !== "none") d.log(`launch verdict: ${failure.kind}${failure.proven ? " (proven)" : ""} — ${failure.detail}`);
|
|
628
|
+
|
|
629
|
+
// ── (b) N IDENTICAL FAILURES ARE A FAULT, NOT N RETRIES ──────────────────
|
|
630
|
+
//
|
|
631
|
+
// The ten-minute backoff was the right instinct and the wrong outcome: every
|
|
632
|
+
// attempt failed identically and the seat stayed quiet about it for days,
|
|
633
|
+
// because the only record was a log on a machine nobody can reach. So the
|
|
634
|
+
// streak is counted across supervisor lifetimes and, at the limit, written
|
|
635
|
+
// where it LEAVES the machine — `state/session/attention.json` rides the
|
|
636
|
+
// presence beat as `machine.sessionNote`, which is the org's view of this
|
|
637
|
+
// seat. A watchdog restart is not counted here: that path has its own
|
|
638
|
+
// bounded budget (`carriedRestarts`) and counting it twice would escalate a
|
|
639
|
+
// fault the seat is already handling.
|
|
640
|
+
let escalation = { escalate: false, fresh: false, streak: 0, limit: d.launchFailureLimit, reason: "" };
|
|
641
|
+
if (!watchdogRestarted) {
|
|
642
|
+
const signature = launchFailureSignature({ mode, kind: failure.kind, exitCode });
|
|
643
|
+
const next = launchFailureStreak(priorFailures, { signature, at: new Date(Number(d.now())).toISOString() });
|
|
644
|
+
if (!next) {
|
|
645
|
+
// The launch worked. Evidence of work clears the record — the same reset
|
|
646
|
+
// rule `carriedRestarts` uses, and for the same reason.
|
|
647
|
+
try { d.unlinkSync(paths.launchFailuresFile); } catch { /* nothing carried */ }
|
|
648
|
+
} else {
|
|
649
|
+
escalation = escalationDecision({ streak: next.streak, limit: d.launchFailureLimit, escalatedAt: next.escalatedAt });
|
|
650
|
+
if (escalation.escalate && !next.escalatedAt) next.escalatedAt = new Date(Number(d.now())).toISOString();
|
|
651
|
+
try { d.writeJsonAtomic(paths.launchFailuresFile, next); } catch { /* the log line still says it */ }
|
|
652
|
+
d.log(`launch failure ${next.streak}× in a row (${signature}) — ${escalation.reason}`);
|
|
653
|
+
if (escalation.escalate) {
|
|
654
|
+
writeLaunchAttention({
|
|
655
|
+
d, paths, mux, muxName, now: d.now(),
|
|
656
|
+
streak: next.streak, limit: escalation.limit, signature,
|
|
657
|
+
detail: failure.detail, sessionId: launched.sessionId, phase: "failed",
|
|
658
|
+
});
|
|
659
|
+
if (escalation.fresh) d.log(`escalating to the org: ${escalationHint({ signature, streak: next.streak, limit: escalation.limit, detail: failure.detail, sessionId: launched.sessionId })}`);
|
|
660
|
+
}
|
|
661
|
+
}
|
|
662
|
+
}
|
|
663
|
+
|
|
664
|
+
const decision = rotationDecision({
|
|
665
|
+
record: launched, mode, exitCode, startedAt, endedAt, now: d.now,
|
|
666
|
+
resumeTargetMissing: failure.kind === "resume-target-missing",
|
|
667
|
+
beatSeen,
|
|
668
|
+
streakAtLimit: escalation.escalate,
|
|
669
|
+
selfStopped: watchdogRestarted,
|
|
670
|
+
});
|
|
510
671
|
if (decision.action === "rotate") {
|
|
511
|
-
const
|
|
672
|
+
const why = decision.why || "the resume failed";
|
|
673
|
+
const rotated = rotateMainSession(launched, { now: d.now, uuid: d.uuid, reason: why });
|
|
512
674
|
const res = saveMainSession(agentRoot, rotated, fsDeps);
|
|
513
|
-
d.log(`resume of ${launched.sessionId}
|
|
675
|
+
d.log(`resume of ${launched.sessionId}: ${why}${decision.proven ? " (proven, so not rationed by the rotation budget)" : ""} — invalidated it and rotated to ${rotated.sessionId}${res.ok ? "" : ` (not saved: ${res.error})`}`);
|
|
514
676
|
} else if (decision.action === "backoff") {
|
|
515
677
|
d.log(`rotation budget spent for this hour — sleeping ${decision.sleepMs / 60000} min before relaunch`);
|
|
516
678
|
await d.sleep(decision.sleepMs);
|
|
@@ -518,6 +680,37 @@ export async function runSupervisor(deps = {}) {
|
|
|
518
680
|
return EXIT_RELAUNCH;
|
|
519
681
|
}
|
|
520
682
|
|
|
683
|
+
/**
|
|
684
|
+
* Write the escalation where it LEAVES THE MACHINE.
|
|
685
|
+
*
|
|
686
|
+
* `state/session/attention.json` is read by `lib/telemetry/collect#sessionNote`
|
|
687
|
+
* and rides the presence beat to hq as `machine.sessionNote` — the only channel
|
|
688
|
+
* out of a seat that accepts no ssh. The record is shaped exactly like the
|
|
689
|
+
* watchdog's (`first-run#attentionRecord` fields) so the existing reader needs
|
|
690
|
+
* no special case; `restarts: 0` is honest and, per `carriedRestarts`, cannot
|
|
691
|
+
* steal budget from the watchdog.
|
|
692
|
+
*
|
|
693
|
+
* @param {object} a
|
|
694
|
+
*/
|
|
695
|
+
export function writeLaunchAttention(a) {
|
|
696
|
+
const { d, paths, mux, muxName } = a;
|
|
697
|
+
const attach = mux && mux.kind === "tmux" ? `tmux attach -t =${muxName}` : `screen -r ${muxName}`;
|
|
698
|
+
const record = {
|
|
699
|
+
reason: "launch-failing",
|
|
700
|
+
since: new Date(Number(a.now)).toISOString(),
|
|
701
|
+
runMs: 0,
|
|
702
|
+
restarts: 0,
|
|
703
|
+
action: a.phase === "retrying" ? "wait" : "give-up",
|
|
704
|
+
attach,
|
|
705
|
+
signature: a.signature,
|
|
706
|
+
streak: a.streak,
|
|
707
|
+
limit: a.limit,
|
|
708
|
+
hint: escalationHint({ signature: a.signature, streak: a.streak, limit: a.limit, detail: a.detail, sessionId: a.sessionId }),
|
|
709
|
+
};
|
|
710
|
+
try { d.writeJsonAtomic(paths.attentionFile, record); } catch { /* the log line still says it */ }
|
|
711
|
+
return record;
|
|
712
|
+
}
|
|
713
|
+
|
|
521
714
|
const isMain = (() => {
|
|
522
715
|
try { return process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href; } catch { return false; }
|
|
523
716
|
})();
|