@cohortapp/agent-sdk 2.18.13 → 2.18.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/maestro.mjs +38 -1
- package/docs/runbooks/fleet-rollout.md +58 -7
- package/docs/runbooks/recovery-and-failover.md +18 -0
- package/lib/assurance/batch.mjs +353 -0
- package/lib/assurance/first-reply.mjs +423 -0
- package/lib/assurance/notice-voice.mjs +357 -0
- package/lib/assurance/plan-note.mjs +43 -0
- package/lib/assurance/room-budget.mjs +55 -6
- package/lib/cadence-failure-class.mjs +245 -0
- package/lib/claude-bin.mjs +26 -7
- package/lib/cli/doctor-checks.mjs +149 -1
- package/lib/comms/send-gate.mjs +59 -0
- package/lib/diagnostics/alerts.mjs +33 -0
- package/lib/engine/agents/usage.mjs +45 -0
- package/lib/engine/budget.mjs +293 -29
- package/lib/engine/cli.mjs +54 -5
- package/lib/engine/loop.mjs +30 -0
- package/lib/engine/output/json.mjs +26 -0
- package/lib/engine/wire/errors.mjs +179 -0
- package/lib/engine/wire/search.mjs +44 -8
- package/lib/identity/persona.mjs +31 -2
- package/lib/org/quota.mjs +27 -0
- package/lib/session/config.mjs +4 -0
- package/lib/session/identity.mjs +71 -7
- package/lib/session/launch-failure.mjs +251 -0
- package/lib/session/resume-target.mjs +86 -0
- package/lib/telemetry/alerts.mjs +94 -0
- package/lib/telemetry/collect.mjs +155 -2
- package/lib/upgrade/pinned-drift.mjs +467 -0
- package/package.json +1 -1
- package/scaffold/config/alerts.yaml +7 -0
- package/scripts/ci/check-cadence-prompts-exist.mjs +96 -0
- package/scripts/ci/check.mjs +3 -0
- package/scripts/daemon/agent-daemon.mjs +75 -5
- package/scripts/daemon/assurance.mjs +709 -44
- package/scripts/daemon/cadence-consumer.mjs +281 -34
- package/scripts/daemon/deliver.mjs +109 -0
- package/scripts/daemon/dispatcher.mjs +21 -3
- package/scripts/daemon/inbox-deferral.mjs +102 -9
- package/scripts/daemon/session-lock.mjs +41 -1
- package/scripts/emergency-stop.sh +114 -13
- package/scripts/fleet/rollout.mjs +256 -10
- package/scripts/healthcheck.sh +131 -33
- package/scripts/local-triggers/autoupdate.sh +144 -11
- package/scripts/resume-operations.sh +101 -6
- package/scripts/session/supervisor.mjs +198 -5
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lib/session/launch-failure.mjs — WHY a main-session launch failed, and when
|
|
3
|
+
* the seat should stop trying quietly and say so. Pure; no clock, no fs.
|
|
4
|
+
*
|
|
5
|
+
* ── THE FAULT THIS MODULE EXISTS TO NOT HAVE ────────────────────────────────
|
|
6
|
+
* Measured on James Kirkland's seat, 2026-09-25. The supervisor relaunched
|
|
7
|
+
* every ten minutes for DAYS running
|
|
8
|
+
*
|
|
9
|
+
* claude --resume e953a33e-97a0-4d13-9fd4-6a77cd7a68c9
|
|
10
|
+
* No conversation found with session ID: e953a33e-…
|
|
11
|
+
* exit 1, after ~2 s
|
|
12
|
+
*
|
|
13
|
+
* for a conversation that no longer existed on that machine. Every layer
|
|
14
|
+
* behaved "correctly": launchd relaunched a job that exited non-zero, the
|
|
15
|
+
* rotation budget throttled to one attempt per ten minutes, the log recorded
|
|
16
|
+
* each failure. The seat looked UP the whole time — a live launchd job, a
|
|
17
|
+
* supervisor process, a mux session flickering into existence — and never
|
|
18
|
+
* beat once. Nobody reads that log; nobody could ssh in to read it.
|
|
19
|
+
*
|
|
20
|
+
* Two things were missing and both are here:
|
|
21
|
+
*
|
|
22
|
+
* 1. THE FAILURE HAS A NAME AND THE NAME WAS ON THE SCREEN. "No conversation
|
|
23
|
+
* found with session ID" is not a symptom to infer from timing — it is the
|
|
24
|
+
* runtime telling us the resume target is gone. {@link classifyLaunchFailure}
|
|
25
|
+
* reads it, so the repair (mint a fresh id) is taken on evidence rather
|
|
26
|
+
* than on the 20-second heuristic in `identity#rotationDecision`, which
|
|
27
|
+
* misses a failure that takes 21 seconds and cannot distinguish a dead
|
|
28
|
+
* transcript from a missing binary.
|
|
29
|
+
*
|
|
30
|
+
* 2. N IDENTICAL FAILURES ARE NOT N CHANCES — they are one fault, repeated.
|
|
31
|
+
* A launch that fails the same way three times in a row will fail the
|
|
32
|
+
* fourth time too, and the only useful next step is to tell somebody who
|
|
33
|
+
* is not on this machine. {@link launchFailureStreak} counts consecutive
|
|
34
|
+
* failures WITH THE SAME SIGNATURE across supervisor lifetimes (each
|
|
35
|
+
* launch is a fresh process, so an in-memory counter is always zero —
|
|
36
|
+
* the same trap `first-run#carriedRestarts` exists for), and
|
|
37
|
+
* {@link escalationDecision} says when it stops being a retry and becomes
|
|
38
|
+
* an escalation.
|
|
39
|
+
*
|
|
40
|
+
* The signature deliberately does NOT include the session id: rotating to a
|
|
41
|
+
* fresh id and failing the same way again is the SAME fault (that is exactly
|
|
42
|
+
* what a missing binary does), and a signature that changed every rotation
|
|
43
|
+
* would reset the streak forever — which is how this failure stayed invisible.
|
|
44
|
+
*
|
|
45
|
+
* @module lib/session/launch-failure
|
|
46
|
+
*/
|
|
47
|
+
|
|
48
|
+
"use strict";
|
|
49
|
+
|
|
50
|
+
/** Consecutive identical launch failures before the seat escalates. */
|
|
51
|
+
export const IDENTICAL_FAILURE_LIMIT = 3;
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* The runtime's own words for "the conversation you asked me to resume is not
|
|
55
|
+
* on this machine". Matched case-insensitively against the tail of the pane /
|
|
56
|
+
* the launcher's stderr. Kept as a list because the wording is the CLI's, not
|
|
57
|
+
* ours, and a second phrasing must be addable without touching the classifier.
|
|
58
|
+
*/
|
|
59
|
+
export const RESUME_MISSING_PATTERNS = Object.freeze([
|
|
60
|
+
/no conversation found with session id/i,
|
|
61
|
+
/no conversation found matching/i,
|
|
62
|
+
/session .{0,80}? not found/i,
|
|
63
|
+
]);
|
|
64
|
+
|
|
65
|
+
/** "the binary is not there" — a shell's 127, and the two spellings of ENOENT. */
|
|
66
|
+
export const BINARY_MISSING_PATTERNS = Object.freeze([
|
|
67
|
+
/command not found/i,
|
|
68
|
+
/no such file or directory/i,
|
|
69
|
+
/\bENOENT\b/,
|
|
70
|
+
]);
|
|
71
|
+
|
|
72
|
+
const anyMatch = (patterns, text) => patterns.some((re) => re.test(text));
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Pure: name the failure of one launch.
|
|
76
|
+
*
|
|
77
|
+
* `output` is whatever text the launch left behind — a pane capture, the
|
|
78
|
+
* launcher's stderr, or "" when nothing was captured. It is EVIDENCE, and its
|
|
79
|
+
* absence is never treated as proof of anything: with no output the verdict
|
|
80
|
+
* falls back to what the exit code and the mode can carry on their own.
|
|
81
|
+
*
|
|
82
|
+
* @param {object} a
|
|
83
|
+
* @param {"session-id"|"resume"} a.mode how this launch addressed the session
|
|
84
|
+
* @param {number|null} a.exitCode the recorded exit (null = the exit file was never written)
|
|
85
|
+
* @param {string} [a.output] pane tail / stderr, may be ""
|
|
86
|
+
* @param {boolean} [a.beatSeen] did this launch produce a heartbeat?
|
|
87
|
+
* @returns {{kind:"none"|"resume-target-missing"|"binary-missing"|"never-beaten"|"nonzero-exit", proven:boolean, detail:string}}
|
|
88
|
+
*/
|
|
89
|
+
export function classifyLaunchFailure(a = {}) {
|
|
90
|
+
const exitCode = a.exitCode === null || a.exitCode === undefined ? null : Number(a.exitCode);
|
|
91
|
+
const output = typeof a.output === "string" ? a.output : "";
|
|
92
|
+
const beatSeen = a.beatSeen === true;
|
|
93
|
+
const failed = exitCode === null || exitCode !== 0;
|
|
94
|
+
|
|
95
|
+
// A launch that beat and then exited 0 is a session that RAN. Nothing to name.
|
|
96
|
+
if (!failed && beatSeen) return { kind: "none", proven: true, detail: "the session ran and exited cleanly" };
|
|
97
|
+
if (!failed) return { kind: "none", proven: false, detail: "the session exited cleanly without beating" };
|
|
98
|
+
|
|
99
|
+
// PROVEN verdicts first: the runtime said what was wrong, in words.
|
|
100
|
+
if (a.mode === "resume" && anyMatch(RESUME_MISSING_PATTERNS, output)) {
|
|
101
|
+
return {
|
|
102
|
+
kind: "resume-target-missing",
|
|
103
|
+
proven: true,
|
|
104
|
+
detail: "the runtime reported that the conversation this launch tried to resume does not exist on this machine",
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
if (exitCode === 127 || anyMatch(BINARY_MISSING_PATTERNS, output)) {
|
|
108
|
+
return {
|
|
109
|
+
kind: "binary-missing",
|
|
110
|
+
proven: exitCode === 127 || anyMatch(BINARY_MISSING_PATTERNS, output),
|
|
111
|
+
detail: "the launcher could not find the binary it was told to run",
|
|
112
|
+
};
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// INFERRED verdicts. `never-beaten` is the honest name for James's seat when
|
|
116
|
+
// the pane was not captured: the launch ended non-zero having never reported
|
|
117
|
+
// in, so whatever it was, it was never a working session.
|
|
118
|
+
if (!beatSeen) {
|
|
119
|
+
return { kind: "never-beaten", proven: false, detail: `the launch ended (exit ${exitCode === null ? "unknown" : exitCode}) without ever writing a heartbeat` };
|
|
120
|
+
}
|
|
121
|
+
return { kind: "nonzero-exit", proven: false, detail: `the session beat, then ended with exit ${exitCode === null ? "unknown" : exitCode}` };
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Pure: the stable identity of a failure, for counting repeats.
|
|
126
|
+
*
|
|
127
|
+
* Mode and kind only. NOT the session id (see the module docblock) and not the
|
|
128
|
+
* exit code when the kind already names the fault — otherwise a flapping exit
|
|
129
|
+
* status would read as a different fault each time and never accumulate.
|
|
130
|
+
*
|
|
131
|
+
* @param {{mode:string, kind:string, exitCode?:number|null}} a
|
|
132
|
+
* @returns {string} e.g. `resume:resume-target-missing`
|
|
133
|
+
*/
|
|
134
|
+
export function launchFailureSignature(a = {}) {
|
|
135
|
+
const kind = String(a.kind || "unknown");
|
|
136
|
+
const mode = a.mode === "resume" ? "resume" : "session-id";
|
|
137
|
+
if (kind === "none") return "";
|
|
138
|
+
if (kind === "never-beaten" || kind === "nonzero-exit") {
|
|
139
|
+
const exit = a.exitCode === null || a.exitCode === undefined ? "unknown" : String(Number(a.exitCode));
|
|
140
|
+
return `${mode}:${kind}:${exit}`;
|
|
141
|
+
}
|
|
142
|
+
return `${mode}:${kind}`;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Pure: parse `state/session/launch-failures.json`. Anything unreadable is
|
|
147
|
+
* "no streak" — a corrupt counter must never be able to escalate on its own.
|
|
148
|
+
* @param {string|null|undefined} text
|
|
149
|
+
* @returns {{signature:string, streak:number, firstAt:string, lastAt:string, escalatedAt:string}|null}
|
|
150
|
+
*/
|
|
151
|
+
export function parseLaunchFailures(text) {
|
|
152
|
+
if (typeof text !== "string" || !text.trim()) return null;
|
|
153
|
+
let raw;
|
|
154
|
+
try { raw = JSON.parse(text); } catch { return null; }
|
|
155
|
+
if (!raw || typeof raw !== "object" || Array.isArray(raw)) return null;
|
|
156
|
+
const signature = typeof raw.signature === "string" ? raw.signature : "";
|
|
157
|
+
const streak = Number.isInteger(raw.streak) && raw.streak > 0 ? raw.streak : 0;
|
|
158
|
+
if (!signature || !streak) return null;
|
|
159
|
+
return {
|
|
160
|
+
signature,
|
|
161
|
+
streak,
|
|
162
|
+
firstAt: typeof raw.firstAt === "string" ? raw.firstAt : "",
|
|
163
|
+
lastAt: typeof raw.lastAt === "string" ? raw.lastAt : "",
|
|
164
|
+
escalatedAt: typeof raw.escalatedAt === "string" ? raw.escalatedAt : "",
|
|
165
|
+
};
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* Pure: the streak after one more launch outcome.
|
|
170
|
+
*
|
|
171
|
+
* A DIFFERENT signature restarts the count at 1 rather than adding to it: two
|
|
172
|
+
* different faults in a row are two problems, and escalating on their sum
|
|
173
|
+
* would cry wolf. A successful launch (`signature` empty) clears the record
|
|
174
|
+
* entirely — evidence of work, the same reset rule `carriedRestarts` uses.
|
|
175
|
+
*
|
|
176
|
+
* @param {object|null} prior {@link parseLaunchFailures} output
|
|
177
|
+
* @param {{signature:string, at:string}} outcome
|
|
178
|
+
* @returns {{signature:string, streak:number, firstAt:string, lastAt:string, escalatedAt:string}|null} null = clear the file
|
|
179
|
+
*/
|
|
180
|
+
export function launchFailureStreak(prior, outcome = {}) {
|
|
181
|
+
const signature = typeof outcome.signature === "string" ? outcome.signature : "";
|
|
182
|
+
const at = typeof outcome.at === "string" ? outcome.at : "";
|
|
183
|
+
if (!signature) return null; // the launch worked: nothing to carry
|
|
184
|
+
if (prior && prior.signature === signature) {
|
|
185
|
+
return { signature, streak: prior.streak + 1, firstAt: prior.firstAt || at, lastAt: at, escalatedAt: prior.escalatedAt || "" };
|
|
186
|
+
}
|
|
187
|
+
return { signature, streak: 1, firstAt: at, lastAt: at, escalatedAt: "" };
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/**
|
|
191
|
+
* Pure: has this stopped being a retry?
|
|
192
|
+
*
|
|
193
|
+
* `escalate` is true on the launch that REACHES the limit and on every launch
|
|
194
|
+
* after it, because the escalation channel is a beat field that the next
|
|
195
|
+
* supervisor run clears (`attention.json` is unlinked at launch): a one-shot
|
|
196
|
+
* escalation would flicker off and the org would see a seat that healed. The
|
|
197
|
+
* `fresh` flag distinguishes the first crossing, which is the one worth a log
|
|
198
|
+
* line and a distinct wording.
|
|
199
|
+
*
|
|
200
|
+
* @param {{streak:number, limit?:number, escalatedAt?:string}} a
|
|
201
|
+
* @returns {{escalate:boolean, fresh:boolean, streak:number, limit:number, reason:string}}
|
|
202
|
+
*/
|
|
203
|
+
export function escalationDecision(a = {}) {
|
|
204
|
+
const limit = Number.isInteger(a.limit) && a.limit > 0 ? a.limit : IDENTICAL_FAILURE_LIMIT;
|
|
205
|
+
const streak = Number.isInteger(a.streak) && a.streak > 0 ? a.streak : 0;
|
|
206
|
+
if (streak < limit) {
|
|
207
|
+
return { escalate: false, fresh: false, streak, limit, reason: `launch failure ${streak}/${limit} — retrying` };
|
|
208
|
+
}
|
|
209
|
+
const fresh = !a.escalatedAt;
|
|
210
|
+
return {
|
|
211
|
+
escalate: true,
|
|
212
|
+
fresh,
|
|
213
|
+
streak,
|
|
214
|
+
limit,
|
|
215
|
+
reason: `${streak} consecutive launches have failed the same way (limit ${limit}) — this is a fault, not a retry`,
|
|
216
|
+
};
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* Pure: the one sentence an operator in a browser gets, via the beat's
|
|
221
|
+
* `machine.sessionNote.detail`.
|
|
222
|
+
*
|
|
223
|
+
* THE VERDICT LEADS, THE EXPLANATION TRAILS — because the channel TRUNCATES.
|
|
224
|
+
* `telemetry/collect#sanitizeNoteDetail` caps the detail at 200 characters and
|
|
225
|
+
* appends an ellipsis, so anything after the first sentence may never leave the
|
|
226
|
+
* machine. The first draft of this line ended "…this needs a person" and that
|
|
227
|
+
* clause — the only part that asks for an action — was the part that got cut.
|
|
228
|
+
* So: how many, that retries are not working, then the diagnosis.
|
|
229
|
+
*
|
|
230
|
+
* @param {{signature:string, streak:number, limit?:number, detail:string, sessionId?:string}} a
|
|
231
|
+
* @returns {string}
|
|
232
|
+
*/
|
|
233
|
+
export function escalationHint(a = {}) {
|
|
234
|
+
const sig = String(a.signature || "unknown");
|
|
235
|
+
const streak = Number(a.streak) || 0;
|
|
236
|
+
const detail = String(a.detail || "").trim();
|
|
237
|
+
const id = typeof a.sessionId === "string" && a.sessionId ? ` Session ${a.sessionId}.` : "";
|
|
238
|
+
return `${streak}× in a row, identically — retries are not fixing it and this needs a person. ${detail || sig}.${id}`;
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
export default {
|
|
242
|
+
IDENTICAL_FAILURE_LIMIT,
|
|
243
|
+
RESUME_MISSING_PATTERNS,
|
|
244
|
+
BINARY_MISSING_PATTERNS,
|
|
245
|
+
classifyLaunchFailure,
|
|
246
|
+
launchFailureSignature,
|
|
247
|
+
parseLaunchFailures,
|
|
248
|
+
launchFailureStreak,
|
|
249
|
+
escalationDecision,
|
|
250
|
+
escalationHint,
|
|
251
|
+
};
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lib/session/resume-target.mjs — is the conversation we are about to `--resume`
|
|
3
|
+
* actually on this machine?
|
|
4
|
+
*
|
|
5
|
+
* ── WHY THIS IS A PREFLIGHT AND NOT A POST-MORTEM ───────────────────────────
|
|
6
|
+
* The supervisor already had a post-mortem: a resume that died non-zero inside
|
|
7
|
+
* twenty seconds rotates to a fresh id (`identity#rotationDecision`). It is a
|
|
8
|
+
* good heuristic and it is not enough — it cannot tell a dead transcript from a
|
|
9
|
+
* missing binary, it misses a failure that takes twenty-one seconds, and every
|
|
10
|
+
* one of its verdicts costs a full launch, a mux session and a launchd cycle to
|
|
11
|
+
* reach. On James Kirkland's seat that cycle ran every ten minutes for days
|
|
12
|
+
* against a session id that had not existed for just as long.
|
|
13
|
+
*
|
|
14
|
+
* The answer is on disk before we launch. Claude Code keeps each project's
|
|
15
|
+
* transcripts as `~/.claude/projects/<project-slug>/<session-id>.jsonl`, where
|
|
16
|
+
* the slug is the project directory with every non-alphanumeric character
|
|
17
|
+
* replaced by `-`. `--resume` is project-scoped, so the seat's own slug is the
|
|
18
|
+
* only place the id could be.
|
|
19
|
+
*
|
|
20
|
+
* ── FAIL-OPEN, DELIBERATELY ─────────────────────────────────────────────────
|
|
21
|
+
* A wrong "missing" throws away a live transcript; a wrong "unknown" costs one
|
|
22
|
+
* launch and lands in the post-mortem that was already there. So `missing` is
|
|
23
|
+
* returned ONLY from positive evidence: the project directory exists, it holds
|
|
24
|
+
* transcripts, and this id is not among them. An absent directory, an empty
|
|
25
|
+
* one, an unreadable one, an engine that does not use this layout at all — all
|
|
26
|
+
* `unknown`, and the launch proceeds exactly as before.
|
|
27
|
+
*
|
|
28
|
+
* One I/O shell (`readdirSync`/`existsSync`), both injected.
|
|
29
|
+
*
|
|
30
|
+
* @module lib/session/resume-target
|
|
31
|
+
*/
|
|
32
|
+
|
|
33
|
+
"use strict";
|
|
34
|
+
|
|
35
|
+
import { readdirSync as fsReaddirSync } from "node:fs";
|
|
36
|
+
import { join } from "node:path";
|
|
37
|
+
|
|
38
|
+
/** Where Claude Code keeps per-project transcripts, relative to the home dir. */
|
|
39
|
+
export const TRANSCRIPT_ROOT = join(".claude", "projects");
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Pure: Claude Code's directory name for a project path — every character that
|
|
43
|
+
* is not a letter or a digit becomes `-` (`/Users/x/hq` → `-Users-x-hq`).
|
|
44
|
+
* @param {string} projectDir
|
|
45
|
+
* @returns {string}
|
|
46
|
+
*/
|
|
47
|
+
export function projectSlug(projectDir) {
|
|
48
|
+
return String(projectDir || "").replace(/[^A-Za-z0-9]/g, "-");
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Is `sessionId`'s transcript present for `projectDir`?
|
|
53
|
+
*
|
|
54
|
+
* @param {object} a
|
|
55
|
+
* @param {string} a.homeDir the user's home directory
|
|
56
|
+
* @param {string} a.projectDir the directory the session runs in (the agent root)
|
|
57
|
+
* @param {string} a.sessionId the id this launch would resume
|
|
58
|
+
* @param {"claude"|"cohort"} [a.engine] a non-claude engine does not use this layout → always "unknown"
|
|
59
|
+
* @param {{readdirSync?:Function}} [deps]
|
|
60
|
+
* @returns {{state:"present"|"missing"|"unknown", dir:string, reason:string}}
|
|
61
|
+
*/
|
|
62
|
+
export function resumeTargetState(a = {}, deps = {}) {
|
|
63
|
+
const readdirSync = deps.readdirSync || fsReaddirSync;
|
|
64
|
+
const homeDir = typeof a.homeDir === "string" ? a.homeDir : "";
|
|
65
|
+
const sessionId = String(a.sessionId || "").toLowerCase();
|
|
66
|
+
if (a.engine && a.engine !== "claude") return { state: "unknown", dir: "", reason: `engine ${a.engine} does not keep transcripts here` };
|
|
67
|
+
if (!homeDir || !sessionId) return { state: "unknown", dir: "", reason: "no home directory or session id to check" };
|
|
68
|
+
const dir = join(homeDir, TRANSCRIPT_ROOT, projectSlug(a.projectDir));
|
|
69
|
+
let entries;
|
|
70
|
+
try { entries = readdirSync(dir); } catch { return { state: "unknown", dir, reason: "the project's transcript directory is absent or unreadable" }; }
|
|
71
|
+
const names = (Array.isArray(entries) ? entries : []).map((e) => String(e && e.name ? e.name : e));
|
|
72
|
+
const transcripts = names.filter((n) => n.toLowerCase().endsWith(".jsonl"));
|
|
73
|
+
if (transcripts.some((n) => n.toLowerCase() === `${sessionId}.jsonl`)) {
|
|
74
|
+
return { state: "present", dir, reason: "the transcript is on disk" };
|
|
75
|
+
}
|
|
76
|
+
// An EMPTY directory proves nothing: it is equally what a seat looks like the
|
|
77
|
+
// moment before its first session writes, and rotating on it would be a guess.
|
|
78
|
+
if (transcripts.length === 0) return { state: "unknown", dir, reason: "the project's transcript directory holds no transcripts at all" };
|
|
79
|
+
return {
|
|
80
|
+
state: "missing",
|
|
81
|
+
dir,
|
|
82
|
+
reason: `${transcripts.length} transcript(s) on disk for this project and none of them is ${sessionId}`,
|
|
83
|
+
};
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
export default { TRANSCRIPT_ROOT, projectSlug, resumeTargetState };
|
package/lib/telemetry/alerts.mjs
CHANGED
|
@@ -19,6 +19,10 @@
|
|
|
19
19
|
* - subAgentsRunning over the cap → INFO (more children than expected).
|
|
20
20
|
* - machine.sessionNote present → WARNING/CRITICAL (frontdoor) — the
|
|
21
21
|
* seat's front door is not answering.
|
|
22
|
+
* - machine.upgrade.failStreak >= N → WARNING/CRITICAL (rollout) — N
|
|
23
|
+
* consecutive attempts to reach
|
|
24
|
+
* @latest have failed; the seat is
|
|
25
|
+
* stranded on old code.
|
|
22
26
|
*
|
|
23
27
|
* Output: a deterministic array `[{ id, severity, kind, detail }, ...]` (the
|
|
24
28
|
* snapshot's alert shape) sorted critical→warning→info then by id, so identical
|
|
@@ -90,6 +94,30 @@ export const DEFAULT_THRESHOLDS = Object.freeze({
|
|
|
90
94
|
warnMs: 15 * 60 * 1000,
|
|
91
95
|
critMs: 2 * 60 * 60 * 1000,
|
|
92
96
|
}),
|
|
97
|
+
/**
|
|
98
|
+
* How many CONSECUTIVE failed attempts to reach @latest make a seat STUCK.
|
|
99
|
+
*
|
|
100
|
+
* `warnStreak` 3, not 1: one failure is a bad release or a slow mirror and
|
|
101
|
+
* the hourly job is entitled to try again; the failed-target hold already
|
|
102
|
+
* spaces those out. Three separate, completed attempts have failed, which
|
|
103
|
+
* means every fix the seat can apply to itself has been applied three times
|
|
104
|
+
* and the seat is still where it was.
|
|
105
|
+
*
|
|
106
|
+
* `critStreak` 6: past this the seat has been refusing @latest for the best
|
|
107
|
+
* part of a week under the default 24 h hold — which is exactly how long
|
|
108
|
+
* three seats sat on 2.17.0 while every hourly log line was individually
|
|
109
|
+
* correct. A release that cannot reach a colleague is not a machine problem
|
|
110
|
+
* that resolves itself; it waits for a person, like a scope fault.
|
|
111
|
+
*/
|
|
112
|
+
upgradeStuck: Object.freeze({
|
|
113
|
+
// THE ONE DEFINITION of the stuck threshold. scripts/fleet/rollout.mjs
|
|
114
|
+
// imports `warnStreak` as its STUCK_STREAK rather than restating it, and
|
|
115
|
+
// the shell copy (autoupdate.sh) is pinned against this value by a
|
|
116
|
+
// source-reading test — because three literals agreeing by comment is the
|
|
117
|
+
// prose-promise-instead-of-a-check pattern this repo bans.
|
|
118
|
+
warnStreak: 3,
|
|
119
|
+
critStreak: 6,
|
|
120
|
+
}),
|
|
93
121
|
replyDebt: Object.freeze({
|
|
94
122
|
// ONE withheld reply is already worth saying. This is not a resource gauge
|
|
95
123
|
// where a low reading is normal noise — it counts people who asked this
|
|
@@ -175,6 +203,7 @@ export function deriveAlerts(status, thresholds, now = Date.now()) {
|
|
|
175
203
|
push(alerts, frontDoorRule(machine, t.frontDoor, now));
|
|
176
204
|
push(alerts, scopeFaultRule(machine));
|
|
177
205
|
push(alerts, repliesWithheldRule(machine, t.replyDebt));
|
|
206
|
+
push(alerts, upgradeStuckRule(machine, t.upgradeStuck, now));
|
|
178
207
|
|
|
179
208
|
alerts.sort((a, b) => {
|
|
180
209
|
const r = (SEVERITY_RANK[a.severity] ?? 9) - (SEVERITY_RANK[b.severity] ?? 9);
|
|
@@ -282,6 +311,71 @@ function repliesWithheldRule(machine, cfg = {}) {
|
|
|
282
311
|
};
|
|
283
312
|
}
|
|
284
313
|
|
|
314
|
+
/**
|
|
315
|
+
* THIS SEAT CANNOT GET TO @latest, AND HAS PROVED IT N TIMES.
|
|
316
|
+
*
|
|
317
|
+
* Every other rule in this file reads a gauge. This one reads a COUNT of
|
|
318
|
+
* completed failures, because the failure it names has no gauge: on 2026-09-25
|
|
319
|
+
* three seats were found sitting on 2.17.0 while the fleet ran 2.18.13, and
|
|
320
|
+
* nothing anywhere was in an error state. The launchd job fired hourly, npm
|
|
321
|
+
* answered, the install ran, the health gate did its job, the rollback worked,
|
|
322
|
+
* `last.json` recorded the outcome honestly and the beat carried it. Each hour
|
|
323
|
+
* was a correct, self-contained failure — and the org has no way to see the
|
|
324
|
+
* difference between the first one and the fortieth, which is the only thing
|
|
325
|
+
* that distinguishes "a release is settling" from "a colleague is stranded on
|
|
326
|
+
* code from three weeks ago and will never leave it unaided".
|
|
327
|
+
*
|
|
328
|
+
* WHY THIS RIDES THE EXISTING BEAT AND NOTHING ELSE. The seats this rule is
|
|
329
|
+
* about are, by construction, running the OLDEST code in the fleet — so any
|
|
330
|
+
* new channel it might use is the one channel they do not have. `machine.upgrade`
|
|
331
|
+
* is a field their beat already carries; the count is added to it, and this
|
|
332
|
+
* rule turns it into an alert on the seats new enough to run this file.
|
|
333
|
+
*
|
|
334
|
+
* FOR THE ONES THAT ARE NOT, THIS RULE IS STRUCTURALLY BLIND AND SAYING SO IS
|
|
335
|
+
* PART OF THE DESIGN. A seat too stale to install the SDK runs a collector that
|
|
336
|
+
* never writes `failStreak` and an alerts file that has never heard of this
|
|
337
|
+
* rule — so the alert cannot reach exactly the seats it was written for. The
|
|
338
|
+
* fleet-side derivation covers them from outside: `seatStuck()` in
|
|
339
|
+
* scripts/fleet/rollout.mjs reads the same fact out of a 2.17-era beat, and
|
|
340
|
+
* `propagationOutcome()` there returns reason `"stuck"` so that stage 3 of
|
|
341
|
+
* every publish exits non-zero and names the machine. That path is automatic
|
|
342
|
+
* because publishing is; nobody has to decide to go looking.
|
|
343
|
+
*
|
|
344
|
+
* Both answer (d) the same way: a streak counts ATTEMPTS, and a seat that has
|
|
345
|
+
* not been asked has made none.
|
|
346
|
+
*
|
|
347
|
+
* DISTINCT FROM DRIFT. A `stranded` seat (framework paths pinned in
|
|
348
|
+
* `.maestroignore`, `machine.stranded`) cannot RECEIVE parts of a release it
|
|
349
|
+
* did install; a STUCK seat never installs the release at all. Different
|
|
350
|
+
* cause, different fix, different alert id — they can be true at once.
|
|
351
|
+
*/
|
|
352
|
+
function upgradeStuckRule(machine, cfg = {}, now = Date.now()) {
|
|
353
|
+
const u = machine && machine.upgrade;
|
|
354
|
+
if (!u || typeof u !== "object") return null;
|
|
355
|
+
// Strict: the producer (autoupdate.sh via collect.upgradeSummary) writes an
|
|
356
|
+
// integer, and both validate it. A string here means something else wrote
|
|
357
|
+
// this record, and a rule that coerces would be reporting on a shape it does
|
|
358
|
+
// not understand.
|
|
359
|
+
const n = Number.isInteger(u.failStreak) ? u.failStreak : 0;
|
|
360
|
+
if (!(n > 0)) return null;
|
|
361
|
+
const sev = severityFor(n, num(cfg.warnStreak, Infinity), num(cfg.critStreak, Infinity));
|
|
362
|
+
if (!sev) return null;
|
|
363
|
+
const target = typeof u.streakTarget === "string" && u.streakTarget ? u.streakTarget : (typeof u.to === "string" ? u.to : "@latest");
|
|
364
|
+
// A DURATION, not only a tally: four attempts since Monday and four attempts
|
|
365
|
+
// in the last hour are different seats with different urgency, and the count
|
|
366
|
+
// alone cannot tell them apart.
|
|
367
|
+
const sinceMs = typeof u.stuckSince === "string" ? Date.parse(u.stuckSince) : NaN;
|
|
368
|
+
const days = Number.isFinite(sinceMs) ? Math.floor((now - sinceMs) / 86400000) : null;
|
|
369
|
+
const forHow = days === null ? "" : days >= 1 ? ` over ${days} ${days === 1 ? "day" : "days"}` : " today";
|
|
370
|
+
const why = typeof u.reason === "string" && u.reason ? ` Last failure: ${u.reason.split(":")[0]}.` : "";
|
|
371
|
+
return {
|
|
372
|
+
id: "upgrade_stuck",
|
|
373
|
+
severity: sev,
|
|
374
|
+
kind: "rollout",
|
|
375
|
+
detail: `${n} consecutive upgrade attempts to ${target} have failed${forHow} — this seat is running ${typeof u.from === "string" && u.from ? u.from : "older code"} and will not arrive on its own.${why} It needs a person at the machine.`,
|
|
376
|
+
};
|
|
377
|
+
}
|
|
378
|
+
|
|
285
379
|
function subAgentsRule(status, cfg = {}) {
|
|
286
380
|
const v = num(status.subAgentsRunning);
|
|
287
381
|
const cap = num(cfg.infoCap, Infinity);
|