@cohortapp/agent-sdk 2.18.13 → 2.18.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,251 @@
1
+ /**
2
+ * lib/session/launch-failure.mjs — WHY a main-session launch failed, and when
3
+ * the seat should stop trying quietly and say so. Pure; no clock, no fs.
4
+ *
5
+ * ── THE FAULT THIS MODULE EXISTS TO NOT HAVE ────────────────────────────────
6
+ * Measured on James Kirkland's seat, 2026-09-25. The supervisor relaunched
7
+ * every ten minutes for DAYS running
8
+ *
9
+ * claude --resume e953a33e-97a0-4d13-9fd4-6a77cd7a68c9
10
+ * No conversation found with session ID: e953a33e-…
11
+ * exit 1, after ~2 s
12
+ *
13
+ * for a conversation that no longer existed on that machine. Every layer
14
+ * behaved "correctly": launchd relaunched a job that exited non-zero, the
15
+ * rotation budget throttled to one attempt per ten minutes, the log recorded
16
+ * each failure. The seat looked UP the whole time — a live launchd job, a
17
+ * supervisor process, a mux session flickering into existence — and never
18
+ * beat once. Nobody reads that log; nobody could ssh in to read it.
19
+ *
20
+ * Two things were missing and both are here:
21
+ *
22
+ * 1. THE FAILURE HAS A NAME AND THE NAME WAS ON THE SCREEN. "No conversation
23
+ * found with session ID" is not a symptom to infer from timing — it is the
24
+ * runtime telling us the resume target is gone. {@link classifyLaunchFailure}
25
+ * reads it, so the repair (mint a fresh id) is taken on evidence rather
26
+ * than on the 20-second heuristic in `identity#rotationDecision`, which
27
+ * misses a failure that takes 21 seconds and cannot distinguish a dead
28
+ * transcript from a missing binary.
29
+ *
30
+ * 2. N IDENTICAL FAILURES ARE NOT N CHANCES — they are one fault, repeated.
31
+ * A launch that fails the same way three times in a row will fail the
32
+ * fourth time too, and the only useful next step is to tell somebody who
33
+ * is not on this machine. {@link launchFailureStreak} counts consecutive
34
+ * failures WITH THE SAME SIGNATURE across supervisor lifetimes (each
35
+ * launch is a fresh process, so an in-memory counter is always zero —
36
+ * the same trap `first-run#carriedRestarts` exists for), and
37
+ * {@link escalationDecision} says when it stops being a retry and becomes
38
+ * an escalation.
39
+ *
40
+ * The signature deliberately does NOT include the session id: rotating to a
41
+ * fresh id and failing the same way again is the SAME fault (that is exactly
42
+ * what a missing binary does), and a signature that changed every rotation
43
+ * would reset the streak forever — which is how this failure stayed invisible.
44
+ *
45
+ * @module lib/session/launch-failure
46
+ */
47
+
48
+ "use strict";
49
+
50
+ /** Consecutive identical launch failures before the seat escalates. */
51
+ export const IDENTICAL_FAILURE_LIMIT = 3;
52
+
53
+ /**
54
+ * The runtime's own words for "the conversation you asked me to resume is not
55
+ * on this machine". Matched case-insensitively against the tail of the pane /
56
+ * the launcher's stderr. Kept as a list because the wording is the CLI's, not
57
+ * ours, and a second phrasing must be addable without touching the classifier.
58
+ */
59
+ export const RESUME_MISSING_PATTERNS = Object.freeze([
60
+ /no conversation found with session id/i,
61
+ /no conversation found matching/i,
62
+ /session .{0,80}? not found/i,
63
+ ]);
64
+
65
+ /** "the binary is not there" — a shell's 127, and the two spellings of ENOENT. */
66
+ export const BINARY_MISSING_PATTERNS = Object.freeze([
67
+ /command not found/i,
68
+ /no such file or directory/i,
69
+ /\bENOENT\b/,
70
+ ]);
71
+
72
+ const anyMatch = (patterns, text) => patterns.some((re) => re.test(text));
73
+
74
+ /**
75
+ * Pure: name the failure of one launch.
76
+ *
77
+ * `output` is whatever text the launch left behind — a pane capture, the
78
+ * launcher's stderr, or "" when nothing was captured. It is EVIDENCE, and its
79
+ * absence is never treated as proof of anything: with no output the verdict
80
+ * falls back to what the exit code and the mode can carry on their own.
81
+ *
82
+ * @param {object} a
83
+ * @param {"session-id"|"resume"} a.mode how this launch addressed the session
84
+ * @param {number|null} a.exitCode the recorded exit (null = the exit file was never written)
85
+ * @param {string} [a.output] pane tail / stderr, may be ""
86
+ * @param {boolean} [a.beatSeen] did this launch produce a heartbeat?
87
+ * @returns {{kind:"none"|"resume-target-missing"|"binary-missing"|"never-beaten"|"nonzero-exit", proven:boolean, detail:string}}
88
+ */
89
+ export function classifyLaunchFailure(a = {}) {
90
+ const exitCode = a.exitCode === null || a.exitCode === undefined ? null : Number(a.exitCode);
91
+ const output = typeof a.output === "string" ? a.output : "";
92
+ const beatSeen = a.beatSeen === true;
93
+ const failed = exitCode === null || exitCode !== 0;
94
+
95
+ // A launch that beat and then exited 0 is a session that RAN. Nothing to name.
96
+ if (!failed && beatSeen) return { kind: "none", proven: true, detail: "the session ran and exited cleanly" };
97
+ if (!failed) return { kind: "none", proven: false, detail: "the session exited cleanly without beating" };
98
+
99
+ // PROVEN verdicts first: the runtime said what was wrong, in words.
100
+ if (a.mode === "resume" && anyMatch(RESUME_MISSING_PATTERNS, output)) {
101
+ return {
102
+ kind: "resume-target-missing",
103
+ proven: true,
104
+ detail: "the runtime reported that the conversation this launch tried to resume does not exist on this machine",
105
+ };
106
+ }
107
+ if (exitCode === 127 || anyMatch(BINARY_MISSING_PATTERNS, output)) {
108
+ return {
109
+ kind: "binary-missing",
110
+ proven: exitCode === 127 || anyMatch(BINARY_MISSING_PATTERNS, output),
111
+ detail: "the launcher could not find the binary it was told to run",
112
+ };
113
+ }
114
+
115
+ // INFERRED verdicts. `never-beaten` is the honest name for James's seat when
116
+ // the pane was not captured: the launch ended non-zero having never reported
117
+ // in, so whatever it was, it was never a working session.
118
+ if (!beatSeen) {
119
+ return { kind: "never-beaten", proven: false, detail: `the launch ended (exit ${exitCode === null ? "unknown" : exitCode}) without ever writing a heartbeat` };
120
+ }
121
+ return { kind: "nonzero-exit", proven: false, detail: `the session beat, then ended with exit ${exitCode === null ? "unknown" : exitCode}` };
122
+ }
123
+
124
+ /**
125
+ * Pure: the stable identity of a failure, for counting repeats.
126
+ *
127
+ * Mode and kind only. NOT the session id (see the module docblock) and not the
128
+ * exit code when the kind already names the fault — otherwise a flapping exit
129
+ * status would read as a different fault each time and never accumulate.
130
+ *
131
+ * @param {{mode:string, kind:string, exitCode?:number|null}} a
132
+ * @returns {string} e.g. `resume:resume-target-missing`
133
+ */
134
+ export function launchFailureSignature(a = {}) {
135
+ const kind = String(a.kind || "unknown");
136
+ const mode = a.mode === "resume" ? "resume" : "session-id";
137
+ if (kind === "none") return "";
138
+ if (kind === "never-beaten" || kind === "nonzero-exit") {
139
+ const exit = a.exitCode === null || a.exitCode === undefined ? "unknown" : String(Number(a.exitCode));
140
+ return `${mode}:${kind}:${exit}`;
141
+ }
142
+ return `${mode}:${kind}`;
143
+ }
144
+
145
+ /**
146
+ * Pure: parse `state/session/launch-failures.json`. Anything unreadable is
147
+ * "no streak" — a corrupt counter must never be able to escalate on its own.
148
+ * @param {string|null|undefined} text
149
+ * @returns {{signature:string, streak:number, firstAt:string, lastAt:string, escalatedAt:string}|null}
150
+ */
151
+ export function parseLaunchFailures(text) {
152
+ if (typeof text !== "string" || !text.trim()) return null;
153
+ let raw;
154
+ try { raw = JSON.parse(text); } catch { return null; }
155
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) return null;
156
+ const signature = typeof raw.signature === "string" ? raw.signature : "";
157
+ const streak = Number.isInteger(raw.streak) && raw.streak > 0 ? raw.streak : 0;
158
+ if (!signature || !streak) return null;
159
+ return {
160
+ signature,
161
+ streak,
162
+ firstAt: typeof raw.firstAt === "string" ? raw.firstAt : "",
163
+ lastAt: typeof raw.lastAt === "string" ? raw.lastAt : "",
164
+ escalatedAt: typeof raw.escalatedAt === "string" ? raw.escalatedAt : "",
165
+ };
166
+ }
167
+
168
+ /**
169
+ * Pure: the streak after one more launch outcome.
170
+ *
171
+ * A DIFFERENT signature restarts the count at 1 rather than adding to it: two
172
+ * different faults in a row are two problems, and escalating on their sum
173
+ * would cry wolf. A successful launch (`signature` empty) clears the record
174
+ * entirely — evidence of work, the same reset rule `carriedRestarts` uses.
175
+ *
176
+ * @param {object|null} prior {@link parseLaunchFailures} output
177
+ * @param {{signature:string, at:string}} outcome
178
+ * @returns {{signature:string, streak:number, firstAt:string, lastAt:string, escalatedAt:string}|null} null = clear the file
179
+ */
180
+ export function launchFailureStreak(prior, outcome = {}) {
181
+ const signature = typeof outcome.signature === "string" ? outcome.signature : "";
182
+ const at = typeof outcome.at === "string" ? outcome.at : "";
183
+ if (!signature) return null; // the launch worked: nothing to carry
184
+ if (prior && prior.signature === signature) {
185
+ return { signature, streak: prior.streak + 1, firstAt: prior.firstAt || at, lastAt: at, escalatedAt: prior.escalatedAt || "" };
186
+ }
187
+ return { signature, streak: 1, firstAt: at, lastAt: at, escalatedAt: "" };
188
+ }
189
+
190
+ /**
191
+ * Pure: has this stopped being a retry?
192
+ *
193
+ * `escalate` is true on the launch that REACHES the limit and on every launch
194
+ * after it, because the escalation channel is a beat field that the next
195
+ * supervisor run clears (`attention.json` is unlinked at launch): a one-shot
196
+ * escalation would flicker off and the org would see a seat that healed. The
197
+ * `fresh` flag distinguishes the first crossing, which is the one worth a log
198
+ * line and a distinct wording.
199
+ *
200
+ * @param {{streak:number, limit?:number, escalatedAt?:string}} a
201
+ * @returns {{escalate:boolean, fresh:boolean, streak:number, limit:number, reason:string}}
202
+ */
203
+ export function escalationDecision(a = {}) {
204
+ const limit = Number.isInteger(a.limit) && a.limit > 0 ? a.limit : IDENTICAL_FAILURE_LIMIT;
205
+ const streak = Number.isInteger(a.streak) && a.streak > 0 ? a.streak : 0;
206
+ if (streak < limit) {
207
+ return { escalate: false, fresh: false, streak, limit, reason: `launch failure ${streak}/${limit} — retrying` };
208
+ }
209
+ const fresh = !a.escalatedAt;
210
+ return {
211
+ escalate: true,
212
+ fresh,
213
+ streak,
214
+ limit,
215
+ reason: `${streak} consecutive launches have failed the same way (limit ${limit}) — this is a fault, not a retry`,
216
+ };
217
+ }
218
+
219
+ /**
220
+ * Pure: the one sentence an operator in a browser gets, via the beat's
221
+ * `machine.sessionNote.detail`.
222
+ *
223
+ * THE VERDICT LEADS, THE EXPLANATION TRAILS — because the channel TRUNCATES.
224
+ * `telemetry/collect#sanitizeNoteDetail` caps the detail at 200 characters and
225
+ * appends an ellipsis, so anything after the first sentence may never leave the
226
+ * machine. The first draft of this line ended "…this needs a person" and that
227
+ * clause — the only part that asks for an action — was the part that got cut.
228
+ * So: how many, that retries are not working, then the diagnosis.
229
+ *
230
+ * @param {{signature:string, streak:number, limit?:number, detail:string, sessionId?:string}} a
231
+ * @returns {string}
232
+ */
233
+ export function escalationHint(a = {}) {
234
+ const sig = String(a.signature || "unknown");
235
+ const streak = Number(a.streak) || 0;
236
+ const detail = String(a.detail || "").trim();
237
+ const id = typeof a.sessionId === "string" && a.sessionId ? ` Session ${a.sessionId}.` : "";
238
+ return `${streak}× in a row, identically — retries are not fixing it and this needs a person. ${detail || sig}.${id}`;
239
+ }
240
+
241
+ export default {
242
+ IDENTICAL_FAILURE_LIMIT,
243
+ RESUME_MISSING_PATTERNS,
244
+ BINARY_MISSING_PATTERNS,
245
+ classifyLaunchFailure,
246
+ launchFailureSignature,
247
+ parseLaunchFailures,
248
+ launchFailureStreak,
249
+ escalationDecision,
250
+ escalationHint,
251
+ };
@@ -0,0 +1,86 @@
1
+ /**
2
+ * lib/session/resume-target.mjs — is the conversation we are about to `--resume`
3
+ * actually on this machine?
4
+ *
5
+ * ── WHY THIS IS A PREFLIGHT AND NOT A POST-MORTEM ───────────────────────────
6
+ * The supervisor already had a post-mortem: a resume that died non-zero inside
7
+ * twenty seconds rotates to a fresh id (`identity#rotationDecision`). It is a
8
+ * good heuristic and it is not enough — it cannot tell a dead transcript from a
9
+ * missing binary, it misses a failure that takes twenty-one seconds, and every
10
+ * one of its verdicts costs a full launch, a mux session and a launchd cycle to
11
+ * reach. On James Kirkland's seat that cycle ran every ten minutes for days
12
+ * against a session id that had not existed for just as long.
13
+ *
14
+ * The answer is on disk before we launch. Claude Code keeps each project's
15
+ * transcripts as `~/.claude/projects/<project-slug>/<session-id>.jsonl`, where
16
+ * the slug is the project directory with every non-alphanumeric character
17
+ * replaced by `-`. `--resume` is project-scoped, so the seat's own slug is the
18
+ * only place the id could be.
19
+ *
20
+ * ── FAIL-OPEN, DELIBERATELY ─────────────────────────────────────────────────
21
+ * A wrong "missing" throws away a live transcript; a wrong "unknown" costs one
22
+ * launch and lands in the post-mortem that was already there. So `missing` is
23
+ * returned ONLY from positive evidence: the project directory exists, it holds
24
+ * transcripts, and this id is not among them. An absent directory, an empty
25
+ * one, an unreadable one, an engine that does not use this layout at all — all
26
+ * `unknown`, and the launch proceeds exactly as before.
27
+ *
28
+ * One I/O shell (`readdirSync`/`existsSync`), both injected.
29
+ *
30
+ * @module lib/session/resume-target
31
+ */
32
+
33
+ "use strict";
34
+
35
+ import { readdirSync as fsReaddirSync } from "node:fs";
36
+ import { join } from "node:path";
37
+
38
+ /** Where Claude Code keeps per-project transcripts, relative to the home dir. */
39
+ export const TRANSCRIPT_ROOT = join(".claude", "projects");
40
+
41
+ /**
42
+ * Pure: Claude Code's directory name for a project path — every character that
43
+ * is not a letter or a digit becomes `-` (`/Users/x/hq` → `-Users-x-hq`).
44
+ * @param {string} projectDir
45
+ * @returns {string}
46
+ */
47
+ export function projectSlug(projectDir) {
48
+ return String(projectDir || "").replace(/[^A-Za-z0-9]/g, "-");
49
+ }
50
+
51
+ /**
52
+ * Is `sessionId`'s transcript present for `projectDir`?
53
+ *
54
+ * @param {object} a
55
+ * @param {string} a.homeDir the user's home directory
56
+ * @param {string} a.projectDir the directory the session runs in (the agent root)
57
+ * @param {string} a.sessionId the id this launch would resume
58
+ * @param {"claude"|"cohort"} [a.engine] a non-claude engine does not use this layout → always "unknown"
59
+ * @param {{readdirSync?:Function}} [deps]
60
+ * @returns {{state:"present"|"missing"|"unknown", dir:string, reason:string}}
61
+ */
62
+ export function resumeTargetState(a = {}, deps = {}) {
63
+ const readdirSync = deps.readdirSync || fsReaddirSync;
64
+ const homeDir = typeof a.homeDir === "string" ? a.homeDir : "";
65
+ const sessionId = String(a.sessionId || "").toLowerCase();
66
+ if (a.engine && a.engine !== "claude") return { state: "unknown", dir: "", reason: `engine ${a.engine} does not keep transcripts here` };
67
+ if (!homeDir || !sessionId) return { state: "unknown", dir: "", reason: "no home directory or session id to check" };
68
+ const dir = join(homeDir, TRANSCRIPT_ROOT, projectSlug(a.projectDir));
69
+ let entries;
70
+ try { entries = readdirSync(dir); } catch { return { state: "unknown", dir, reason: "the project's transcript directory is absent or unreadable" }; }
71
+ const names = (Array.isArray(entries) ? entries : []).map((e) => String(e && e.name ? e.name : e));
72
+ const transcripts = names.filter((n) => n.toLowerCase().endsWith(".jsonl"));
73
+ if (transcripts.some((n) => n.toLowerCase() === `${sessionId}.jsonl`)) {
74
+ return { state: "present", dir, reason: "the transcript is on disk" };
75
+ }
76
+ // An EMPTY directory proves nothing: it is equally what a seat looks like the
77
+ // moment before its first session writes, and rotating on it would be a guess.
78
+ if (transcripts.length === 0) return { state: "unknown", dir, reason: "the project's transcript directory holds no transcripts at all" };
79
+ return {
80
+ state: "missing",
81
+ dir,
82
+ reason: `${transcripts.length} transcript(s) on disk for this project and none of them is ${sessionId}`,
83
+ };
84
+ }
85
+
86
+ export default { TRANSCRIPT_ROOT, projectSlug, resumeTargetState };
@@ -99,6 +99,28 @@
99
99
  * daemon?: { pid?: number, bootAt?: ISO8601, uptimeS?: number,
100
100
  * sdkVersion?: string, dashboardAt?: ISO8601,
101
101
  * healthy?: boolean, healthReason?: string },
102
+ * // WHICH UPSTREAM FIXES CANNOT REACH THIS SEAT. A `.maestroignore`
103
+ * // pin on a FRAMEWORK path is a fork, and upstream moves under it:
104
+ * // five of sixteen seats were degraded by exactly this on 2026-09-25,
105
+ * // two of them crash-looping at import time on a pinned deliver.mjs.
106
+ * // `stranded` counts framework pins where upstream holds lines this
107
+ * // seat refuses (`onlyUpstream > 0` in .maestro/ignored-drift.json) —
108
+ * // that, and not "a file differs", is the property that means no
109
+ * // release can ever reach this colleague. ABSENT ENTIRELY on a seat
110
+ * // that pins nothing; `{unknown:true}` when the seat has pins and
111
+ * // could not read its own report, which must never read as clean.
112
+ * // INVENTORY: carries the `at`/`sdkVersion` of the upgrade that took
113
+ * // it, not the snapshot's.
114
+ * // `patterns` counts lines in `.maestroignore`; `matched` counts the
115
+ * // entries the drift report covers. Two numbers, two names, on both
116
+ * // branches — one glob can match forty files, or none.
117
+ * pinnedDrift?: { patterns: number|null, matched: number, framework: number,
118
+ * seatOwned: number, drifting: number, stranded: number,
119
+ * strandedCode: number, localOnly: number,
120
+ * at: ISO8601|null, sdkVersion: string|null,
121
+ * worst: [{ path: string, behind: number, code: boolean }] }
122
+ * | { unknown: true, patterns: number|null, matched: null,
123
+ * reason: string },
102
124
  * },
103
125
  * claudeAuth: "ok"|"relogin_required"|"unknown",
104
126
  * alerts: [{ id, severity, kind, detail }],
@@ -142,6 +164,8 @@ import { liveClaudeStats } from "../resource-governor.mjs";
142
164
  import { agentFirstName } from "../session/identity.mjs";
143
165
  import { snapshot as countersSnapshot } from "../diagnostics/counters.mjs";
144
166
  import { replyDebtFromCounters } from "../daemon/reply-debt.mjs";
167
+ import { summarisePinnedDrift, countPins } from "../upgrade/pinned-drift.mjs";
168
+ import { IGNORED_DRIFT_REL } from "../upgrade/ignored-drift.mjs";
145
169
 
146
170
  /** Default temperature probe ceiling — used only as a guard in alerts; here we just report. */
147
171
  const VALID_STATES = new Set(["active", "idle", "busy", "error", "offline"]);
@@ -1229,6 +1253,13 @@ const ATTENTION_WHY = {
1229
1253
  "no-heartbeat": "up, never beaten",
1230
1254
  "stale-heartbeat": "no beat since this launch",
1231
1255
  "beat-stopped": "beat, then stopped",
1256
+ // The session is not wedged — it never STARTS. Written by the supervisor
1257
+ // once N consecutive launches have failed identically
1258
+ // (scripts/session/supervisor.mjs, lib/session/launch-failure.mjs). It is a
1259
+ // different fault from the three above, with a different fix, and it is the
1260
+ // one that looked like health for days on a seat relaunching a dead resume
1261
+ // id every ten minutes.
1262
+ "launch-failing": "cannot launch",
1232
1263
  };
1233
1264
 
1234
1265
  /**
@@ -1509,6 +1540,86 @@ export function daemonSummary(dash, last) {
1509
1540
  return Object.keys(out).length > 0 ? out : null;
1510
1541
  }
1511
1542
 
1543
+ // ---------------------------------------------------------------------------
1544
+ // Pinned framework files — WHICH UPSTREAM FIXES CANNOT REACH THIS SEAT
1545
+ // (`machine.pinnedDrift`)
1546
+ // ---------------------------------------------------------------------------
1547
+
1548
+ /**
1549
+ * Pure: the beat's `machine.pinnedDrift`.
1550
+ *
1551
+ * WHY THIS FIELD EXISTS. A `.maestroignore` entry on a framework path is a
1552
+ * fork, and upstream keeps moving under it. MEASURED 2026-09-25, repairing
1553
+ * five of sixteen seats by hand: two were crash-looping at import time on a
1554
+ * pinned `lib/org/inbound/deliver.mjs`; two more pinned
1555
+ * `scripts/daemon/agent-daemon.mjs` and therefore could not receive the
1556
+ * front-door revive fix (2.18.11/12) OR the persona fix — one of them went on
1557
+ * emitting an AI self-introduction for days after it was fixed upstream,
1558
+ * because the fix could not physically arrive.
1559
+ *
1560
+ * Every one of those seats already knew. `maestro upgrade` has written the
1561
+ * per-file answer to `.maestro/ignored-drift.json` for weeks and NOTHING read
1562
+ * it — no warning, no beat field, no alert. This is the half that makes the
1563
+ * knowledge leave the machine it is about.
1564
+ *
1565
+ * THE PROPERTY WORTH CARRYING is not "a file differs". It is "upstream holds
1566
+ * lines this pin refuses", which is `onlyUpstream > 0` on a framework path —
1567
+ * see `lib/upgrade/pinned-drift.mjs` for why (on this seat the same day, two
1568
+ * of seven drifting pins refused nothing at all, which is what a pin is FOR).
1569
+ *
1570
+ * SMALL AND BOUNDED, like every other field on `machine`: a handful of counts
1571
+ * and at most three paths, each capped (the size floor is pinned by
1572
+ * `pinned-drift-beat.test.mjs`). It rides `machine` for the reason
1573
+ * `frontDoor` and `daemon` do — hq validates `machine` as an OPEN record
1574
+ * (`presence/beat.ts#statusSchema`) while `session` is `.strict()`, so a new
1575
+ * seat-level fact lands without a lock-step hq deploy.
1576
+ *
1577
+ * THREE ANSWERS, AND THE THIRD IS THE POINT:
1578
+ * · a seat that pins nothing returns `null` — the caller drops the key, and
1579
+ * hq grows no field and no alert for a seat with nothing wrong;
1580
+ * · a seat that pins something and cannot read its own report returns
1581
+ * `{unknown:true}` — never a clean bill. "Could not tell" and "nothing to
1582
+ * tell" are opposite instructions, and collapsing them is the exact
1583
+ * failure this whole field exists to end.
1584
+ *
1585
+ * INVENTORY, not a momentary reading: the report is written by the last
1586
+ * upgrade and stays true until the next one, so it carries its own `at` and
1587
+ * the `sdkVersion` it was taken against rather than borrowing the snapshot's.
1588
+ *
1589
+ * @param {object|null} report parsed `.maestro/ignored-drift.json`, or null
1590
+ * @param {number|null} pinCount pattern lines in `.maestroignore`, or null
1591
+ * when even that could not be read
1592
+ */
1593
+ export function pinnedDriftSummary(report, pinCount) {
1594
+ return summarisePinnedDrift(report, { pinCount });
1595
+ }
1596
+
1597
+ /** `.maestro/ignored-drift.json`, or null — fail-open. */
1598
+ function readIgnoredDriftReport(agentRoot) {
1599
+ if (!agentRoot) return null;
1600
+ return safeReadJson(join(resolve(agentRoot), IGNORED_DRIFT_REL));
1601
+ }
1602
+
1603
+ /**
1604
+ * How many real patterns `.maestroignore` holds — `0` when the file is absent
1605
+ * (a seat with no pins), `null` when it exists and could not be read.
1606
+ *
1607
+ * The distinction is load-bearing: `0` is what lets the field be dropped
1608
+ * entirely, and `null` is what forces `unknown` instead of a clean answer.
1609
+ * A missing file is a definite zero; an unreadable one is not.
1610
+ */
1611
+ function readPinCount(agentRoot) {
1612
+ if (!agentRoot) return null;
1613
+ const file = join(resolve(agentRoot), ".maestroignore");
1614
+ let text;
1615
+ try {
1616
+ text = readFileSync(file, "utf8");
1617
+ } catch (e) {
1618
+ return e && e.code === "ENOENT" ? 0 : null;
1619
+ }
1620
+ return countPins(text);
1621
+ }
1622
+
1512
1623
  /** state/autoupdate/last.json, or null. */
1513
1624
  function readAutoupdateLast(agentRoot) {
1514
1625
  if (!agentRoot) return null;
@@ -1974,6 +2085,21 @@ export async function collectStatus(o = {}) {
1974
2085
  if (d) machine.daemon = d;
1975
2086
  } catch { /* no daemon field */ }
1976
2087
 
2088
+ // 4e. WHICH UPSTREAM FIXES CANNOT REACH THIS SEAT — `machine.pinnedDrift`,
2089
+ // absent on a seat that pins nothing. See `pinnedDriftSummary` for the five
2090
+ // seats this was measured on. Fail-open like every probe here: a throw drops
2091
+ // the field, never the beat — but note that a *readable* seat with pins and
2092
+ // an unreadable report deliberately lands `{unknown:true}` rather than
2093
+ // nothing, because "I could not tell" must never render as "clean".
2094
+ try {
2095
+ const pinReport = opt.ignoredDrift !== undefined
2096
+ ? opt.ignoredDrift
2097
+ : readIgnoredDriftReport(opt.agentRoot);
2098
+ const pins = opt.pinCount !== undefined ? opt.pinCount : readPinCount(opt.agentRoot);
2099
+ const pd = pinnedDriftSummary(pinReport, pins);
2100
+ if (pd) machine.pinnedDrift = pd;
2101
+ } catch { /* no pinnedDrift field */ }
2102
+
1977
2103
  const status = {
1978
2104
  state,
1979
2105
  activity: act.activity || "idle",
@@ -2026,6 +2152,9 @@ export const _internals = {
2026
2152
  upgradeSummary,
2027
2153
  parseDaemonHealthYaml,
2028
2154
  daemonSummary,
2155
+ pinnedDriftSummary,
2156
+ readIgnoredDriftReport,
2157
+ readPinCount,
2029
2158
  detectClaudeAuth,
2030
2159
  resetAuthProbeCache,
2031
2160
  scanForRelogin,