@cohortapp/agent-sdk 2.18.15 → 2.18.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,9 +22,10 @@
22
22
  *
23
23
  * The daemon is the honest home for this. It is a separate process, it is
24
24
  * already alive on every seat that beats, it already reads the front-door
25
- * state every poll to decide who owns the inbox, and it runs every thirty
26
- * seconds. A seat whose door shuts is then measured in a minute rather than a
27
- * day, by something the shut door cannot take down with it.
25
+ * state every poll to decide who owns the inbox, and it re-checks the door on
26
+ * its own sixty-second cadence (REVIVE_CHECK_INTERVAL_MS). A seat whose door
27
+ * shuts is then measured in a minute or two rather than a day, by something
28
+ * the shut door cannot take down with it.
28
29
  *
29
30
  * Pure: the decision only. The daemon owns the kickstart.
30
31
  */
@@ -38,19 +39,33 @@ export const DEFAULT_REVIVE_BACKOFF_MS = 15 * 60 * 1000;
38
39
  /** How many restarts before the daemon stops and leaves it to a person. */
39
40
  export const DEFAULT_REVIVE_MAX = 3;
40
41
 
42
+ /**
43
+ * Consecutive not-live reads required before the FIRST kickstart.
44
+ *
45
+ * A kickstart drops whatever the session was doing. One bad read is not
46
+ * enough evidence to pay that cost — a heartbeat file can be caught
47
+ * mid-write, or a poll can land in the one second between a session ending
48
+ * a tool call and starting the next. Two reads in a row, SIXTY seconds
49
+ * apart at the daemon's revive-check cadence (REVIVE_CHECK_INTERVAL_MS), is
50
+ * the line Ravi drew after the fleet had already seen what a single-read
51
+ * trigger costs.
52
+ */
53
+ export const DEFAULT_REVIVE_CONFIRM_READS = 2;
54
+
41
55
  /**
42
56
  * Should the daemon restart the front door right now?
43
57
  *
44
58
  * The caller supplies the front-door state it already reads for dispatch, the
45
59
  * age of the session heartbeat, and what this daemon has already tried. Every
46
60
  * bound is explicit because the failure mode of getting this wrong is a seat
47
- * that restarts its own session every thirty seconds forever — which is worse
61
+ * that restarts its own session every sixty seconds forever — which is worse
48
62
  * than the shut door, and is the reason the ladder ends in "stop and say so"
49
63
  * rather than in another attempt.
50
64
  *
51
65
  * @param {{frontDoor?:string, sessionLive?:boolean, silentMs?:number|null,
52
66
  * attempts?:number, lastAttemptAt?:number|null, now:number,
53
- * reviveAfterMs?:number, backoffMs?:number, maxAttempts?:number}} a
67
+ * reviveAfterMs?:number, backoffMs?:number, maxAttempts?:number,
68
+ * notLiveReads?:number, confirmReads?:number}} a
54
69
  * @returns {{revive:boolean, reason:string}}
55
70
  */
56
71
  export function shouldReviveFrontDoor(a) {
@@ -59,6 +74,7 @@ export function shouldReviveFrontDoor(a) {
59
74
  const after = Number.isFinite(x.reviveAfterMs) && x.reviveAfterMs > 0 ? x.reviveAfterMs : DEFAULT_REVIVE_AFTER_MS;
60
75
  const backoff = Number.isFinite(x.backoffMs) && x.backoffMs > 0 ? x.backoffMs : DEFAULT_REVIVE_BACKOFF_MS;
61
76
  const max = Number.isFinite(x.maxAttempts) && x.maxAttempts >= 0 ? x.maxAttempts : DEFAULT_REVIVE_MAX;
77
+ const confirm = Number.isFinite(x.confirmReads) && x.confirmReads > 0 ? x.confirmReads : DEFAULT_REVIVE_CONFIRM_READS;
62
78
 
63
79
  // A seat whose lane is the daemon has no front door to revive: the daemon
64
80
  // itself is answering, and restarting a session job it does not depend on
@@ -82,6 +98,16 @@ export function shouldReviveFrontDoor(a) {
82
98
  if (!Number.isFinite(silentMs)) return { revive: false, reason: "silence-unknown" };
83
99
  if (silentMs < after) return { revive: false, reason: "within-grace" };
84
100
 
101
+ // Two-read hysteresis, gating only the FIRST kickstart: a kickstart drops
102
+ // whatever the session was mid-way through, so the daemon must not act on
103
+ // a single not-live read before it has confirmed the verdict on the next
104
+ // poll. `notLiveReads` undefined means a caller that predates this streak
105
+ // (every existing test above) — treat it as already confirmed so those
106
+ // callers see no change in behaviour; the daemon is the one caller that
107
+ // threads the real streak through.
108
+ const reads = x.notLiveReads === undefined ? confirm : Number(x.notLiveReads);
109
+ if (!(reads >= confirm)) return { revive: false, reason: "confirming" };
110
+
85
111
  const attempts = Number(x.attempts) || 0;
86
112
  if (attempts >= max) return { revive: false, reason: "budget-spent" };
87
113
 
@@ -93,6 +119,277 @@ export function shouldReviveFrontDoor(a) {
93
119
  return { revive: true, reason: `front door silent ${Math.round(silentMs / 1000)}s (attempt ${attempts + 1}/${max})` };
94
120
  }
95
121
 
122
+ /**
123
+ * Should this seat's daemon job be restarted? Local check only — the
124
+ * reciprocal of `shouldReviveFrontDoor`'s job, but it CANNOT be run by the
125
+ * daemon on itself: a process that is alive to ask the question is a process
126
+ * launchctl reports a live pid for, so a live daemon can never observe its own
127
+ * down state. The actor is therefore the hourly `autoupdate.sh` backstop (it
128
+ * already parses `launchctl list` for `daemon_last_exit`), which is exactly
129
+ * the seat-local watchdog that catches a daemon that has stopped while the
130
+ * front door kept beating (Jacob's 2026-09-25 case: heartbeat.json ticking, no
131
+ * overdue handoff, daemon dead). `autoupdate.sh#revive_daemon` is the caller.
132
+ *
133
+ * `launchctl` is read by the caller from `launchctl list`, the same source
134
+ * `sibling_job_audit` and `autoupdate.sh` already parse: "pid" when the
135
+ * column holds a number, "dash" when it holds `-` (loaded, no pid), "absent"
136
+ * when the label is not in the list at all.
137
+ *
138
+ * PHANTOM PID: a silent crash loop can leave `launchctl list` holding a pid
139
+ * whose process has genuinely exited — the label still carries a number, but a
140
+ * `kill -0` on that pid fails because no such process is running, a kickstart
141
+ * yields another dead pid, and nothing lands on stderr. A LISTED pid is not a
142
+ * RUNNING daemon. So the caller decides liveness by `kill -0` on the pid and
143
+ * passes `pidAlive`; a phantom pid (`launchctl:"pid"` but `pidAlive:false`) is
144
+ * treated exactly like "dash" — loaded with no live process — and the beat age
145
+ * decides, instead of the listed pid rubber-stamping the daemon as alive.
146
+ *
147
+ * @param {{launchctl?:string, beatAgeMs?:number, staleMs?:number, pidAlive?:boolean}} a
148
+ * @returns {{revive:boolean, reason:string}}
149
+ */
150
+ export function shouldReviveDaemon(a) {
151
+ const x = a && typeof a === "object" ? a : {};
152
+ const stale = Number.isFinite(x.staleMs) && x.staleMs > 0 ? x.staleMs : DEFAULT_REVIVE_AFTER_MS;
153
+
154
+ // A pid that launchctl lists but the OS does not actually run is a phantom —
155
+ // the same down state as a dash, dressed up as alive. Only an EXPLICIT
156
+ // `pidAlive === false` demotes it; a caller that does not probe liveness
157
+ // (`pidAlive` undefined) keeps the old behaviour, so nothing that never
158
+ // measured it changes.
159
+ const phantom = x.launchctl === "pid" && x.pidAlive === false;
160
+ const state = phantom ? "dash" : x.launchctl;
161
+
162
+ if (state === "pid") return { revive: false, reason: "daemon-alive" };
163
+
164
+ // No job at all is not a wedge to clear — it is a seat that was never
165
+ // provisioned with one, or had it removed, and a kickstart has nothing to
166
+ // aim at.
167
+ if (state === "absent") return { revive: false, reason: "daemon-job-absent" };
168
+
169
+ if (state === "dash") {
170
+ const beatAgeMs = Number(x.beatAgeMs);
171
+ // A loaded job with no live pid and a beat that has gone stale is the down
172
+ // state this exists to catch. A fresh beat under the same launchctl
173
+ // reading means the prior instance is still exiting, not that this one
174
+ // needs help.
175
+ if (!(beatAgeMs > stale)) return { revive: false, reason: "daemon-settling" };
176
+ return { revive: true, reason: phantom ? "daemon-phantom-pid" : "daemon-down" };
177
+ }
178
+
179
+ // Anything else — missing, empty, an error string from a failed parse — is
180
+ // a reading this cannot name a state from, and an unnamed state is never
181
+ // grounds to act.
182
+ return { revive: false, reason: "daemon-state-unknown" };
183
+ }
184
+
185
+ /** How many consecutive observations across the settle window prove continuity. */
186
+ export const DEFAULT_CONFIRM_SAMPLES = 2;
187
+
188
+ /**
189
+ * Did a revive attempt actually work? Checked from OUTSIDE the restart helper,
190
+ * on purpose: a helper that exits 0 has told you it ran, not that it succeeded,
191
+ * and trusting that exit code is the gap that let a shut front door sit
192
+ * unrestarted for days while the job meant to fix it kept reporting clean.
193
+ *
194
+ * ── WHY A ONE-SHOT PID CHECK IS NOT ENOUGH (Jacob, incident-validated) ──────
195
+ *
196
+ * A single "is there a pid, is the beat fresh" read passes two things that are
197
+ * NOT a revive: a PHANTOM pid (launchctl lists it, the process is dead),
198
+ * and a COME-UP-THEN-CRASH loop that beats once, dies, and is relaunched — its
199
+ * local beat file always looks fresh because the crash-looping restart keeps
200
+ * re-stamping it. The bar is "alive-AND-still-beating-server-side", proven two
201
+ * ways, either of which alone is sufficient:
202
+ *
203
+ * 1. UPTIME CONTINUITY — the one thing a crash loop cannot fake. One pid,
204
+ * alive on every read (`pidAlive`), the SAME process identity (start-time,
205
+ * `ps -o lstart`) across the whole settle window: no restart boundary. A
206
+ * crash loop moves its start-time on every restart, so identity drift is
207
+ * the tell. This alone still admits an alive-but-WEDGED process, so it is
208
+ * paired with a fresh presence beat read back FROM THE SERVER — not the
209
+ * local beat file, which the wedged/crash-looping process keeps stamping.
210
+ * `serverBeatFresh:false` (server reachable, beat stale = alive-but-not-
211
+ * publishing) blocks; `undefined` (could not reach the server) does NOT —
212
+ * recovery is never held hostage to org reachability, continuity carries.
213
+ *
214
+ * 2. A COMPLETED UNIT OF WORK — the STRONGER signal when it exists, and
215
+ * SUFFICIENT on its own (Jacob's positive control: A001 recovered dark →
216
+ * 12 authored posts). A crash loop cannot author a real unit of work.
217
+ *
218
+ * Work is sufficient, not necessary: a seat that recovered into an EMPTY queue
219
+ * does real nothing and MUST still pass on continuity alone — an assert that
220
+ * demanded work would fire a revive on every idle seat overnight and get
221
+ * itself disabled at 3am. Continuity-OR-work separates busy-recovered,
222
+ * crash-loop, phantom, AND quiet-recovered.
223
+ *
224
+ * ── THREE COMPLEMENTARY WITHIN-WINDOW CONDITIONS ────────────────────────────
225
+ *
226
+ * Continuity is three independent checks, each catching a failure the other two
227
+ * cannot: pid-recycling breaks start-time, a crash loop breaks pid, and a WRONG
228
+ * process (right pid, wrong binary) breaks comm. The third exists because `ps`
229
+ * is aliased to `pnpm start` on this fleet — a stray match on the wrong process
230
+ * would otherwise read as a revive. When the caller supplies `expectedComm` and
231
+ * a `comm` per sample (from `/bin/ps -p <pid> -o comm=`), a mismatch fails and
232
+ * NAMES `wrong-comm`; a null comm is unknown, not wrong, and continuity carries
233
+ * it. A caller that passes no `expectedComm` sees the pre-comm-check behaviour.
234
+ *
235
+ * @param {{
236
+ * samples?:Array<{pid?:number, alive?:boolean, startTime?:string|null, comm?:string|null}>,
237
+ * serverBeatFresh?:boolean, workObserved?:boolean, minSamples?:number,
238
+ * expectedComm?:string
239
+ * }} a `samples` are the per-read observations across the settle window;
240
+ * `serverBeatFresh` is a fresh presence beat read back from the SERVER;
241
+ * `workObserved` is a completed unit of work since the kickstart;
242
+ * `expectedComm` is the binary name the revived pid must be running.
243
+ * @returns {{ok:boolean, reason:string}}
244
+ */
245
+ export function confirmRevived(a) {
246
+ const x = a && typeof a === "object" ? a : {};
247
+
248
+ // Work is sufficient on its own and is the strongest evidence there is — a
249
+ // crash loop cannot produce a completed unit of work.
250
+ if (x.workObserved === true) return { ok: true, reason: "revived-work" };
251
+
252
+ const min = Number.isFinite(x.minSamples) && x.minSamples > 0 ? x.minSamples : DEFAULT_CONFIRM_SAMPLES;
253
+ const samples = Array.isArray(x.samples) ? x.samples : null;
254
+
255
+ // Fewer reads than the settle window needs cannot establish continuity: the
256
+ // process may be about to crash on its next breath. Not a failure to
257
+ // escalate — just not yet confirmed, so the caller re-checks within budget.
258
+ if (!samples || samples.length < min) return { ok: false, reason: "settling" };
259
+
260
+ // A live pid must appear SOMEWHERE in the window. If none ever did — every
261
+ // read dead — the helper claimed a process that is not there: a phantom pid
262
+ // (launchctl lists it, `kill -0` says dead) or an exit-0-with-no-pid. That is
263
+ // the failure-to-escalate that hands the seat to a person.
264
+ const isAlive = (s) => s && Number.isInteger(s.pid) && s.pid > 0 && s.alive === true;
265
+ const aliveSamples = samples.filter(isAlive);
266
+ if (aliveSamples.length === 0) return { ok: false, reason: "no-pid" };
267
+
268
+ // A dead read FOLLOWED BY a live one is a daemon mid-boot, not a failed
269
+ // revive. The first settle sample is taken the instant the kickstart returns,
270
+ // before a normal slow boot has drawn its first breath — so a dead first read
271
+ // then an alive one is boot-in-progress. Latching that as terminal aborts a
272
+ // revive that was working. A window that is not yet all-alive is still
273
+ // SETTLING: re-check within budget rather than escalate. Only two genuinely
274
+ // ALIVE reads can prove — or refute — continuity.
275
+ if (aliveSamples.length < samples.length) return { ok: false, reason: "settling" };
276
+
277
+ // Every read is alive. Same pid AND same start-time across the window = one
278
+ // instance that never restarted. A restart boundary between two ALIVE reads —
279
+ // a different pid, or the same pid with a different start-time (PID reuse by
280
+ // the relaunched process) — is the fingerprint of a crash loop, never a
281
+ // revive. `startTime` may be null on both when `ps` was unavailable;
282
+ // identical-null still agrees, and the pid continuity carries the degraded
283
+ // case.
284
+ const first = aliveSamples[0];
285
+ const continuous = aliveSamples.every((s) => s.pid === first.pid && s.startTime === first.startTime);
286
+ if (!continuous) return { ok: false, reason: "restart-boundary" };
287
+
288
+ // THIRD within-window condition: the binary under the pid must be the one we
289
+ // kickstarted. Right pid, right start-time, WRONG process is still not a
290
+ // revive — and `ps` aliased to `pnpm start` on this fleet makes a stray match
291
+ // real. Opt-in: only when the caller supplies `expectedComm`. A null comm is
292
+ // unknown (ps unavailable / pid vanished), not wrong — continuity carries it,
293
+ // as a null start-time does; only an OBSERVED, mismatching comm fails. Matched
294
+ // by suffix because `ps -o comm=` prints the absolute path and the expected
295
+ // value is a bare binary name.
296
+ if (x.expectedComm) {
297
+ const want = String(x.expectedComm);
298
+ const observed = aliveSamples.filter((s) => s.comm != null);
299
+ const commOk = observed.every((s) => {
300
+ const got = String(s.comm);
301
+ return got === want || got.endsWith(`/${want}`) || want.endsWith(`/${got}`);
302
+ });
303
+ if (!commOk) return { ok: false, reason: "wrong-comm" };
304
+ }
305
+
306
+ // Alive and continuous, but is it actually WORKING? The local beat file lies
307
+ // in a wedge, so the confirming signal is the presence beat the server
308
+ // recorded. An explicit stale reading (reachable, not publishing) blocks; an
309
+ // unknown reading (unreachable) does not — continuity carries it.
310
+ if (x.serverBeatFresh === false) return { ok: false, reason: "beat-not-advancing" };
311
+
312
+ return { ok: true, reason: "revived" };
313
+ }
314
+
315
+ /**
316
+ * A confirm outcome that must STOP the revive and hand the seat to a person,
317
+ * rather than fire another blind kickstart. A second attempt after a crash-loop
318
+ * assert buries the FIRST failure's log under the loop, so these outcomes are
319
+ * terminal: `no-pid` (phantom / exit-0-with-no-pid), `restart-boundary`
320
+ * (came-up-then-crashed within the settle window), `wrong-comm` (a live pid
321
+ * running the wrong binary — not the process we kickstarted), and
322
+ * `crash-loop-cross-poll` (an unrequested restart caught a poll later — the
323
+ * crash loop whose period outran the settle window). `settling` and
324
+ * `beat-not-advancing` are "not yet" — the caller may re-check within budget.
325
+ *
326
+ * @param {{ok?:boolean, reason?:string}} confirmed
327
+ * @returns {boolean}
328
+ */
329
+ export function isTerminalReviveFailure(confirmed) {
330
+ const r = confirmed && confirmed.reason;
331
+ return r === "no-pid" || r === "restart-boundary" || r === "wrong-comm" || r === "crash-loop-cross-poll";
332
+ }
333
+
334
+ /**
335
+ * Did the session job restart WITHOUT this daemon asking it to, BETWEEN two
336
+ * daemon polls? The unbounded companion to `confirmRevived`'s within-window
337
+ * continuity check.
338
+ *
339
+ * ── WHY WITHIN-WINDOW CONTINUITY IS NOT ENOUGH ──────────────────────────────
340
+ *
341
+ * `confirmRevived` proves continuity across a SETTLE WINDOW (a few seconds). A
342
+ * crash loop whose period EXCEEDS that window shows the SAME identity twice
343
+ * inside it — two alive reads, same pid, same start-time — and false-confirms
344
+ * "revived". A 25s crash-loop period against a 5s settle window is invisible to
345
+ * a within-window check, and WIDENING the window only relocates the threshold:
346
+ * a 65s period beats a 60s window just as cleanly.
347
+ *
348
+ * The fix has no threshold to tune. The daemon persists the last-confirmed
349
+ * {pid, startTime} between its 60s poll iterations. On the NEXT poll, if the job
350
+ * holds a DIFFERENT identity (a different pid, OR the same pid with a moved
351
+ * start-time from PID reuse by the relaunched process) AND this daemon issued NO
352
+ * kickstart in the interval, that is an UNREQUESTED restart — the crash loop, at
353
+ * ANY period. A changed identity WITH a kickstart in between is exactly what a
354
+ * legitimate revive looks like and is never flagged.
355
+ *
356
+ * Pure: the verdict only. The daemon owns the persistence, the kickstart flag,
357
+ * and the halt.
358
+ *
359
+ * @param {{prev?:{pid?:number, startTime?:string|null}|null,
360
+ * curr?:{pid?:number, startTime?:string|null}|null,
361
+ * kickstartIssuedSince?:boolean}} a
362
+ * `prev` is the last poll's confirmed identity (null on the first poll);
363
+ * `curr` is this poll's observed identity; `kickstartIssuedSince` is whether
364
+ * THIS daemon kickstarted the job between the two polls.
365
+ * @returns {{restarted:boolean, reason:string}}
366
+ */
367
+ export function detectUnrequestedRestart(a) {
368
+ const x = a && typeof a === "object" ? a : {};
369
+ const prev = x.prev && typeof x.prev === "object" ? x.prev : null;
370
+ const curr = x.curr && typeof x.curr === "object" ? x.curr : null;
371
+
372
+ // Nothing to compare against yet — the first poll after boot establishes the
373
+ // baseline, it cannot judge drift from one.
374
+ if (!prev || !(Number.isInteger(prev.pid) && prev.pid > 0)) return { restarted: false, reason: "no-prior" };
375
+
376
+ // No live identity this poll is a DOWN door, not an identity change — that is
377
+ // the revive ladder's job. This function speaks only to drift between two
378
+ // observed identities.
379
+ if (!curr || !(Number.isInteger(curr.pid) && curr.pid > 0)) return { restarted: false, reason: "no-current" };
380
+
381
+ const sameIdentity = curr.pid === prev.pid && curr.startTime === prev.startTime;
382
+ if (sameIdentity) return { restarted: false, reason: "continuous" };
383
+
384
+ // Identity changed. If this daemon asked for it, that is a legitimate revive,
385
+ // not the loop.
386
+ if (x.kickstartIssuedSince === true) return { restarted: false, reason: "kickstart-issued" };
387
+
388
+ // Changed identity, no kickstart in between: an unrequested restart — the
389
+ // crash loop, at whatever period it runs.
390
+ return { restarted: true, reason: "unrequested-restart" };
391
+ }
392
+
96
393
  /**
97
394
  * The command that restarts the front-door job. Pure — the caller runs it.
98
395
  *
@@ -172,6 +172,7 @@ import { snapshot as countersSnapshot } from "../diagnostics/counters.mjs";
172
172
  import { replyDebtFromCounters } from "../daemon/reply-debt.mjs";
173
173
  import { summarisePinnedDrift, countPins } from "../upgrade/pinned-drift.mjs";
174
174
  import { IGNORED_DRIFT_REL } from "../upgrade/ignored-drift.mjs";
175
+ import { listHandoffs, DEFAULT_HANDOFF_DEADLINE_MS } from "../session/handoffs.mjs";
175
176
 
176
177
  /** Default temperature probe ceiling — used only as a guard in alerts; here we just report. */
177
178
  const VALID_STATES = new Set(["active", "idle", "busy", "error", "offline"]);
@@ -1135,6 +1136,44 @@ async function collectIdentity(o, osImpl) {
1135
1136
  /** A main-session heartbeat older than this is not live. */
1136
1137
  export const SESSION_STALE_MS = 90_000;
1137
1138
 
1139
+ /**
1140
+ * A FAILED REVIVE, written by the on-host actors (the daemon's front-door
1141
+ * revive and `autoupdate.sh#revive_daemon`) when a kickstart returned but no
1142
+ * live pid could be confirmed — a failure to escalate, which has to reach the
1143
+ * fleet, not just the seat's log. Relative to the agent root; fresh window
1144
+ * bounds how long it rides the beat after the last failed attempt.
1145
+ */
1146
+ export const REVIVE_NOTE_REL = "state/telemetry/revive-note.json";
1147
+ export const REVIVE_NOTE_FRESH_MS = 90 * 60 * 1000;
1148
+
1149
+ /**
1150
+ * Pure: a revive-note that NAMES budget exhaustion, or null.
1151
+ *
1152
+ * A separate legibility defect from a failed revive: when the revive ladder
1153
+ * spends its whole restart budget and STOPS (`shouldReviveFrontDoor` →
1154
+ * `budget-spent`), nothing is written — the seat goes quiet and a quiet system
1155
+ * looks healthy. A ladder that gave up is a failure that does not name itself.
1156
+ * This turns that silence into a beat note carried through the SAME channel as
1157
+ * `machine.reviveNote`, so the beat says "stopped because budget spent" rather
1158
+ * than nothing. Only the budget-spent verdict qualifies — any other verdict
1159
+ * (still confirming, backing off, answering) is not the ladder giving up.
1160
+ *
1161
+ * @param {{verdict?:string, target?:string, attempts?:number, at?:string}} a
1162
+ * @returns {{reason:string, target?:string, detail:string, at:string}|null}
1163
+ */
1164
+ export function budgetSpentNote(a = {}) {
1165
+ const x = a && typeof a === "object" ? a : {};
1166
+ if (x.verdict !== "budget-spent") return null;
1167
+ const attempts = Number.isFinite(x.attempts) ? x.attempts : 0;
1168
+ const at = typeof x.at === "string" && x.at ? x.at : new Date().toISOString();
1169
+ return {
1170
+ reason: "revive-budget-spent",
1171
+ ...(typeof x.target === "string" && x.target ? { target: x.target } : {}),
1172
+ detail: `revive budget spent after ${attempts} attempts`,
1173
+ at,
1174
+ };
1175
+ }
1176
+
1138
1177
  /**
1139
1178
  * Pure: derive `{frontDoor, sessionLive}` from the heartbeat the main session's
1140
1179
  * feed writes every 15 s (`state/session/heartbeat.json` `{pid, ppid, sessionId,
@@ -1395,6 +1434,136 @@ export function sessionNote(a = {}) {
1395
1434
  };
1396
1435
  }
1397
1436
 
1437
+ /**
1438
+ * Front-door-routed inbox lanes. `sweep`, `held` and `deferred` are the
1439
+ * consumer's OWN maintenance lanes — items it parked on purpose, not items
1440
+ * waiting on a person — and a `cohort-held-*` id is the sibling hold-queue
1441
+ * shape (see meeting-capture's `kind==HOLD`). Neither is evidence the front
1442
+ * door is failing to read; counting them would wedge a seat that is doing
1443
+ * exactly what it was told to do.
1444
+ */
1445
+ const WEDGE_EXCLUDED_LANES = new Set(["sweep", "held", "deferred"]);
1446
+ const WEDGE_HELD_ID_RE = /^cohort-held-/;
1447
+
1448
+ /** An inbox item this file's `sessionWedge` treats as front-door-routed. */
1449
+ function isFrontDoorInboxItem(item) {
1450
+ if (!item || typeof item !== "object") return false;
1451
+ const id = typeof item.id === "string" ? item.id : "";
1452
+ if (WEDGE_HELD_ID_RE.test(id)) return false;
1453
+ const lane = typeof item.lane === "string" ? item.lane : "";
1454
+ if (WEDGE_EXCLUDED_LANES.has(lane)) return false;
1455
+ return true;
1456
+ }
1457
+
1458
+ /**
1459
+ * Pure: the wedge verdict — "beating but not reading". A seat can have a
1460
+ * perfectly live heartbeat while the process behind it has stopped actually
1461
+ * answering inbound (wedged on a dialog with no watchdog for THIS failure
1462
+ * mode, a deadlocked event loop, anything short of the process dying). Two
1463
+ * independent signals, checked in order, because they catch different
1464
+ * failures and the first is the harder evidence:
1465
+ *
1466
+ * 1. PRIMARY — an un-acked handoff (`lib/session/handoffs.mjs`) past its
1467
+ * deadline. The consumer handed a cadence tick to the front door and
1468
+ * nobody acked it in time; this is true regardless of which front door is
1469
+ * nominally live, because a handoff is only ever written TO whichever
1470
+ * door claimed the tick. Deadline math mirrors `expireHandoffs` exactly
1471
+ * (`deadlineAt`, else `enqueuedAt + deadlineMs`) so the two never
1472
+ * disagree about which handoff is overdue.
1473
+ * 2. SECONDARY — inbox-unconsumed. DEFERRED and HELD OFF in this build
1474
+ * (`SECONDARY_WEDGE_ENABLED === false`), because it cannot yet fire
1475
+ * correctly end-to-end (Hannah, PR #70 review #2). Three things must land
1476
+ * together first, NONE of them in this PR:
1477
+ * (i) a consume-stamp on EVERY front-door path — today only
1478
+ * `ackHandoff` stamps `lastConsumedAt`, so it advances only on
1479
+ * handoff acks (~every 30 min), never on the inbound
1480
+ * claim/reply/done path in session-runtime;
1481
+ * (ii) a real front-door inbox reader feeding `frontDoorInbox` into
1482
+ * `collectStatus` — nothing feeds it, so `inbox` is always `[]`;
1483
+ * (iii) lane/held keying matched to the real item shape — on this fleet
1484
+ * inbox items carry `channel`/`kind`/`claimed_by`, not `lane`, and
1485
+ * no id begins `cohort-held-`, so `isFrontDoorInboxItem`'s
1486
+ * exclusion is keyed on fields the items do not carry.
1487
+ * Shipped live at the 90s bound it would read most seats wedged, and
1488
+ * `rollout`'s `done` (which requires zero wedged) would never clear. The
1489
+ * branch, `isFrontDoorInboxItem` and `WEDGE_EXCLUDED_LANES` are kept
1490
+ * intact behind the flag so the follow-up wires them rather than rebuilds
1491
+ * them; the PRIMARY handoff-overdue tell is the real signal and is fully
1492
+ * live.
1493
+ *
1494
+ * Defensive by construction: every read is behind a type check, nothing here
1495
+ * touches a clock or the filesystem, and no input shape can make it throw.
1496
+ *
1497
+ * @param {object} [a]
1498
+ * @param {object[]} [a.handoffs] `listHandoffs()` records (or equivalent)
1499
+ * @param {object[]} [a.inbox] raw inbox items, each `{id?, lane?}`
1500
+ * @param {number|null} [a.lastConsumedAt] epoch ms, or null/undefined = never
1501
+ * @param {"session"|"daemon"} [a.frontDoor]
1502
+ * @param {boolean} [a.sessionLive]
1503
+ * @param {number} [a.now] epoch ms
1504
+ * @param {number} [a.deadlineMs] handoff deadline (default {@link DEFAULT_HANDOFF_DEADLINE_MS})
1505
+ * @param {number} [a.staleMs] inbox staleness bound (default {@link SESSION_STALE_MS})
1506
+ * @returns {{wedged:false}|{wedged:true, reason:"handoff-overdue"|"inbox-unconsumed", since?:string}}
1507
+ */
1508
+ // SECONDARY (inbox-unconsumed) is deferred — see the header above. A named
1509
+ // const, not a deleted branch: the follow-up flips this once the consume-stamp,
1510
+ // the frontDoorInbox feed and the real lane/held keying are all wired.
1511
+ export const SECONDARY_WEDGE_ENABLED = false;
1512
+
1513
+ export function sessionWedge(a = {}) {
1514
+ const o = a && typeof a === "object" ? a : {};
1515
+ const now = Number(o.now);
1516
+ if (!Number.isFinite(now)) return { wedged: false };
1517
+ const deadlineMs = Number.isFinite(o.deadlineMs) && o.deadlineMs > 0 ? o.deadlineMs : DEFAULT_HANDOFF_DEADLINE_MS;
1518
+ const staleMs = Number.isFinite(o.staleMs) ? o.staleMs : SESSION_STALE_MS;
1519
+
1520
+ // 1. PRIMARY — mirrors expireHandoffs' deadline math exactly.
1521
+ const handoffs = Array.isArray(o.handoffs) ? o.handoffs : [];
1522
+ for (const rec of handoffs) {
1523
+ if (!rec || typeof rec !== "object" || rec.status !== "open") continue;
1524
+ let deadline = Date.parse(rec.deadlineAt || "");
1525
+ if (!Number.isFinite(deadline)) {
1526
+ const enq = Date.parse(rec.enqueuedAt || "");
1527
+ deadline = Number.isFinite(enq) ? enq + deadlineMs : now; // undated → already due
1528
+ }
1529
+ if (now >= deadline) {
1530
+ return { wedged: true, reason: "handoff-overdue", since: new Date(deadline).toISOString() };
1531
+ }
1532
+ }
1533
+
1534
+ // 2. SECONDARY — DEFERRED (see header): held off until the consume-stamp,
1535
+ // the frontDoorInbox feed and the real lane/held keying are wired end-to-end.
1536
+ // Kept behind the flag so it is not shipped reading seats wedged spuriously.
1537
+ if (SECONDARY_WEDGE_ENABLED && o.frontDoor === "session" && o.sessionLive === true) {
1538
+ const inbox = Array.isArray(o.inbox) ? o.inbox : [];
1539
+ const routed = inbox.filter(isFrontDoorInboxItem);
1540
+ if (routed.length > 0) {
1541
+ const lastConsumedAt = Number.isFinite(o.lastConsumedAt) ? o.lastConsumedAt : null;
1542
+ const staleConsume = lastConsumedAt === null || now - lastConsumedAt > staleMs;
1543
+ if (staleConsume) {
1544
+ const since = lastConsumedAt !== null ? lastConsumedAt : oldestInboxAt(routed);
1545
+ return {
1546
+ wedged: true,
1547
+ reason: "inbox-unconsumed",
1548
+ ...(Number.isFinite(since) ? { since: new Date(since).toISOString() } : {}),
1549
+ };
1550
+ }
1551
+ }
1552
+ }
1553
+
1554
+ return { wedged: false };
1555
+ }
1556
+
1557
+ /** Oldest `enqueuedAt`/`ts` among inbox items, epoch ms, else NaN. */
1558
+ function oldestInboxAt(items) {
1559
+ let oldest = NaN;
1560
+ for (const it of items) {
1561
+ const ms = Date.parse((it && (it.enqueuedAt || it.ts)) || "");
1562
+ if (Number.isFinite(ms) && (Number.isNaN(oldest) || ms < oldest)) oldest = ms;
1563
+ }
1564
+ return oldest;
1565
+ }
1566
+
1398
1567
  /**
1399
1568
  * Pure: the beat's `machine.upgrade` from state/autoupdate/last.json
1400
1569
  * (`{from,to,at,ok,healthy,reason}` as autoupdate.sh writes it). Only the four
@@ -2069,6 +2238,60 @@ export async function collectStatus(o = {}) {
2069
2238
  } catch { /* no sessionNote — the beat still carries frontDoor/sessionLive */ }
2070
2239
  }
2071
2240
 
2241
+ // 4b-iii. BEATING BUT NOT READING (DARK-SEAT) — `machine.wedge`, present
2242
+ // ONLY when wedged, exactly like `sessionNote` above. The decision is the
2243
+ // pure `sessionWedge`; this is the reads that feed it — the open handoff
2244
+ // ledger — and it is fail-open in the same way every other probe here is: a
2245
+ // throw anywhere (a corrupt handoff file) drops the field, never the beat.
2246
+ // `opt.handoffs` overrides disk, mirroring `opt.attention` above, so a
2247
+ // caller that already has the ledger in memory (the daemon, a test) is never
2248
+ // made to round-trip it through the filesystem.
2249
+ //
2250
+ // The `frontDoorInbox`/`lastConsumedAt` reads that fed the SECONDARY
2251
+ // (inbox-unconsumed) signal are gone from here: that signal is DEFERRED
2252
+ // (`SECONDARY_WEDGE_ENABLED === false` in sessionWedge), so feeding it would
2253
+ // be inert work. They return when the follow-up wires the secondary.
2254
+ try {
2255
+ const handoffs = opt.handoffs !== undefined ? opt.handoffs : listHandoffs(opt.agentRoot);
2256
+ const wedge = sessionWedge({
2257
+ handoffs,
2258
+ frontDoor: machine.frontDoor,
2259
+ sessionLive: machine.sessionLive,
2260
+ now: nowMs,
2261
+ deadlineMs: opt.handoffDeadlineMs,
2262
+ });
2263
+ if (wedge && wedge.wedged) machine.wedge = { reason: wedge.reason, ...(wedge.since ? { since: wedge.since } : {}) };
2264
+ } catch { /* no wedge field — the beat still carries frontDoor/sessionLive/sessionNote */ }
2265
+
2266
+ // 4b-iv. A FAILED REVIVE — `machine.reviveNote`, present only when a revive
2267
+ // kickstarted a job and then could NOT confirm a live pid (Jacob's
2268
+ // drain-then-restart-daemon.sh exited 0 with a dead daemon). The two on-host
2269
+ // actors — the daemon's front-door revive (agent-daemon.mjs) and the hourly
2270
+ // daemon backstop (autoupdate.sh#revive_daemon) — write it to
2271
+ // state/telemetry/revive-note.json; a no-pid-after-kickstart is a FAILURE TO
2272
+ // ESCALATE and has to reach the fleet, not just this seat's log — the seat
2273
+ // needs a person. Surfaced only while fresh (REVIVE_NOTE_FRESH_MS) so a
2274
+ // recovered seat ages out; the writers refresh it every attempt and clear it
2275
+ // on a confirmed revive. `machine` is an OPEN record (like wedge/upgrade), so
2276
+ // this lands without an hq schema change. Fail-open. `opt.reviveNote`
2277
+ // overrides disk for tests.
2278
+ try {
2279
+ const rn = opt.reviveNote !== undefined
2280
+ ? opt.reviveNote
2281
+ : (opt.agentRoot ? safeReadJson(join(resolve(opt.agentRoot), REVIVE_NOTE_REL)) : null);
2282
+ if (rn && typeof rn === "object" && typeof rn.reason === "string") {
2283
+ const at = Date.parse(rn.at || "");
2284
+ const freshMs = Number.isFinite(opt.reviveNoteFreshMs) ? opt.reviveNoteFreshMs : REVIVE_NOTE_FRESH_MS;
2285
+ if (Number.isFinite(at) && nowMs - at <= freshMs) {
2286
+ machine.reviveNote = {
2287
+ reason: rn.reason,
2288
+ at: new Date(at).toISOString(),
2289
+ ...(typeof rn.target === "string" && rn.target ? { target: rn.target } : {}),
2290
+ };
2291
+ }
2292
+ }
2293
+ } catch { /* no reviveNote — the beat still carries everything else */ }
2294
+
2072
2295
  // 4c. last upgrade outcome (WP-M6) — `machine.upgrade`, absent until the
2073
2296
  // first autoupdate attempt; a corrupt file drops the field, never the beat.
2074
2297
  try {
@@ -2169,6 +2392,7 @@ export const _internals = {
2169
2392
  readSdkVersion,
2170
2393
  sessionLiveness,
2171
2394
  sessionNote,
2395
+ sessionWedge,
2172
2396
  sanitizeNoteDetail,
2173
2397
  sessionJobLabel,
2174
2398
  sessionJobInstalled,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@cohortapp/agent-sdk",
3
- "version": "2.18.15",
3
+ "version": "2.18.16",
4
4
  "description": "Cohort Agent SDK — autonomous AI colleague runtime. Deploy senior AI colleagues on dedicated Mac minis, wired to the Cohort operating surface.",
5
5
  "type": "module",
6
6
  "bin": {