@cohortapp/agent-sdk 2.18.15 → 2.18.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/org/inbound/hydrate.mjs +35 -1
- package/lib/org/inbound/project.mjs +3 -0
- package/lib/session/frontdoor.mjs +105 -8
- package/lib/session/handoffs.mjs +57 -0
- package/lib/session/inbox-claims.mjs +106 -2
- package/lib/session/revive.mjs +302 -5
- package/lib/telemetry/collect.mjs +224 -0
- package/package.json +1 -1
- package/scripts/daemon/agent-daemon.mjs +262 -10
- package/scripts/daemon/assurance.mjs +56 -1
- package/scripts/daemon/deliver.mjs +80 -0
- package/scripts/fleet/rollout.mjs +48 -4
- package/scripts/hooks/pre-write-yaml-validate.mjs +63 -2
- package/scripts/local-triggers/autoupdate.sh +287 -22
package/lib/session/revive.mjs
CHANGED
|
@@ -22,9 +22,10 @@
|
|
|
22
22
|
*
|
|
23
23
|
* The daemon is the honest home for this. It is a separate process, it is
|
|
24
24
|
* already alive on every seat that beats, it already reads the front-door
|
|
25
|
-
* state every poll to decide who owns the inbox, and it
|
|
26
|
-
*
|
|
27
|
-
*
|
|
25
|
+
* state every poll to decide who owns the inbox, and it re-checks the door on
|
|
26
|
+
* its own sixty-second cadence (REVIVE_CHECK_INTERVAL_MS). A seat whose door
|
|
27
|
+
* shuts is then measured in a minute or two rather than a day, by something
|
|
28
|
+
* the shut door cannot take down with it.
|
|
28
29
|
*
|
|
29
30
|
* Pure: the decision only. The daemon owns the kickstart.
|
|
30
31
|
*/
|
|
@@ -38,19 +39,33 @@ export const DEFAULT_REVIVE_BACKOFF_MS = 15 * 60 * 1000;
|
|
|
38
39
|
/** How many restarts before the daemon stops and leaves it to a person. */
|
|
39
40
|
export const DEFAULT_REVIVE_MAX = 3;
|
|
40
41
|
|
|
42
|
+
/**
|
|
43
|
+
* Consecutive not-live reads required before the FIRST kickstart.
|
|
44
|
+
*
|
|
45
|
+
* A kickstart drops whatever the session was doing. One bad read is not
|
|
46
|
+
* enough evidence to pay that cost — a heartbeat file can be caught
|
|
47
|
+
* mid-write, or a poll can land in the one second between a session ending
|
|
48
|
+
* a tool call and starting the next. Two reads in a row, SIXTY seconds
|
|
49
|
+
* apart at the daemon's revive-check cadence (REVIVE_CHECK_INTERVAL_MS), is
|
|
50
|
+
* the line Ravi drew after the fleet had already seen what a single-read
|
|
51
|
+
* trigger costs.
|
|
52
|
+
*/
|
|
53
|
+
export const DEFAULT_REVIVE_CONFIRM_READS = 2;
|
|
54
|
+
|
|
41
55
|
/**
|
|
42
56
|
* Should the daemon restart the front door right now?
|
|
43
57
|
*
|
|
44
58
|
* The caller supplies the front-door state it already reads for dispatch, the
|
|
45
59
|
* age of the session heartbeat, and what this daemon has already tried. Every
|
|
46
60
|
* bound is explicit because the failure mode of getting this wrong is a seat
|
|
47
|
-
* that restarts its own session every
|
|
61
|
+
* that restarts its own session every sixty seconds forever — which is worse
|
|
48
62
|
* than the shut door, and is the reason the ladder ends in "stop and say so"
|
|
49
63
|
* rather than in another attempt.
|
|
50
64
|
*
|
|
51
65
|
* @param {{frontDoor?:string, sessionLive?:boolean, silentMs?:number|null,
|
|
52
66
|
* attempts?:number, lastAttemptAt?:number|null, now:number,
|
|
53
|
-
* reviveAfterMs?:number, backoffMs?:number, maxAttempts?:number
|
|
67
|
+
* reviveAfterMs?:number, backoffMs?:number, maxAttempts?:number,
|
|
68
|
+
* notLiveReads?:number, confirmReads?:number}} a
|
|
54
69
|
* @returns {{revive:boolean, reason:string}}
|
|
55
70
|
*/
|
|
56
71
|
export function shouldReviveFrontDoor(a) {
|
|
@@ -59,6 +74,7 @@ export function shouldReviveFrontDoor(a) {
|
|
|
59
74
|
const after = Number.isFinite(x.reviveAfterMs) && x.reviveAfterMs > 0 ? x.reviveAfterMs : DEFAULT_REVIVE_AFTER_MS;
|
|
60
75
|
const backoff = Number.isFinite(x.backoffMs) && x.backoffMs > 0 ? x.backoffMs : DEFAULT_REVIVE_BACKOFF_MS;
|
|
61
76
|
const max = Number.isFinite(x.maxAttempts) && x.maxAttempts >= 0 ? x.maxAttempts : DEFAULT_REVIVE_MAX;
|
|
77
|
+
const confirm = Number.isFinite(x.confirmReads) && x.confirmReads > 0 ? x.confirmReads : DEFAULT_REVIVE_CONFIRM_READS;
|
|
62
78
|
|
|
63
79
|
// A seat whose lane is the daemon has no front door to revive: the daemon
|
|
64
80
|
// itself is answering, and restarting a session job it does not depend on
|
|
@@ -82,6 +98,16 @@ export function shouldReviveFrontDoor(a) {
|
|
|
82
98
|
if (!Number.isFinite(silentMs)) return { revive: false, reason: "silence-unknown" };
|
|
83
99
|
if (silentMs < after) return { revive: false, reason: "within-grace" };
|
|
84
100
|
|
|
101
|
+
// Two-read hysteresis, gating only the FIRST kickstart: a kickstart drops
|
|
102
|
+
// whatever the session was mid-way through, so the daemon must not act on
|
|
103
|
+
// a single not-live read before it has confirmed the verdict on the next
|
|
104
|
+
// poll. `notLiveReads` undefined means a caller that predates this streak
|
|
105
|
+
// (every existing test above) — treat it as already confirmed so those
|
|
106
|
+
// callers see no change in behaviour; the daemon is the one caller that
|
|
107
|
+
// threads the real streak through.
|
|
108
|
+
const reads = x.notLiveReads === undefined ? confirm : Number(x.notLiveReads);
|
|
109
|
+
if (!(reads >= confirm)) return { revive: false, reason: "confirming" };
|
|
110
|
+
|
|
85
111
|
const attempts = Number(x.attempts) || 0;
|
|
86
112
|
if (attempts >= max) return { revive: false, reason: "budget-spent" };
|
|
87
113
|
|
|
@@ -93,6 +119,277 @@ export function shouldReviveFrontDoor(a) {
|
|
|
93
119
|
return { revive: true, reason: `front door silent ${Math.round(silentMs / 1000)}s (attempt ${attempts + 1}/${max})` };
|
|
94
120
|
}
|
|
95
121
|
|
|
122
|
+
/**
|
|
123
|
+
* Should this seat's daemon job be restarted? Local check only — the
|
|
124
|
+
* reciprocal of `shouldReviveFrontDoor`'s job, but it CANNOT be run by the
|
|
125
|
+
* daemon on itself: a process that is alive to ask the question is a process
|
|
126
|
+
* launchctl reports a live pid for, so a live daemon can never observe its own
|
|
127
|
+
* down state. The actor is therefore the hourly `autoupdate.sh` backstop (it
|
|
128
|
+
* already parses `launchctl list` for `daemon_last_exit`), which is exactly
|
|
129
|
+
* the seat-local watchdog that catches a daemon that has stopped while the
|
|
130
|
+
* front door kept beating (Jacob's 2026-09-25 case: heartbeat.json ticking, no
|
|
131
|
+
* overdue handoff, daemon dead). `autoupdate.sh#revive_daemon` is the caller.
|
|
132
|
+
*
|
|
133
|
+
* `launchctl` is read by the caller from `launchctl list`, the same source
|
|
134
|
+
* `sibling_job_audit` and `autoupdate.sh` already parse: "pid" when the
|
|
135
|
+
* column holds a number, "dash" when it holds `-` (loaded, no pid), "absent"
|
|
136
|
+
* when the label is not in the list at all.
|
|
137
|
+
*
|
|
138
|
+
* PHANTOM PID: a silent crash loop can leave `launchctl list` holding a pid
|
|
139
|
+
* whose process has genuinely exited — the label still carries a number, but a
|
|
140
|
+
* `kill -0` on that pid fails because no such process is running, a kickstart
|
|
141
|
+
* yields another dead pid, and nothing lands on stderr. A LISTED pid is not a
|
|
142
|
+
* RUNNING daemon. So the caller decides liveness by `kill -0` on the pid and
|
|
143
|
+
* passes `pidAlive`; a phantom pid (`launchctl:"pid"` but `pidAlive:false`) is
|
|
144
|
+
* treated exactly like "dash" — loaded with no live process — and the beat age
|
|
145
|
+
* decides, instead of the listed pid rubber-stamping the daemon as alive.
|
|
146
|
+
*
|
|
147
|
+
* @param {{launchctl?:string, beatAgeMs?:number, staleMs?:number, pidAlive?:boolean}} a
|
|
148
|
+
* @returns {{revive:boolean, reason:string}}
|
|
149
|
+
*/
|
|
150
|
+
export function shouldReviveDaemon(a) {
|
|
151
|
+
const x = a && typeof a === "object" ? a : {};
|
|
152
|
+
const stale = Number.isFinite(x.staleMs) && x.staleMs > 0 ? x.staleMs : DEFAULT_REVIVE_AFTER_MS;
|
|
153
|
+
|
|
154
|
+
// A pid that launchctl lists but the OS does not actually run is a phantom —
|
|
155
|
+
// the same down state as a dash, dressed up as alive. Only an EXPLICIT
|
|
156
|
+
// `pidAlive === false` demotes it; a caller that does not probe liveness
|
|
157
|
+
// (`pidAlive` undefined) keeps the old behaviour, so nothing that never
|
|
158
|
+
// measured it changes.
|
|
159
|
+
const phantom = x.launchctl === "pid" && x.pidAlive === false;
|
|
160
|
+
const state = phantom ? "dash" : x.launchctl;
|
|
161
|
+
|
|
162
|
+
if (state === "pid") return { revive: false, reason: "daemon-alive" };
|
|
163
|
+
|
|
164
|
+
// No job at all is not a wedge to clear — it is a seat that was never
|
|
165
|
+
// provisioned with one, or had it removed, and a kickstart has nothing to
|
|
166
|
+
// aim at.
|
|
167
|
+
if (state === "absent") return { revive: false, reason: "daemon-job-absent" };
|
|
168
|
+
|
|
169
|
+
if (state === "dash") {
|
|
170
|
+
const beatAgeMs = Number(x.beatAgeMs);
|
|
171
|
+
// A loaded job with no live pid and a beat that has gone stale is the down
|
|
172
|
+
// state this exists to catch. A fresh beat under the same launchctl
|
|
173
|
+
// reading means the prior instance is still exiting, not that this one
|
|
174
|
+
// needs help.
|
|
175
|
+
if (!(beatAgeMs > stale)) return { revive: false, reason: "daemon-settling" };
|
|
176
|
+
return { revive: true, reason: phantom ? "daemon-phantom-pid" : "daemon-down" };
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
// Anything else — missing, empty, an error string from a failed parse — is
|
|
180
|
+
// a reading this cannot name a state from, and an unnamed state is never
|
|
181
|
+
// grounds to act.
|
|
182
|
+
return { revive: false, reason: "daemon-state-unknown" };
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/** How many consecutive observations across the settle window prove continuity. */
|
|
186
|
+
export const DEFAULT_CONFIRM_SAMPLES = 2;
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Did a revive attempt actually work? Checked from OUTSIDE the restart helper,
|
|
190
|
+
* on purpose: a helper that exits 0 has told you it ran, not that it succeeded,
|
|
191
|
+
* and trusting that exit code is the gap that let a shut front door sit
|
|
192
|
+
* unrestarted for days while the job meant to fix it kept reporting clean.
|
|
193
|
+
*
|
|
194
|
+
* ── WHY A ONE-SHOT PID CHECK IS NOT ENOUGH (Jacob, incident-validated) ──────
|
|
195
|
+
*
|
|
196
|
+
* A single "is there a pid, is the beat fresh" read passes two things that are
|
|
197
|
+
* NOT a revive: a PHANTOM pid (launchctl lists it, the process is dead),
|
|
198
|
+
* and a COME-UP-THEN-CRASH loop that beats once, dies, and is relaunched — its
|
|
199
|
+
* local beat file always looks fresh because the crash-looping restart keeps
|
|
200
|
+
* re-stamping it. The bar is "alive-AND-still-beating-server-side", proven two
|
|
201
|
+
* ways, either of which alone is sufficient:
|
|
202
|
+
*
|
|
203
|
+
* 1. UPTIME CONTINUITY — the one thing a crash loop cannot fake. One pid,
|
|
204
|
+
* alive on every read (`pidAlive`), the SAME process identity (start-time,
|
|
205
|
+
* `ps -o lstart`) across the whole settle window: no restart boundary. A
|
|
206
|
+
* crash loop moves its start-time on every restart, so identity drift is
|
|
207
|
+
* the tell. This alone still admits an alive-but-WEDGED process, so it is
|
|
208
|
+
* paired with a fresh presence beat read back FROM THE SERVER — not the
|
|
209
|
+
* local beat file, which the wedged/crash-looping process keeps stamping.
|
|
210
|
+
* `serverBeatFresh:false` (server reachable, beat stale = alive-but-not-
|
|
211
|
+
* publishing) blocks; `undefined` (could not reach the server) does NOT —
|
|
212
|
+
* recovery is never held hostage to org reachability, continuity carries.
|
|
213
|
+
*
|
|
214
|
+
* 2. A COMPLETED UNIT OF WORK — the STRONGER signal when it exists, and
|
|
215
|
+
* SUFFICIENT on its own (Jacob's positive control: A001 recovered dark →
|
|
216
|
+
* 12 authored posts). A crash loop cannot author a real unit of work.
|
|
217
|
+
*
|
|
218
|
+
* Work is sufficient, not necessary: a seat that recovered into an EMPTY queue
|
|
219
|
+
* does real nothing and MUST still pass on continuity alone — an assert that
|
|
220
|
+
* demanded work would fire a revive on every idle seat overnight and get
|
|
221
|
+
* itself disabled at 3am. Continuity-OR-work separates busy-recovered,
|
|
222
|
+
* crash-loop, phantom, AND quiet-recovered.
|
|
223
|
+
*
|
|
224
|
+
* ── THREE COMPLEMENTARY WITHIN-WINDOW CONDITIONS ────────────────────────────
|
|
225
|
+
*
|
|
226
|
+
* Continuity is three independent checks, each catching a failure the other two
|
|
227
|
+
* cannot: pid-recycling breaks start-time, a crash loop breaks pid, and a WRONG
|
|
228
|
+
* process (right pid, wrong binary) breaks comm. The third exists because `ps`
|
|
229
|
+
* is aliased to `pnpm start` on this fleet — a stray match on the wrong process
|
|
230
|
+
* would otherwise read as a revive. When the caller supplies `expectedComm` and
|
|
231
|
+
* a `comm` per sample (from `/bin/ps -p <pid> -o comm=`), a mismatch fails and
|
|
232
|
+
* NAMES `wrong-comm`; a null comm is unknown, not wrong, and continuity carries
|
|
233
|
+
* it. A caller that passes no `expectedComm` sees the pre-comm-check behaviour.
|
|
234
|
+
*
|
|
235
|
+
* @param {{
|
|
236
|
+
* samples?:Array<{pid?:number, alive?:boolean, startTime?:string|null, comm?:string|null}>,
|
|
237
|
+
* serverBeatFresh?:boolean, workObserved?:boolean, minSamples?:number,
|
|
238
|
+
* expectedComm?:string
|
|
239
|
+
* }} a `samples` are the per-read observations across the settle window;
|
|
240
|
+
* `serverBeatFresh` is a fresh presence beat read back from the SERVER;
|
|
241
|
+
* `workObserved` is a completed unit of work since the kickstart;
|
|
242
|
+
* `expectedComm` is the binary name the revived pid must be running.
|
|
243
|
+
* @returns {{ok:boolean, reason:string}}
|
|
244
|
+
*/
|
|
245
|
+
export function confirmRevived(a) {
|
|
246
|
+
const x = a && typeof a === "object" ? a : {};
|
|
247
|
+
|
|
248
|
+
// Work is sufficient on its own and is the strongest evidence there is — a
|
|
249
|
+
// crash loop cannot produce a completed unit of work.
|
|
250
|
+
if (x.workObserved === true) return { ok: true, reason: "revived-work" };
|
|
251
|
+
|
|
252
|
+
const min = Number.isFinite(x.minSamples) && x.minSamples > 0 ? x.minSamples : DEFAULT_CONFIRM_SAMPLES;
|
|
253
|
+
const samples = Array.isArray(x.samples) ? x.samples : null;
|
|
254
|
+
|
|
255
|
+
// Fewer reads than the settle window needs cannot establish continuity: the
|
|
256
|
+
// process may be about to crash on its next breath. Not a failure to
|
|
257
|
+
// escalate — just not yet confirmed, so the caller re-checks within budget.
|
|
258
|
+
if (!samples || samples.length < min) return { ok: false, reason: "settling" };
|
|
259
|
+
|
|
260
|
+
// A live pid must appear SOMEWHERE in the window. If none ever did — every
|
|
261
|
+
// read dead — the helper claimed a process that is not there: a phantom pid
|
|
262
|
+
// (launchctl lists it, `kill -0` says dead) or an exit-0-with-no-pid. That is
|
|
263
|
+
// the failure-to-escalate that hands the seat to a person.
|
|
264
|
+
const isAlive = (s) => s && Number.isInteger(s.pid) && s.pid > 0 && s.alive === true;
|
|
265
|
+
const aliveSamples = samples.filter(isAlive);
|
|
266
|
+
if (aliveSamples.length === 0) return { ok: false, reason: "no-pid" };
|
|
267
|
+
|
|
268
|
+
// A dead read FOLLOWED BY a live one is a daemon mid-boot, not a failed
|
|
269
|
+
// revive. The first settle sample is taken the instant the kickstart returns,
|
|
270
|
+
// before a normal slow boot has drawn its first breath — so a dead first read
|
|
271
|
+
// then an alive one is boot-in-progress. Latching that as terminal aborts a
|
|
272
|
+
// revive that was working. A window that is not yet all-alive is still
|
|
273
|
+
// SETTLING: re-check within budget rather than escalate. Only two genuinely
|
|
274
|
+
// ALIVE reads can prove — or refute — continuity.
|
|
275
|
+
if (aliveSamples.length < samples.length) return { ok: false, reason: "settling" };
|
|
276
|
+
|
|
277
|
+
// Every read is alive. Same pid AND same start-time across the window = one
|
|
278
|
+
// instance that never restarted. A restart boundary between two ALIVE reads —
|
|
279
|
+
// a different pid, or the same pid with a different start-time (PID reuse by
|
|
280
|
+
// the relaunched process) — is the fingerprint of a crash loop, never a
|
|
281
|
+
// revive. `startTime` may be null on both when `ps` was unavailable;
|
|
282
|
+
// identical-null still agrees, and the pid continuity carries the degraded
|
|
283
|
+
// case.
|
|
284
|
+
const first = aliveSamples[0];
|
|
285
|
+
const continuous = aliveSamples.every((s) => s.pid === first.pid && s.startTime === first.startTime);
|
|
286
|
+
if (!continuous) return { ok: false, reason: "restart-boundary" };
|
|
287
|
+
|
|
288
|
+
// THIRD within-window condition: the binary under the pid must be the one we
|
|
289
|
+
// kickstarted. Right pid, right start-time, WRONG process is still not a
|
|
290
|
+
// revive — and `ps` aliased to `pnpm start` on this fleet makes a stray match
|
|
291
|
+
// real. Opt-in: only when the caller supplies `expectedComm`. A null comm is
|
|
292
|
+
// unknown (ps unavailable / pid vanished), not wrong — continuity carries it,
|
|
293
|
+
// as a null start-time does; only an OBSERVED, mismatching comm fails. Matched
|
|
294
|
+
// by suffix because `ps -o comm=` prints the absolute path and the expected
|
|
295
|
+
// value is a bare binary name.
|
|
296
|
+
if (x.expectedComm) {
|
|
297
|
+
const want = String(x.expectedComm);
|
|
298
|
+
const observed = aliveSamples.filter((s) => s.comm != null);
|
|
299
|
+
const commOk = observed.every((s) => {
|
|
300
|
+
const got = String(s.comm);
|
|
301
|
+
return got === want || got.endsWith(`/${want}`) || want.endsWith(`/${got}`);
|
|
302
|
+
});
|
|
303
|
+
if (!commOk) return { ok: false, reason: "wrong-comm" };
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
// Alive and continuous, but is it actually WORKING? The local beat file lies
|
|
307
|
+
// in a wedge, so the confirming signal is the presence beat the server
|
|
308
|
+
// recorded. An explicit stale reading (reachable, not publishing) blocks; an
|
|
309
|
+
// unknown reading (unreachable) does not — continuity carries it.
|
|
310
|
+
if (x.serverBeatFresh === false) return { ok: false, reason: "beat-not-advancing" };
|
|
311
|
+
|
|
312
|
+
return { ok: true, reason: "revived" };
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
/**
|
|
316
|
+
* A confirm outcome that must STOP the revive and hand the seat to a person,
|
|
317
|
+
* rather than fire another blind kickstart. A second attempt after a crash-loop
|
|
318
|
+
* assert buries the FIRST failure's log under the loop, so these outcomes are
|
|
319
|
+
* terminal: `no-pid` (phantom / exit-0-with-no-pid), `restart-boundary`
|
|
320
|
+
* (came-up-then-crashed within the settle window), `wrong-comm` (a live pid
|
|
321
|
+
* running the wrong binary — not the process we kickstarted), and
|
|
322
|
+
* `crash-loop-cross-poll` (an unrequested restart caught a poll later — the
|
|
323
|
+
* crash loop whose period outran the settle window). `settling` and
|
|
324
|
+
* `beat-not-advancing` are "not yet" — the caller may re-check within budget.
|
|
325
|
+
*
|
|
326
|
+
* @param {{ok?:boolean, reason?:string}} confirmed
|
|
327
|
+
* @returns {boolean}
|
|
328
|
+
*/
|
|
329
|
+
export function isTerminalReviveFailure(confirmed) {
|
|
330
|
+
const r = confirmed && confirmed.reason;
|
|
331
|
+
return r === "no-pid" || r === "restart-boundary" || r === "wrong-comm" || r === "crash-loop-cross-poll";
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
/**
|
|
335
|
+
* Did the session job restart WITHOUT this daemon asking it to, BETWEEN two
|
|
336
|
+
* daemon polls? The unbounded companion to `confirmRevived`'s within-window
|
|
337
|
+
* continuity check.
|
|
338
|
+
*
|
|
339
|
+
* ── WHY WITHIN-WINDOW CONTINUITY IS NOT ENOUGH ──────────────────────────────
|
|
340
|
+
*
|
|
341
|
+
* `confirmRevived` proves continuity across a SETTLE WINDOW (a few seconds). A
|
|
342
|
+
* crash loop whose period EXCEEDS that window shows the SAME identity twice
|
|
343
|
+
* inside it — two alive reads, same pid, same start-time — and false-confirms
|
|
344
|
+
* "revived". A 25s crash-loop period against a 5s settle window is invisible to
|
|
345
|
+
* a within-window check, and WIDENING the window only relocates the threshold:
|
|
346
|
+
* a 65s period beats a 60s window just as cleanly.
|
|
347
|
+
*
|
|
348
|
+
* The fix has no threshold to tune. The daemon persists the last-confirmed
|
|
349
|
+
* {pid, startTime} between its 60s poll iterations. On the NEXT poll, if the job
|
|
350
|
+
* holds a DIFFERENT identity (a different pid, OR the same pid with a moved
|
|
351
|
+
* start-time from PID reuse by the relaunched process) AND this daemon issued NO
|
|
352
|
+
* kickstart in the interval, that is an UNREQUESTED restart — the crash loop, at
|
|
353
|
+
* ANY period. A changed identity WITH a kickstart in between is exactly what a
|
|
354
|
+
* legitimate revive looks like and is never flagged.
|
|
355
|
+
*
|
|
356
|
+
* Pure: the verdict only. The daemon owns the persistence, the kickstart flag,
|
|
357
|
+
* and the halt.
|
|
358
|
+
*
|
|
359
|
+
* @param {{prev?:{pid?:number, startTime?:string|null}|null,
|
|
360
|
+
* curr?:{pid?:number, startTime?:string|null}|null,
|
|
361
|
+
* kickstartIssuedSince?:boolean}} a
|
|
362
|
+
* `prev` is the last poll's confirmed identity (null on the first poll);
|
|
363
|
+
* `curr` is this poll's observed identity; `kickstartIssuedSince` is whether
|
|
364
|
+
* THIS daemon kickstarted the job between the two polls.
|
|
365
|
+
* @returns {{restarted:boolean, reason:string}}
|
|
366
|
+
*/
|
|
367
|
+
export function detectUnrequestedRestart(a) {
|
|
368
|
+
const x = a && typeof a === "object" ? a : {};
|
|
369
|
+
const prev = x.prev && typeof x.prev === "object" ? x.prev : null;
|
|
370
|
+
const curr = x.curr && typeof x.curr === "object" ? x.curr : null;
|
|
371
|
+
|
|
372
|
+
// Nothing to compare against yet — the first poll after boot establishes the
|
|
373
|
+
// baseline, it cannot judge drift from one.
|
|
374
|
+
if (!prev || !(Number.isInteger(prev.pid) && prev.pid > 0)) return { restarted: false, reason: "no-prior" };
|
|
375
|
+
|
|
376
|
+
// No live identity this poll is a DOWN door, not an identity change — that is
|
|
377
|
+
// the revive ladder's job. This function speaks only to drift between two
|
|
378
|
+
// observed identities.
|
|
379
|
+
if (!curr || !(Number.isInteger(curr.pid) && curr.pid > 0)) return { restarted: false, reason: "no-current" };
|
|
380
|
+
|
|
381
|
+
const sameIdentity = curr.pid === prev.pid && curr.startTime === prev.startTime;
|
|
382
|
+
if (sameIdentity) return { restarted: false, reason: "continuous" };
|
|
383
|
+
|
|
384
|
+
// Identity changed. If this daemon asked for it, that is a legitimate revive,
|
|
385
|
+
// not the loop.
|
|
386
|
+
if (x.kickstartIssuedSince === true) return { restarted: false, reason: "kickstart-issued" };
|
|
387
|
+
|
|
388
|
+
// Changed identity, no kickstart in between: an unrequested restart — the
|
|
389
|
+
// crash loop, at whatever period it runs.
|
|
390
|
+
return { restarted: true, reason: "unrequested-restart" };
|
|
391
|
+
}
|
|
392
|
+
|
|
96
393
|
/**
|
|
97
394
|
* The command that restarts the front-door job. Pure — the caller runs it.
|
|
98
395
|
*
|
|
@@ -172,6 +172,7 @@ import { snapshot as countersSnapshot } from "../diagnostics/counters.mjs";
|
|
|
172
172
|
import { replyDebtFromCounters } from "../daemon/reply-debt.mjs";
|
|
173
173
|
import { summarisePinnedDrift, countPins } from "../upgrade/pinned-drift.mjs";
|
|
174
174
|
import { IGNORED_DRIFT_REL } from "../upgrade/ignored-drift.mjs";
|
|
175
|
+
import { listHandoffs, DEFAULT_HANDOFF_DEADLINE_MS } from "../session/handoffs.mjs";
|
|
175
176
|
|
|
176
177
|
/** Default temperature probe ceiling — used only as a guard in alerts; here we just report. */
|
|
177
178
|
const VALID_STATES = new Set(["active", "idle", "busy", "error", "offline"]);
|
|
@@ -1135,6 +1136,44 @@ async function collectIdentity(o, osImpl) {
|
|
|
1135
1136
|
/** A main-session heartbeat older than this is not live. */
|
|
1136
1137
|
export const SESSION_STALE_MS = 90_000;
|
|
1137
1138
|
|
|
1139
|
+
/**
|
|
1140
|
+
* A FAILED REVIVE, written by the on-host actors (the daemon's front-door
|
|
1141
|
+
* revive and `autoupdate.sh#revive_daemon`) when a kickstart returned but no
|
|
1142
|
+
* live pid could be confirmed — a failure to escalate, which has to reach the
|
|
1143
|
+
* fleet, not just the seat's log. Relative to the agent root; fresh window
|
|
1144
|
+
* bounds how long it rides the beat after the last failed attempt.
|
|
1145
|
+
*/
|
|
1146
|
+
export const REVIVE_NOTE_REL = "state/telemetry/revive-note.json";
|
|
1147
|
+
export const REVIVE_NOTE_FRESH_MS = 90 * 60 * 1000;
|
|
1148
|
+
|
|
1149
|
+
/**
|
|
1150
|
+
* Pure: a revive-note that NAMES budget exhaustion, or null.
|
|
1151
|
+
*
|
|
1152
|
+
* A separate legibility defect from a failed revive: when the revive ladder
|
|
1153
|
+
* spends its whole restart budget and STOPS (`shouldReviveFrontDoor` →
|
|
1154
|
+
* `budget-spent`), nothing is written — the seat goes quiet and a quiet system
|
|
1155
|
+
* looks healthy. A ladder that gave up is a failure that does not name itself.
|
|
1156
|
+
* This turns that silence into a beat note carried through the SAME channel as
|
|
1157
|
+
* `machine.reviveNote`, so the beat says "stopped because budget spent" rather
|
|
1158
|
+
* than nothing. Only the budget-spent verdict qualifies — any other verdict
|
|
1159
|
+
* (still confirming, backing off, answering) is not the ladder giving up.
|
|
1160
|
+
*
|
|
1161
|
+
* @param {{verdict?:string, target?:string, attempts?:number, at?:string}} a
|
|
1162
|
+
* @returns {{reason:string, target?:string, detail:string, at:string}|null}
|
|
1163
|
+
*/
|
|
1164
|
+
export function budgetSpentNote(a = {}) {
|
|
1165
|
+
const x = a && typeof a === "object" ? a : {};
|
|
1166
|
+
if (x.verdict !== "budget-spent") return null;
|
|
1167
|
+
const attempts = Number.isFinite(x.attempts) ? x.attempts : 0;
|
|
1168
|
+
const at = typeof x.at === "string" && x.at ? x.at : new Date().toISOString();
|
|
1169
|
+
return {
|
|
1170
|
+
reason: "revive-budget-spent",
|
|
1171
|
+
...(typeof x.target === "string" && x.target ? { target: x.target } : {}),
|
|
1172
|
+
detail: `revive budget spent after ${attempts} attempts`,
|
|
1173
|
+
at,
|
|
1174
|
+
};
|
|
1175
|
+
}
|
|
1176
|
+
|
|
1138
1177
|
/**
|
|
1139
1178
|
* Pure: derive `{frontDoor, sessionLive}` from the heartbeat the main session's
|
|
1140
1179
|
* feed writes every 15 s (`state/session/heartbeat.json` `{pid, ppid, sessionId,
|
|
@@ -1395,6 +1434,136 @@ export function sessionNote(a = {}) {
|
|
|
1395
1434
|
};
|
|
1396
1435
|
}
|
|
1397
1436
|
|
|
1437
|
+
/**
|
|
1438
|
+
* Front-door-routed inbox lanes. `sweep`, `held` and `deferred` are the
|
|
1439
|
+
* consumer's OWN maintenance lanes — items it parked on purpose, not items
|
|
1440
|
+
* waiting on a person — and a `cohort-held-*` id is the sibling hold-queue
|
|
1441
|
+
* shape (see meeting-capture's `kind==HOLD`). Neither is evidence the front
|
|
1442
|
+
* door is failing to read; counting them would wedge a seat that is doing
|
|
1443
|
+
* exactly what it was told to do.
|
|
1444
|
+
*/
|
|
1445
|
+
const WEDGE_EXCLUDED_LANES = new Set(["sweep", "held", "deferred"]);
|
|
1446
|
+
const WEDGE_HELD_ID_RE = /^cohort-held-/;
|
|
1447
|
+
|
|
1448
|
+
/** An inbox item this file's `sessionWedge` treats as front-door-routed. */
|
|
1449
|
+
function isFrontDoorInboxItem(item) {
|
|
1450
|
+
if (!item || typeof item !== "object") return false;
|
|
1451
|
+
const id = typeof item.id === "string" ? item.id : "";
|
|
1452
|
+
if (WEDGE_HELD_ID_RE.test(id)) return false;
|
|
1453
|
+
const lane = typeof item.lane === "string" ? item.lane : "";
|
|
1454
|
+
if (WEDGE_EXCLUDED_LANES.has(lane)) return false;
|
|
1455
|
+
return true;
|
|
1456
|
+
}
|
|
1457
|
+
|
|
1458
|
+
/**
|
|
1459
|
+
* Pure: the wedge verdict — "beating but not reading". A seat can have a
|
|
1460
|
+
* perfectly live heartbeat while the process behind it has stopped actually
|
|
1461
|
+
* answering inbound (wedged on a dialog with no watchdog for THIS failure
|
|
1462
|
+
* mode, a deadlocked event loop, anything short of the process dying). Two
|
|
1463
|
+
* independent signals, checked in order, because they catch different
|
|
1464
|
+
* failures and the first is the harder evidence:
|
|
1465
|
+
*
|
|
1466
|
+
* 1. PRIMARY — an un-acked handoff (`lib/session/handoffs.mjs`) past its
|
|
1467
|
+
* deadline. The consumer handed a cadence tick to the front door and
|
|
1468
|
+
* nobody acked it in time; this is true regardless of which front door is
|
|
1469
|
+
* nominally live, because a handoff is only ever written TO whichever
|
|
1470
|
+
* door claimed the tick. Deadline math mirrors `expireHandoffs` exactly
|
|
1471
|
+
* (`deadlineAt`, else `enqueuedAt + deadlineMs`) so the two never
|
|
1472
|
+
* disagree about which handoff is overdue.
|
|
1473
|
+
* 2. SECONDARY — inbox-unconsumed. DEFERRED and HELD OFF in this build
|
|
1474
|
+
* (`SECONDARY_WEDGE_ENABLED === false`), because it cannot yet fire
|
|
1475
|
+
* correctly end-to-end (Hannah, PR #70 review #2). Three things must land
|
|
1476
|
+
* together first, NONE of them in this PR:
|
|
1477
|
+
* (i) a consume-stamp on EVERY front-door path — today only
|
|
1478
|
+
* `ackHandoff` stamps `lastConsumedAt`, so it advances only on
|
|
1479
|
+
* handoff acks (~every 30 min), never on the inbound
|
|
1480
|
+
* claim/reply/done path in session-runtime;
|
|
1481
|
+
* (ii) a real front-door inbox reader feeding `frontDoorInbox` into
|
|
1482
|
+
* `collectStatus` — nothing feeds it, so `inbox` is always `[]`;
|
|
1483
|
+
* (iii) lane/held keying matched to the real item shape — on this fleet
|
|
1484
|
+
* inbox items carry `channel`/`kind`/`claimed_by`, not `lane`, and
|
|
1485
|
+
* no id begins `cohort-held-`, so `isFrontDoorInboxItem`'s
|
|
1486
|
+
* exclusion is keyed on fields the items do not carry.
|
|
1487
|
+
* Shipped live at the 90s bound it would read most seats wedged, and
|
|
1488
|
+
* `rollout`'s `done` (which requires zero wedged) would never clear. The
|
|
1489
|
+
* branch, `isFrontDoorInboxItem` and `WEDGE_EXCLUDED_LANES` are kept
|
|
1490
|
+
* intact behind the flag so the follow-up wires them rather than rebuilds
|
|
1491
|
+
* them; the PRIMARY handoff-overdue tell is the real signal and is fully
|
|
1492
|
+
* live.
|
|
1493
|
+
*
|
|
1494
|
+
* Defensive by construction: every read is behind a type check, nothing here
|
|
1495
|
+
* touches a clock or the filesystem, and no input shape can make it throw.
|
|
1496
|
+
*
|
|
1497
|
+
* @param {object} [a]
|
|
1498
|
+
* @param {object[]} [a.handoffs] `listHandoffs()` records (or equivalent)
|
|
1499
|
+
* @param {object[]} [a.inbox] raw inbox items, each `{id?, lane?}`
|
|
1500
|
+
* @param {number|null} [a.lastConsumedAt] epoch ms, or null/undefined = never
|
|
1501
|
+
* @param {"session"|"daemon"} [a.frontDoor]
|
|
1502
|
+
* @param {boolean} [a.sessionLive]
|
|
1503
|
+
* @param {number} [a.now] epoch ms
|
|
1504
|
+
* @param {number} [a.deadlineMs] handoff deadline (default {@link DEFAULT_HANDOFF_DEADLINE_MS})
|
|
1505
|
+
* @param {number} [a.staleMs] inbox staleness bound (default {@link SESSION_STALE_MS})
|
|
1506
|
+
* @returns {{wedged:false}|{wedged:true, reason:"handoff-overdue"|"inbox-unconsumed", since?:string}}
|
|
1507
|
+
*/
|
|
1508
|
+
// SECONDARY (inbox-unconsumed) is deferred — see the header above. A named
|
|
1509
|
+
// const, not a deleted branch: the follow-up flips this once the consume-stamp,
|
|
1510
|
+
// the frontDoorInbox feed and the real lane/held keying are all wired.
|
|
1511
|
+
export const SECONDARY_WEDGE_ENABLED = false;
|
|
1512
|
+
|
|
1513
|
+
export function sessionWedge(a = {}) {
|
|
1514
|
+
const o = a && typeof a === "object" ? a : {};
|
|
1515
|
+
const now = Number(o.now);
|
|
1516
|
+
if (!Number.isFinite(now)) return { wedged: false };
|
|
1517
|
+
const deadlineMs = Number.isFinite(o.deadlineMs) && o.deadlineMs > 0 ? o.deadlineMs : DEFAULT_HANDOFF_DEADLINE_MS;
|
|
1518
|
+
const staleMs = Number.isFinite(o.staleMs) ? o.staleMs : SESSION_STALE_MS;
|
|
1519
|
+
|
|
1520
|
+
// 1. PRIMARY — mirrors expireHandoffs' deadline math exactly.
|
|
1521
|
+
const handoffs = Array.isArray(o.handoffs) ? o.handoffs : [];
|
|
1522
|
+
for (const rec of handoffs) {
|
|
1523
|
+
if (!rec || typeof rec !== "object" || rec.status !== "open") continue;
|
|
1524
|
+
let deadline = Date.parse(rec.deadlineAt || "");
|
|
1525
|
+
if (!Number.isFinite(deadline)) {
|
|
1526
|
+
const enq = Date.parse(rec.enqueuedAt || "");
|
|
1527
|
+
deadline = Number.isFinite(enq) ? enq + deadlineMs : now; // undated → already due
|
|
1528
|
+
}
|
|
1529
|
+
if (now >= deadline) {
|
|
1530
|
+
return { wedged: true, reason: "handoff-overdue", since: new Date(deadline).toISOString() };
|
|
1531
|
+
}
|
|
1532
|
+
}
|
|
1533
|
+
|
|
1534
|
+
// 2. SECONDARY — DEFERRED (see header): held off until the consume-stamp,
|
|
1535
|
+
// the frontDoorInbox feed and the real lane/held keying are wired end-to-end.
|
|
1536
|
+
// Kept behind the flag so it is not shipped reading seats wedged spuriously.
|
|
1537
|
+
if (SECONDARY_WEDGE_ENABLED && o.frontDoor === "session" && o.sessionLive === true) {
|
|
1538
|
+
const inbox = Array.isArray(o.inbox) ? o.inbox : [];
|
|
1539
|
+
const routed = inbox.filter(isFrontDoorInboxItem);
|
|
1540
|
+
if (routed.length > 0) {
|
|
1541
|
+
const lastConsumedAt = Number.isFinite(o.lastConsumedAt) ? o.lastConsumedAt : null;
|
|
1542
|
+
const staleConsume = lastConsumedAt === null || now - lastConsumedAt > staleMs;
|
|
1543
|
+
if (staleConsume) {
|
|
1544
|
+
const since = lastConsumedAt !== null ? lastConsumedAt : oldestInboxAt(routed);
|
|
1545
|
+
return {
|
|
1546
|
+
wedged: true,
|
|
1547
|
+
reason: "inbox-unconsumed",
|
|
1548
|
+
...(Number.isFinite(since) ? { since: new Date(since).toISOString() } : {}),
|
|
1549
|
+
};
|
|
1550
|
+
}
|
|
1551
|
+
}
|
|
1552
|
+
}
|
|
1553
|
+
|
|
1554
|
+
return { wedged: false };
|
|
1555
|
+
}
|
|
1556
|
+
|
|
1557
|
+
/** Oldest `enqueuedAt`/`ts` among inbox items, epoch ms, else NaN. */
|
|
1558
|
+
function oldestInboxAt(items) {
|
|
1559
|
+
let oldest = NaN;
|
|
1560
|
+
for (const it of items) {
|
|
1561
|
+
const ms = Date.parse((it && (it.enqueuedAt || it.ts)) || "");
|
|
1562
|
+
if (Number.isFinite(ms) && (Number.isNaN(oldest) || ms < oldest)) oldest = ms;
|
|
1563
|
+
}
|
|
1564
|
+
return oldest;
|
|
1565
|
+
}
|
|
1566
|
+
|
|
1398
1567
|
/**
|
|
1399
1568
|
* Pure: the beat's `machine.upgrade` from state/autoupdate/last.json
|
|
1400
1569
|
* (`{from,to,at,ok,healthy,reason}` as autoupdate.sh writes it). Only the four
|
|
@@ -2069,6 +2238,60 @@ export async function collectStatus(o = {}) {
|
|
|
2069
2238
|
} catch { /* no sessionNote — the beat still carries frontDoor/sessionLive */ }
|
|
2070
2239
|
}
|
|
2071
2240
|
|
|
2241
|
+
// 4b-iii. BEATING BUT NOT READING (DARK-SEAT) — `machine.wedge`, present
|
|
2242
|
+
// ONLY when wedged, exactly like `sessionNote` above. The decision is the
|
|
2243
|
+
// pure `sessionWedge`; this is the reads that feed it — the open handoff
|
|
2244
|
+
// ledger — and it is fail-open in the same way every other probe here is: a
|
|
2245
|
+
// throw anywhere (a corrupt handoff file) drops the field, never the beat.
|
|
2246
|
+
// `opt.handoffs` overrides disk, mirroring `opt.attention` above, so a
|
|
2247
|
+
// caller that already has the ledger in memory (the daemon, a test) is never
|
|
2248
|
+
// made to round-trip it through the filesystem.
|
|
2249
|
+
//
|
|
2250
|
+
// The `frontDoorInbox`/`lastConsumedAt` reads that fed the SECONDARY
|
|
2251
|
+
// (inbox-unconsumed) signal are gone from here: that signal is DEFERRED
|
|
2252
|
+
// (`SECONDARY_WEDGE_ENABLED === false` in sessionWedge), so feeding it would
|
|
2253
|
+
// be inert work. They return when the follow-up wires the secondary.
|
|
2254
|
+
try {
|
|
2255
|
+
const handoffs = opt.handoffs !== undefined ? opt.handoffs : listHandoffs(opt.agentRoot);
|
|
2256
|
+
const wedge = sessionWedge({
|
|
2257
|
+
handoffs,
|
|
2258
|
+
frontDoor: machine.frontDoor,
|
|
2259
|
+
sessionLive: machine.sessionLive,
|
|
2260
|
+
now: nowMs,
|
|
2261
|
+
deadlineMs: opt.handoffDeadlineMs,
|
|
2262
|
+
});
|
|
2263
|
+
if (wedge && wedge.wedged) machine.wedge = { reason: wedge.reason, ...(wedge.since ? { since: wedge.since } : {}) };
|
|
2264
|
+
} catch { /* no wedge field — the beat still carries frontDoor/sessionLive/sessionNote */ }
|
|
2265
|
+
|
|
2266
|
+
// 4b-iv. A FAILED REVIVE — `machine.reviveNote`, present only when a revive
|
|
2267
|
+
// kickstarted a job and then could NOT confirm a live pid (Jacob's
|
|
2268
|
+
// drain-then-restart-daemon.sh exited 0 with a dead daemon). The two on-host
|
|
2269
|
+
// actors — the daemon's front-door revive (agent-daemon.mjs) and the hourly
|
|
2270
|
+
// daemon backstop (autoupdate.sh#revive_daemon) — write it to
|
|
2271
|
+
// state/telemetry/revive-note.json; a no-pid-after-kickstart is a FAILURE TO
|
|
2272
|
+
// ESCALATE and has to reach the fleet, not just this seat's log — the seat
|
|
2273
|
+
// needs a person. Surfaced only while fresh (REVIVE_NOTE_FRESH_MS) so a
|
|
2274
|
+
// recovered seat ages out; the writers refresh it every attempt and clear it
|
|
2275
|
+
// on a confirmed revive. `machine` is an OPEN record (like wedge/upgrade), so
|
|
2276
|
+
// this lands without an hq schema change. Fail-open. `opt.reviveNote`
|
|
2277
|
+
// overrides disk for tests.
|
|
2278
|
+
try {
|
|
2279
|
+
const rn = opt.reviveNote !== undefined
|
|
2280
|
+
? opt.reviveNote
|
|
2281
|
+
: (opt.agentRoot ? safeReadJson(join(resolve(opt.agentRoot), REVIVE_NOTE_REL)) : null);
|
|
2282
|
+
if (rn && typeof rn === "object" && typeof rn.reason === "string") {
|
|
2283
|
+
const at = Date.parse(rn.at || "");
|
|
2284
|
+
const freshMs = Number.isFinite(opt.reviveNoteFreshMs) ? opt.reviveNoteFreshMs : REVIVE_NOTE_FRESH_MS;
|
|
2285
|
+
if (Number.isFinite(at) && nowMs - at <= freshMs) {
|
|
2286
|
+
machine.reviveNote = {
|
|
2287
|
+
reason: rn.reason,
|
|
2288
|
+
at: new Date(at).toISOString(),
|
|
2289
|
+
...(typeof rn.target === "string" && rn.target ? { target: rn.target } : {}),
|
|
2290
|
+
};
|
|
2291
|
+
}
|
|
2292
|
+
}
|
|
2293
|
+
} catch { /* no reviveNote — the beat still carries everything else */ }
|
|
2294
|
+
|
|
2072
2295
|
// 4c. last upgrade outcome (WP-M6) — `machine.upgrade`, absent until the
|
|
2073
2296
|
// first autoupdate attempt; a corrupt file drops the field, never the beat.
|
|
2074
2297
|
try {
|
|
@@ -2169,6 +2392,7 @@ export const _internals = {
|
|
|
2169
2392
|
readSdkVersion,
|
|
2170
2393
|
sessionLiveness,
|
|
2171
2394
|
sessionNote,
|
|
2395
|
+
sessionWedge,
|
|
2172
2396
|
sanitizeNoteDetail,
|
|
2173
2397
|
sessionJobLabel,
|
|
2174
2398
|
sessionJobInstalled,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@cohortapp/agent-sdk",
|
|
3
|
-
"version": "2.18.
|
|
3
|
+
"version": "2.18.17",
|
|
4
4
|
"description": "Cohort Agent SDK — autonomous AI colleague runtime. Deploy senior AI colleagues on dedicated Mac minis, wired to the Cohort operating surface.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|