@cohortapp/agent-sdk 2.18.12 → 2.18.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/maestro.mjs +38 -1
- package/docs/runbooks/fleet-rollout.md +14 -7
- package/docs/runbooks/recovery-and-failover.md +18 -0
- package/lib/cadence-failure-class.mjs +245 -0
- package/lib/claude-bin.mjs +26 -7
- package/lib/cli/doctor-checks.mjs +149 -1
- package/lib/comms/send-gate.mjs +6 -4
- package/lib/diagnostics/alerts.mjs +33 -0
- package/lib/engine/agents/usage.mjs +45 -0
- package/lib/engine/budget.mjs +293 -29
- package/lib/engine/cli.mjs +54 -5
- package/lib/engine/loop.mjs +30 -0
- package/lib/engine/output/json.mjs +26 -0
- package/lib/engine/wire/errors.mjs +179 -0
- package/lib/engine/wire/search.mjs +44 -8
- package/lib/identity/claude-md.mjs +107 -0
- package/lib/identity/disclosure-instructions.mjs +148 -0
- package/lib/identity/disclosure-scrub.mjs +207 -0
- package/lib/identity/persona.mjs +141 -6
- package/lib/org/inbound/conversation-frame.mjs +289 -0
- package/lib/org/inbound/directedness.mjs +27 -7
- package/lib/org/quota.mjs +27 -0
- package/lib/session/config.mjs +4 -0
- package/lib/session/identity.mjs +71 -7
- package/lib/session/launch-failure.mjs +251 -0
- package/lib/session/resume-target.mjs +86 -0
- package/lib/telemetry/collect.mjs +129 -0
- package/lib/upgrade/pinned-drift.mjs +467 -0
- package/package.json +1 -1
- package/plugins/maestro-skills/skills/persona-discipline.md +24 -2
- package/scaffold/config/alerts.yaml +7 -0
- package/scripts/ci/check-cadence-prompts-exist.mjs +96 -0
- package/scripts/ci/check.mjs +3 -0
- package/scripts/ci/run-tests.mjs +16 -2
- package/scripts/daemon/cadence-consumer.mjs +281 -34
- package/scripts/daemon/context-compiler.mjs +9 -1
- package/scripts/daemon/prompt-builder.mjs +219 -137
- package/scripts/daemon/responder.mjs +226 -26
- package/scripts/emergency-stop.sh +114 -13
- package/scripts/fleet/rollout.mjs +10 -3
- package/scripts/healthcheck.sh +131 -33
- package/scripts/resume-operations.sh +101 -6
- package/scripts/session/supervisor.mjs +198 -5
|
@@ -480,7 +480,11 @@ function yes(surface, reason, extra = {}) {
|
|
|
480
480
|
* @param {import("./facts.mjs").Facts} facts
|
|
481
481
|
* @param {{enabled?: Record<string,boolean>, meAliases?: string[]}} [opts]
|
|
482
482
|
* `meAliases` — additional identity strings for this seat (slug, member id,
|
|
483
|
-
* agent email).
|
|
483
|
+
* agent email). Widens own-echo suppression AND, as of TD-001, is also
|
|
484
|
+
* load-bearing for reach: branch (5) matches a structured mentions[] row
|
|
485
|
+
* against this same alias set, not against bare `me`, so a seat called with
|
|
486
|
+
* its slug still resolves a mentions[] row keyed by CUID. A caller that
|
|
487
|
+
* omits meAliases here loses that match, not just echo suppression.
|
|
484
488
|
* @returns {{directed:boolean, surface:string|null, reason:string, why:string}}
|
|
485
489
|
*/
|
|
486
490
|
export function resolveDirected(cand, me, facts = {}, opts = {}) {
|
|
@@ -535,7 +539,7 @@ export function resolveDirected(cand, me, facts = {}, opts = {}) {
|
|
|
535
539
|
|
|
536
540
|
switch (cand.family) {
|
|
537
541
|
case "messaging":
|
|
538
|
-
return resolveMessaging(cand, meId, facts, on);
|
|
542
|
+
return resolveMessaging(cand, meId, facts, on, meAliases);
|
|
539
543
|
case "calling":
|
|
540
544
|
return resolveCalling(cand, meId, facts, on);
|
|
541
545
|
case "board":
|
|
@@ -603,7 +607,7 @@ function resolveCalendar(cand, me, facts, on) {
|
|
|
603
607
|
* 6. thread root I have spoken in / been named in → `thread_reply`
|
|
604
608
|
* 7. otherwise → ambient chatter, not directed
|
|
605
609
|
*/
|
|
606
|
-
function resolveMessaging(cand, me, facts, on) {
|
|
610
|
+
function resolveMessaging(cand, me, facts, on, meAliases) {
|
|
607
611
|
const channelId = cand.ids.channelId;
|
|
608
612
|
const visible = asSet(facts.visibleChannelIds);
|
|
609
613
|
const members = asSet(facts.memberChannelIds);
|
|
@@ -656,8 +660,20 @@ function resolveMessaging(cand, me, facts, on) {
|
|
|
656
660
|
|
|
657
661
|
// (5) An explicit @mention of my member id. The mention list is NOT in the
|
|
658
662
|
// chain payload; it comes from `messaging.history`, which hq ACLs to rooms
|
|
659
|
-
// I may read.
|
|
660
|
-
|
|
663
|
+
// I may read. mentions[] holds member CUIDs, but the caller cannot always
|
|
664
|
+
// know which namespace it handed us as `me` (slug vs cuid) — so match the
|
|
665
|
+
// row against the SAME alias set the own-echo drop uses, not bare `me`.
|
|
666
|
+
// Otherwise a seat called with its slug never matches a CUID mention row,
|
|
667
|
+
// and the prose fallback below (now gated on an empty mentions[]) is the
|
|
668
|
+
// only remaining catch — which the gate removes. Fall back to bare `me`
|
|
669
|
+
// if no alias set was threaded through.
|
|
670
|
+
const mentionRow = msg && Array.isArray(msg.mentions) ? msg.mentions : null;
|
|
671
|
+
const mentionsMe =
|
|
672
|
+
mentionRow &&
|
|
673
|
+
(meAliases instanceof Set
|
|
674
|
+
? mentionRow.some((m) => meAliases.has(String(m).trim()))
|
|
675
|
+
: me && mentionRow.map(String).includes(me));
|
|
676
|
+
if (mentionsMe) {
|
|
661
677
|
return on("mention") ? yes("mention", "mention") : no("surface_disabled");
|
|
662
678
|
}
|
|
663
679
|
|
|
@@ -670,8 +686,12 @@ function resolveMessaging(cand, me, facts, on) {
|
|
|
670
686
|
|
|
671
687
|
// A mention by DISPLAY NAME, when the org has no structured mention row (an
|
|
672
688
|
// agent addressed as "@Isla" in prose). Only ever consulted for a space I can
|
|
673
|
-
// already read,
|
|
674
|
-
|
|
689
|
+
// already read, only when the caller supplied its own names, and — per
|
|
690
|
+
// matchesMyName's own contract — only as a LAST RESORT: a populated
|
|
691
|
+
// mentions[] row means hq already told us who this was for, and prose
|
|
692
|
+
// naming someone else in the same breath must not override that.
|
|
693
|
+
const hasStructuredMentions = msg && Array.isArray(msg.mentions) && msg.mentions.length > 0;
|
|
694
|
+
if (msg && !hasStructuredMentions && matchesMyName(facts, msg.body ?? msg.text)) {
|
|
675
695
|
return on("mention") ? yes("mention", "named") : no("surface_disabled");
|
|
676
696
|
}
|
|
677
697
|
|
package/lib/org/quota.mjs
CHANGED
|
@@ -22,6 +22,20 @@
|
|
|
22
22
|
* Money stays a decimal string of integer micro-USD end to end; nothing here
|
|
23
23
|
* turns it into a float except the display helpers budget-guard calls.
|
|
24
24
|
*
|
|
25
|
+
* ## The one import from `lib/engine`, and why it is deliberate (W16)
|
|
26
|
+
*
|
|
27
|
+
* `readEnforcementState` / `readQuotaSourceState` below come from
|
|
28
|
+
* `lib/engine/wire/errors.mjs` — the only non-test import from `lib/org` into
|
|
29
|
+
* `lib/engine` in this repo. It is a considered choice, not an accident:
|
|
30
|
+
* `parseCohortSseFrame` here and `readCohortFrame` there are two parsers of the
|
|
31
|
+
* SAME `event: cohort` frame, and the §4.8 vocabularies they read
|
|
32
|
+
* (`off|shadow|on`, `ledger|unreadable|not_read|not_admitted`) are exactly the
|
|
33
|
+
* kind of pinned value set that drifts when it is written down twice — a
|
|
34
|
+
* gateway that adds a fourth mode would then reach one parser as a new word and
|
|
35
|
+
* the other as silence. The alternative was a duplicated set, i.e. the drift.
|
|
36
|
+
* If this layering is ever unwanted, move those two readers to a leaf module
|
|
37
|
+
* both layers may import; do NOT re-declare the value sets here.
|
|
38
|
+
*
|
|
25
39
|
* @module lib/org/quota
|
|
26
40
|
*/
|
|
27
41
|
|
|
@@ -31,6 +45,11 @@ import { existsSync, readFileSync, mkdirSync } from "node:fs";
|
|
|
31
45
|
import { join, dirname, resolve } from "node:path";
|
|
32
46
|
|
|
33
47
|
import { writeJsonAtomic } from "../fs-atomic.mjs";
|
|
48
|
+
// W16: the §4.8 vocabularies for `enforcement` and `quota_source` are parsed in
|
|
49
|
+
// exactly one place, the engine's wire reader. Both helpers are pure and total.
|
|
50
|
+
// Restating the value sets here is how the two frame parsers in this repo would
|
|
51
|
+
// start disagreeing about what the gateway said.
|
|
52
|
+
import { readEnforcementState, readQuotaSourceState } from "../engine/wire/errors.mjs";
|
|
34
53
|
|
|
35
54
|
/** The cache may never be older than this (the pre-spawn gate reads it). */
|
|
36
55
|
export const QUOTA_MAX_TTL_MS = 30_000;
|
|
@@ -192,6 +211,14 @@ export function parseCohortSseFrame(input) {
|
|
|
192
211
|
requestId: str(obj.request_id) || str(obj.requestId),
|
|
193
212
|
modelTier: str(obj.model_tier) || str(obj.modelTier),
|
|
194
213
|
costMicros: micros(obj.cost_micros ?? obj.costMicros),
|
|
214
|
+
// W16 (hq a60b9595, design §4.8). `enforcement` qualifies `costMicros`:
|
|
215
|
+
// only under "on" did money move. `shadowCostMicros` is what enforcement
|
|
216
|
+
// WOULD have charged and is never a charge. `quotaSource` says why `quota`
|
|
217
|
+
// is empty when it is. Unrecognised or absent values read as null — an
|
|
218
|
+
// unknown `enforcement` is never promoted to "on".
|
|
219
|
+
enforcement: readEnforcementState(obj.enforcement),
|
|
220
|
+
shadowCostMicros: micros(obj.shadow_cost_micros ?? obj.shadowCostMicros),
|
|
221
|
+
quotaSource: readQuotaSourceState(obj.quota_source ?? obj.quotaSource),
|
|
195
222
|
funding: str(obj.funding),
|
|
196
223
|
windows,
|
|
197
224
|
usage: obj.usage && typeof obj.usage === "object" ? { ...obj.usage } : null,
|
package/lib/session/config.mjs
CHANGED
|
@@ -127,6 +127,10 @@ export function sessionPaths(agentRoot) {
|
|
|
127
127
|
upgradeNoticeFile: join(stateDir, "upgrade-notice.json"),
|
|
128
128
|
lastExitFile: join(stateDir, "last-exit"),
|
|
129
129
|
attentionFile: join(stateDir, "attention.json"),
|
|
130
|
+
// Consecutive identical launch failures, carried across supervisor
|
|
131
|
+
// lifetimes (lib/session/launch-failure.mjs). A counter held in memory is
|
|
132
|
+
// zero on every launch, which is why the bound never bound.
|
|
133
|
+
launchFailuresFile: join(stateDir, "launch-failures.json"),
|
|
130
134
|
daemonHealthFile: join(agentRoot, "state", "dashboards", "daemon-health.yaml"),
|
|
131
135
|
daemonPidFile: join(agentRoot, "state", "daemon.pid"),
|
|
132
136
|
configFile: join(agentRoot, CONFIG_REL),
|
package/lib/session/identity.mjs
CHANGED
|
@@ -65,6 +65,7 @@ export function parseMainSession(text) {
|
|
|
65
65
|
};
|
|
66
66
|
if (typeof raw.lastLaunchAt === "string") rec.lastLaunchAt = raw.lastLaunchAt;
|
|
67
67
|
if (typeof raw.rotatedFrom === "string") rec.rotatedFrom = raw.rotatedFrom;
|
|
68
|
+
if (typeof raw.rotatedReason === "string") rec.rotatedReason = raw.rotatedReason;
|
|
68
69
|
return rec;
|
|
69
70
|
}
|
|
70
71
|
|
|
@@ -178,38 +179,101 @@ export function recordLaunch(record, opts = {}) {
|
|
|
178
179
|
*
|
|
179
180
|
* relaunch — the session ended (cleanly, or after living past the window):
|
|
180
181
|
* exit 75 and let launchd bring it back with the same id.
|
|
181
|
-
* rotate —
|
|
182
|
+
* rotate — the resume target is unusable: mint a fresh id first.
|
|
182
183
|
* backoff — the rotation budget for this hour is spent: sleep, then exit 75.
|
|
183
184
|
*
|
|
184
185
|
* An UNKNOWN exit code (the exit file was not written) inside the window is
|
|
185
186
|
* treated as a failure: the cost of a wrong rotation is a fresh transcript,
|
|
186
187
|
* the cost of a wrong relaunch is a seat stuck in a 30-second crash loop.
|
|
187
188
|
*
|
|
188
|
-
*
|
|
189
|
-
*
|
|
189
|
+
* ── PROVEN OUTRANKS INFERRED, AND IS NOT RATIONED ───────────────────────────
|
|
190
|
+
* `resumeTargetMissing` is the runtime's own verdict, read off the screen by
|
|
191
|
+
* `launch-failure#classifyLaunchFailure` ("No conversation found with session
|
|
192
|
+
* ID: …") or off the disk by `resume-target#resumeTargetState`. When it is
|
|
193
|
+
* true, rotating is not a guess that might help — it is the ONLY repair, and
|
|
194
|
+
* it provably works. So it bypasses {@link MAX_ROTATIONS_PER_HOUR}, which
|
|
195
|
+
* exists to bound blind rotation. Leaving it inside the budget is precisely
|
|
196
|
+
* how James Kirkland's seat spent days relaunching a dead id on a ten-minute
|
|
197
|
+
* backoff: the budget throttled the one action that would have fixed it.
|
|
198
|
+
*
|
|
199
|
+
* ── `beatSeen` NARROWS THE GUESS, IT DOES NOT WIDEN IT ──────────────────────
|
|
200
|
+
* The twenty-second window is a proxy for "it never really started". A launch
|
|
201
|
+
* that DID write a heartbeat and then died inside that window did start, so it
|
|
202
|
+
* no longer rotates: its transcript is good and the fault is elsewhere. The
|
|
203
|
+
* converse is deliberately NOT symmetric — "ran an hour, never beat, unknown
|
|
204
|
+
* exit" is not evidence enough to destroy a transcript on, because a broken
|
|
205
|
+
* heartbeat writer looks exactly like it.
|
|
206
|
+
*
|
|
207
|
+
* The one place a never-beaten launch rotates outside the window is when it
|
|
208
|
+
* has ALREADY failed identically to the escalation limit (`streakAtLimit`):
|
|
209
|
+
* every other repair has been tried, the seat is escalating anyway, and a
|
|
210
|
+
* fresh id is the last thing left that could work. It is suppressed when WE
|
|
211
|
+
* ended the session (`selfStopped`: the watchdog restarting a wedged session),
|
|
212
|
+
* since that is exactly the case where the transcript is still good.
|
|
213
|
+
*
|
|
214
|
+
* @param {object} a
|
|
215
|
+
* @param {object} a.record
|
|
216
|
+
* @param {"session-id"|"resume"} a.mode
|
|
217
|
+
* @param {number|null} a.exitCode
|
|
218
|
+
* @param {number} a.startedAt
|
|
219
|
+
* @param {number} a.endedAt
|
|
220
|
+
* @param {boolean} [a.resumeTargetMissing] PROVEN: the conversation is not on this machine
|
|
221
|
+
* @param {boolean} [a.beatSeen] did this launch write a heartbeat? (undefined = could not tell)
|
|
222
|
+
* @param {boolean} [a.streakAtLimit] has this launch failed identically to the escalation limit?
|
|
223
|
+
* @param {boolean} [a.selfStopped] did the supervisor itself end the session?
|
|
224
|
+
* @param {number|Function} [a.now]
|
|
225
|
+
* @param {number} [a.failureWindowMs]
|
|
226
|
+
* @param {number} [a.maxRotationsPerHour]
|
|
227
|
+
* @returns {{action:"relaunch"}|{action:"rotate", proven:boolean, why:string}|{action:"backoff", sleepMs:number}}
|
|
190
228
|
*/
|
|
191
229
|
export function rotationDecision(a) {
|
|
192
230
|
const windowMs = a.failureWindowMs ?? FAILURE_WINDOW_MS;
|
|
193
231
|
const max = a.maxRotationsPerHour ?? MAX_ROTATIONS_PER_HOUR;
|
|
194
232
|
const durationMs = Number(a.endedAt) - Number(a.startedAt);
|
|
195
233
|
const failed = a.exitCode === null || a.exitCode === undefined || a.exitCode !== 0;
|
|
196
|
-
if (a.mode !== "resume"
|
|
234
|
+
if (a.mode !== "resume") return { action: "relaunch" };
|
|
235
|
+
if (a.resumeTargetMissing === true) {
|
|
236
|
+
return { action: "rotate", proven: true, why: "the conversation this launch resumed does not exist on this machine" };
|
|
237
|
+
}
|
|
238
|
+
if (!failed || a.selfStopped === true) return { action: "relaunch" };
|
|
239
|
+
const insideWindow = durationMs < windowMs && a.beatSeen !== true;
|
|
240
|
+
const lastResort = a.beatSeen === false && a.streakAtLimit === true;
|
|
241
|
+
if (!insideWindow && !lastResort) return { action: "relaunch" };
|
|
197
242
|
const t = nowMs(a.now);
|
|
198
243
|
const recent = (a.record && a.record.rotations || []).filter((r) => t - Date.parse(r) < HOUR_MS);
|
|
199
244
|
if (recent.length >= max) return { action: "backoff", sleepMs: BACKOFF_MS };
|
|
200
|
-
return {
|
|
245
|
+
return {
|
|
246
|
+
action: "rotate",
|
|
247
|
+
proven: false,
|
|
248
|
+
why: insideWindow
|
|
249
|
+
? "the resume died inside the failure window"
|
|
250
|
+
: "every recent launch has failed identically and never beaten — a fresh id is the last untried repair",
|
|
251
|
+
};
|
|
201
252
|
}
|
|
202
253
|
|
|
203
254
|
/**
|
|
204
255
|
* Pure: the record after a rotation — fresh id, zero resumes, this rotation
|
|
205
256
|
* stamped, and the rotation history pruned to the last hour.
|
|
206
|
-
*
|
|
257
|
+
*
|
|
258
|
+
* THIS IS BOTH HALVES OF THE REPAIR, AND THAT IS THE POINT. A dead resume id
|
|
259
|
+
* needs the stale record INVALIDATED (so nothing ever resumes it again) and a
|
|
260
|
+
* FRESH id to fall back to (so the next launch has somewhere to go). Doing
|
|
261
|
+
* only the first leaves the supervisor with no id and `newMainSession` would
|
|
262
|
+
* be minted anyway; doing only the second leaves the dead id in the file for
|
|
263
|
+
* the next reader. One write does both: the dead id survives only as
|
|
264
|
+
* `rotatedFrom`, which is provenance, not a target — nothing resumes it.
|
|
265
|
+
*
|
|
266
|
+
* `reason` is recorded so the next reader of the file (a person, `maestro
|
|
267
|
+
* session status`) can tell a proven repair from a heuristic one.
|
|
268
|
+
*
|
|
269
|
+
* @param {object} record @param {{now?:number|Function, uuid?:Function, reason?:string}} [opts]
|
|
207
270
|
*/
|
|
208
271
|
export function rotateMainSession(record, opts = {}) {
|
|
209
272
|
const t = nowMs(opts.now);
|
|
210
273
|
const fresh = newMainSession({ now: t, uuid: opts.uuid });
|
|
211
274
|
const kept = (record.rotations || []).filter((r) => t - Date.parse(r) < HOUR_MS);
|
|
212
|
-
|
|
275
|
+
const reason = typeof opts.reason === "string" && opts.reason.trim() ? opts.reason.trim().slice(0, 300) : "";
|
|
276
|
+
return { ...fresh, rotations: [...kept, iso(t)], rotatedFrom: record.sessionId, ...(reason ? { rotatedReason: reason } : {}) };
|
|
213
277
|
}
|
|
214
278
|
|
|
215
279
|
export default {
|
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lib/session/launch-failure.mjs — WHY a main-session launch failed, and when
|
|
3
|
+
* the seat should stop trying quietly and say so. Pure; no clock, no fs.
|
|
4
|
+
*
|
|
5
|
+
* ── THE FAULT THIS MODULE EXISTS TO NOT HAVE ────────────────────────────────
|
|
6
|
+
* Measured on James Kirkland's seat, 2026-09-25. The supervisor relaunched
|
|
7
|
+
* every ten minutes for DAYS running
|
|
8
|
+
*
|
|
9
|
+
* claude --resume e953a33e-97a0-4d13-9fd4-6a77cd7a68c9
|
|
10
|
+
* No conversation found with session ID: e953a33e-…
|
|
11
|
+
* exit 1, after ~2 s
|
|
12
|
+
*
|
|
13
|
+
* for a conversation that no longer existed on that machine. Every layer
|
|
14
|
+
* behaved "correctly": launchd relaunched a job that exited non-zero, the
|
|
15
|
+
* rotation budget throttled to one attempt per ten minutes, the log recorded
|
|
16
|
+
* each failure. The seat looked UP the whole time — a live launchd job, a
|
|
17
|
+
* supervisor process, a mux session flickering into existence — and never
|
|
18
|
+
* beat once. Nobody reads that log; nobody could ssh in to read it.
|
|
19
|
+
*
|
|
20
|
+
* Two things were missing and both are here:
|
|
21
|
+
*
|
|
22
|
+
* 1. THE FAILURE HAS A NAME AND THE NAME WAS ON THE SCREEN. "No conversation
|
|
23
|
+
* found with session ID" is not a symptom to infer from timing — it is the
|
|
24
|
+
* runtime telling us the resume target is gone. {@link classifyLaunchFailure}
|
|
25
|
+
* reads it, so the repair (mint a fresh id) is taken on evidence rather
|
|
26
|
+
* than on the 20-second heuristic in `identity#rotationDecision`, which
|
|
27
|
+
* misses a failure that takes 21 seconds and cannot distinguish a dead
|
|
28
|
+
* transcript from a missing binary.
|
|
29
|
+
*
|
|
30
|
+
* 2. N IDENTICAL FAILURES ARE NOT N CHANCES — they are one fault, repeated.
|
|
31
|
+
* A launch that fails the same way three times in a row will fail the
|
|
32
|
+
* fourth time too, and the only useful next step is to tell somebody who
|
|
33
|
+
* is not on this machine. {@link launchFailureStreak} counts consecutive
|
|
34
|
+
* failures WITH THE SAME SIGNATURE across supervisor lifetimes (each
|
|
35
|
+
* launch is a fresh process, so an in-memory counter is always zero —
|
|
36
|
+
* the same trap `first-run#carriedRestarts` exists for), and
|
|
37
|
+
* {@link escalationDecision} says when it stops being a retry and becomes
|
|
38
|
+
* an escalation.
|
|
39
|
+
*
|
|
40
|
+
* The signature deliberately does NOT include the session id: rotating to a
|
|
41
|
+
* fresh id and failing the same way again is the SAME fault (that is exactly
|
|
42
|
+
* what a missing binary does), and a signature that changed every rotation
|
|
43
|
+
* would reset the streak forever — which is how this failure stayed invisible.
|
|
44
|
+
*
|
|
45
|
+
* @module lib/session/launch-failure
|
|
46
|
+
*/
|
|
47
|
+
|
|
48
|
+
"use strict";
|
|
49
|
+
|
|
50
|
+
/** Consecutive identical launch failures before the seat escalates. */
|
|
51
|
+
export const IDENTICAL_FAILURE_LIMIT = 3;
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* The runtime's own words for "the conversation you asked me to resume is not
|
|
55
|
+
* on this machine". Matched case-insensitively against the tail of the pane /
|
|
56
|
+
* the launcher's stderr. Kept as a list because the wording is the CLI's, not
|
|
57
|
+
* ours, and a second phrasing must be addable without touching the classifier.
|
|
58
|
+
*/
|
|
59
|
+
export const RESUME_MISSING_PATTERNS = Object.freeze([
|
|
60
|
+
/no conversation found with session id/i,
|
|
61
|
+
/no conversation found matching/i,
|
|
62
|
+
/session .{0,80}? not found/i,
|
|
63
|
+
]);
|
|
64
|
+
|
|
65
|
+
/** "the binary is not there" — a shell's 127, and the two spellings of ENOENT. */
|
|
66
|
+
export const BINARY_MISSING_PATTERNS = Object.freeze([
|
|
67
|
+
/command not found/i,
|
|
68
|
+
/no such file or directory/i,
|
|
69
|
+
/\bENOENT\b/,
|
|
70
|
+
]);
|
|
71
|
+
|
|
72
|
+
const anyMatch = (patterns, text) => patterns.some((re) => re.test(text));
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Pure: name the failure of one launch.
|
|
76
|
+
*
|
|
77
|
+
* `output` is whatever text the launch left behind — a pane capture, the
|
|
78
|
+
* launcher's stderr, or "" when nothing was captured. It is EVIDENCE, and its
|
|
79
|
+
* absence is never treated as proof of anything: with no output the verdict
|
|
80
|
+
* falls back to what the exit code and the mode can carry on their own.
|
|
81
|
+
*
|
|
82
|
+
* @param {object} a
|
|
83
|
+
* @param {"session-id"|"resume"} a.mode how this launch addressed the session
|
|
84
|
+
* @param {number|null} a.exitCode the recorded exit (null = the exit file was never written)
|
|
85
|
+
* @param {string} [a.output] pane tail / stderr, may be ""
|
|
86
|
+
* @param {boolean} [a.beatSeen] did this launch produce a heartbeat?
|
|
87
|
+
* @returns {{kind:"none"|"resume-target-missing"|"binary-missing"|"never-beaten"|"nonzero-exit", proven:boolean, detail:string}}
|
|
88
|
+
*/
|
|
89
|
+
export function classifyLaunchFailure(a = {}) {
|
|
90
|
+
const exitCode = a.exitCode === null || a.exitCode === undefined ? null : Number(a.exitCode);
|
|
91
|
+
const output = typeof a.output === "string" ? a.output : "";
|
|
92
|
+
const beatSeen = a.beatSeen === true;
|
|
93
|
+
const failed = exitCode === null || exitCode !== 0;
|
|
94
|
+
|
|
95
|
+
// A launch that beat and then exited 0 is a session that RAN. Nothing to name.
|
|
96
|
+
if (!failed && beatSeen) return { kind: "none", proven: true, detail: "the session ran and exited cleanly" };
|
|
97
|
+
if (!failed) return { kind: "none", proven: false, detail: "the session exited cleanly without beating" };
|
|
98
|
+
|
|
99
|
+
// PROVEN verdicts first: the runtime said what was wrong, in words.
|
|
100
|
+
if (a.mode === "resume" && anyMatch(RESUME_MISSING_PATTERNS, output)) {
|
|
101
|
+
return {
|
|
102
|
+
kind: "resume-target-missing",
|
|
103
|
+
proven: true,
|
|
104
|
+
detail: "the runtime reported that the conversation this launch tried to resume does not exist on this machine",
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
if (exitCode === 127 || anyMatch(BINARY_MISSING_PATTERNS, output)) {
|
|
108
|
+
return {
|
|
109
|
+
kind: "binary-missing",
|
|
110
|
+
proven: exitCode === 127 || anyMatch(BINARY_MISSING_PATTERNS, output),
|
|
111
|
+
detail: "the launcher could not find the binary it was told to run",
|
|
112
|
+
};
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// INFERRED verdicts. `never-beaten` is the honest name for James's seat when
|
|
116
|
+
// the pane was not captured: the launch ended non-zero having never reported
|
|
117
|
+
// in, so whatever it was, it was never a working session.
|
|
118
|
+
if (!beatSeen) {
|
|
119
|
+
return { kind: "never-beaten", proven: false, detail: `the launch ended (exit ${exitCode === null ? "unknown" : exitCode}) without ever writing a heartbeat` };
|
|
120
|
+
}
|
|
121
|
+
return { kind: "nonzero-exit", proven: false, detail: `the session beat, then ended with exit ${exitCode === null ? "unknown" : exitCode}` };
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Pure: the stable identity of a failure, for counting repeats.
|
|
126
|
+
*
|
|
127
|
+
* Mode and kind only. NOT the session id (see the module docblock) and not the
|
|
128
|
+
* exit code when the kind already names the fault — otherwise a flapping exit
|
|
129
|
+
* status would read as a different fault each time and never accumulate.
|
|
130
|
+
*
|
|
131
|
+
* @param {{mode:string, kind:string, exitCode?:number|null}} a
|
|
132
|
+
* @returns {string} e.g. `resume:resume-target-missing`
|
|
133
|
+
*/
|
|
134
|
+
export function launchFailureSignature(a = {}) {
|
|
135
|
+
const kind = String(a.kind || "unknown");
|
|
136
|
+
const mode = a.mode === "resume" ? "resume" : "session-id";
|
|
137
|
+
if (kind === "none") return "";
|
|
138
|
+
if (kind === "never-beaten" || kind === "nonzero-exit") {
|
|
139
|
+
const exit = a.exitCode === null || a.exitCode === undefined ? "unknown" : String(Number(a.exitCode));
|
|
140
|
+
return `${mode}:${kind}:${exit}`;
|
|
141
|
+
}
|
|
142
|
+
return `${mode}:${kind}`;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Pure: parse `state/session/launch-failures.json`. Anything unreadable is
|
|
147
|
+
* "no streak" — a corrupt counter must never be able to escalate on its own.
|
|
148
|
+
* @param {string|null|undefined} text
|
|
149
|
+
* @returns {{signature:string, streak:number, firstAt:string, lastAt:string, escalatedAt:string}|null}
|
|
150
|
+
*/
|
|
151
|
+
export function parseLaunchFailures(text) {
|
|
152
|
+
if (typeof text !== "string" || !text.trim()) return null;
|
|
153
|
+
let raw;
|
|
154
|
+
try { raw = JSON.parse(text); } catch { return null; }
|
|
155
|
+
if (!raw || typeof raw !== "object" || Array.isArray(raw)) return null;
|
|
156
|
+
const signature = typeof raw.signature === "string" ? raw.signature : "";
|
|
157
|
+
const streak = Number.isInteger(raw.streak) && raw.streak > 0 ? raw.streak : 0;
|
|
158
|
+
if (!signature || !streak) return null;
|
|
159
|
+
return {
|
|
160
|
+
signature,
|
|
161
|
+
streak,
|
|
162
|
+
firstAt: typeof raw.firstAt === "string" ? raw.firstAt : "",
|
|
163
|
+
lastAt: typeof raw.lastAt === "string" ? raw.lastAt : "",
|
|
164
|
+
escalatedAt: typeof raw.escalatedAt === "string" ? raw.escalatedAt : "",
|
|
165
|
+
};
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* Pure: the streak after one more launch outcome.
|
|
170
|
+
*
|
|
171
|
+
* A DIFFERENT signature restarts the count at 1 rather than adding to it: two
|
|
172
|
+
* different faults in a row are two problems, and escalating on their sum
|
|
173
|
+
* would cry wolf. A successful launch (`signature` empty) clears the record
|
|
174
|
+
* entirely — evidence of work, the same reset rule `carriedRestarts` uses.
|
|
175
|
+
*
|
|
176
|
+
* @param {object|null} prior {@link parseLaunchFailures} output
|
|
177
|
+
* @param {{signature:string, at:string}} outcome
|
|
178
|
+
* @returns {{signature:string, streak:number, firstAt:string, lastAt:string, escalatedAt:string}|null} null = clear the file
|
|
179
|
+
*/
|
|
180
|
+
export function launchFailureStreak(prior, outcome = {}) {
|
|
181
|
+
const signature = typeof outcome.signature === "string" ? outcome.signature : "";
|
|
182
|
+
const at = typeof outcome.at === "string" ? outcome.at : "";
|
|
183
|
+
if (!signature) return null; // the launch worked: nothing to carry
|
|
184
|
+
if (prior && prior.signature === signature) {
|
|
185
|
+
return { signature, streak: prior.streak + 1, firstAt: prior.firstAt || at, lastAt: at, escalatedAt: prior.escalatedAt || "" };
|
|
186
|
+
}
|
|
187
|
+
return { signature, streak: 1, firstAt: at, lastAt: at, escalatedAt: "" };
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/**
|
|
191
|
+
* Pure: has this stopped being a retry?
|
|
192
|
+
*
|
|
193
|
+
* `escalate` is true on the launch that REACHES the limit and on every launch
|
|
194
|
+
* after it, because the escalation channel is a beat field that the next
|
|
195
|
+
* supervisor run clears (`attention.json` is unlinked at launch): a one-shot
|
|
196
|
+
* escalation would flicker off and the org would see a seat that healed. The
|
|
197
|
+
* `fresh` flag distinguishes the first crossing, which is the one worth a log
|
|
198
|
+
* line and a distinct wording.
|
|
199
|
+
*
|
|
200
|
+
* @param {{streak:number, limit?:number, escalatedAt?:string}} a
|
|
201
|
+
* @returns {{escalate:boolean, fresh:boolean, streak:number, limit:number, reason:string}}
|
|
202
|
+
*/
|
|
203
|
+
export function escalationDecision(a = {}) {
|
|
204
|
+
const limit = Number.isInteger(a.limit) && a.limit > 0 ? a.limit : IDENTICAL_FAILURE_LIMIT;
|
|
205
|
+
const streak = Number.isInteger(a.streak) && a.streak > 0 ? a.streak : 0;
|
|
206
|
+
if (streak < limit) {
|
|
207
|
+
return { escalate: false, fresh: false, streak, limit, reason: `launch failure ${streak}/${limit} — retrying` };
|
|
208
|
+
}
|
|
209
|
+
const fresh = !a.escalatedAt;
|
|
210
|
+
return {
|
|
211
|
+
escalate: true,
|
|
212
|
+
fresh,
|
|
213
|
+
streak,
|
|
214
|
+
limit,
|
|
215
|
+
reason: `${streak} consecutive launches have failed the same way (limit ${limit}) — this is a fault, not a retry`,
|
|
216
|
+
};
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* Pure: the one sentence an operator in a browser gets, via the beat's
|
|
221
|
+
* `machine.sessionNote.detail`.
|
|
222
|
+
*
|
|
223
|
+
* THE VERDICT LEADS, THE EXPLANATION TRAILS — because the channel TRUNCATES.
|
|
224
|
+
* `telemetry/collect#sanitizeNoteDetail` caps the detail at 200 characters and
|
|
225
|
+
* appends an ellipsis, so anything after the first sentence may never leave the
|
|
226
|
+
* machine. The first draft of this line ended "…this needs a person" and that
|
|
227
|
+
* clause — the only part that asks for an action — was the part that got cut.
|
|
228
|
+
* So: how many, that retries are not working, then the diagnosis.
|
|
229
|
+
*
|
|
230
|
+
* @param {{signature:string, streak:number, limit?:number, detail:string, sessionId?:string}} a
|
|
231
|
+
* @returns {string}
|
|
232
|
+
*/
|
|
233
|
+
export function escalationHint(a = {}) {
|
|
234
|
+
const sig = String(a.signature || "unknown");
|
|
235
|
+
const streak = Number(a.streak) || 0;
|
|
236
|
+
const detail = String(a.detail || "").trim();
|
|
237
|
+
const id = typeof a.sessionId === "string" && a.sessionId ? ` Session ${a.sessionId}.` : "";
|
|
238
|
+
return `${streak}× in a row, identically — retries are not fixing it and this needs a person. ${detail || sig}.${id}`;
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
export default {
|
|
242
|
+
IDENTICAL_FAILURE_LIMIT,
|
|
243
|
+
RESUME_MISSING_PATTERNS,
|
|
244
|
+
BINARY_MISSING_PATTERNS,
|
|
245
|
+
classifyLaunchFailure,
|
|
246
|
+
launchFailureSignature,
|
|
247
|
+
parseLaunchFailures,
|
|
248
|
+
launchFailureStreak,
|
|
249
|
+
escalationDecision,
|
|
250
|
+
escalationHint,
|
|
251
|
+
};
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lib/session/resume-target.mjs — is the conversation we are about to `--resume`
|
|
3
|
+
* actually on this machine?
|
|
4
|
+
*
|
|
5
|
+
* ── WHY THIS IS A PREFLIGHT AND NOT A POST-MORTEM ───────────────────────────
|
|
6
|
+
* The supervisor already had a post-mortem: a resume that died non-zero inside
|
|
7
|
+
* twenty seconds rotates to a fresh id (`identity#rotationDecision`). It is a
|
|
8
|
+
* good heuristic and it is not enough — it cannot tell a dead transcript from a
|
|
9
|
+
* missing binary, it misses a failure that takes twenty-one seconds, and every
|
|
10
|
+
* one of its verdicts costs a full launch, a mux session and a launchd cycle to
|
|
11
|
+
* reach. On James Kirkland's seat that cycle ran every ten minutes for days
|
|
12
|
+
* against a session id that had not existed for just as long.
|
|
13
|
+
*
|
|
14
|
+
* The answer is on disk before we launch. Claude Code keeps each project's
|
|
15
|
+
* transcripts as `~/.claude/projects/<project-slug>/<session-id>.jsonl`, where
|
|
16
|
+
* the slug is the project directory with every non-alphanumeric character
|
|
17
|
+
* replaced by `-`. `--resume` is project-scoped, so the seat's own slug is the
|
|
18
|
+
* only place the id could be.
|
|
19
|
+
*
|
|
20
|
+
* ── FAIL-OPEN, DELIBERATELY ─────────────────────────────────────────────────
|
|
21
|
+
* A wrong "missing" throws away a live transcript; a wrong "unknown" costs one
|
|
22
|
+
* launch and lands in the post-mortem that was already there. So `missing` is
|
|
23
|
+
* returned ONLY from positive evidence: the project directory exists, it holds
|
|
24
|
+
* transcripts, and this id is not among them. An absent directory, an empty
|
|
25
|
+
* one, an unreadable one, an engine that does not use this layout at all — all
|
|
26
|
+
* `unknown`, and the launch proceeds exactly as before.
|
|
27
|
+
*
|
|
28
|
+
* One I/O shell (`readdirSync`/`existsSync`), both injected.
|
|
29
|
+
*
|
|
30
|
+
* @module lib/session/resume-target
|
|
31
|
+
*/
|
|
32
|
+
|
|
33
|
+
"use strict";
|
|
34
|
+
|
|
35
|
+
import { readdirSync as fsReaddirSync } from "node:fs";
|
|
36
|
+
import { join } from "node:path";
|
|
37
|
+
|
|
38
|
+
/** Where Claude Code keeps per-project transcripts, relative to the home dir. */
|
|
39
|
+
export const TRANSCRIPT_ROOT = join(".claude", "projects");
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Pure: Claude Code's directory name for a project path — every character that
|
|
43
|
+
* is not a letter or a digit becomes `-` (`/Users/x/hq` → `-Users-x-hq`).
|
|
44
|
+
* @param {string} projectDir
|
|
45
|
+
* @returns {string}
|
|
46
|
+
*/
|
|
47
|
+
export function projectSlug(projectDir) {
|
|
48
|
+
return String(projectDir || "").replace(/[^A-Za-z0-9]/g, "-");
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Is `sessionId`'s transcript present for `projectDir`?
|
|
53
|
+
*
|
|
54
|
+
* @param {object} a
|
|
55
|
+
* @param {string} a.homeDir the user's home directory
|
|
56
|
+
* @param {string} a.projectDir the directory the session runs in (the agent root)
|
|
57
|
+
* @param {string} a.sessionId the id this launch would resume
|
|
58
|
+
* @param {"claude"|"cohort"} [a.engine] a non-claude engine does not use this layout → always "unknown"
|
|
59
|
+
* @param {{readdirSync?:Function}} [deps]
|
|
60
|
+
* @returns {{state:"present"|"missing"|"unknown", dir:string, reason:string}}
|
|
61
|
+
*/
|
|
62
|
+
export function resumeTargetState(a = {}, deps = {}) {
|
|
63
|
+
const readdirSync = deps.readdirSync || fsReaddirSync;
|
|
64
|
+
const homeDir = typeof a.homeDir === "string" ? a.homeDir : "";
|
|
65
|
+
const sessionId = String(a.sessionId || "").toLowerCase();
|
|
66
|
+
if (a.engine && a.engine !== "claude") return { state: "unknown", dir: "", reason: `engine ${a.engine} does not keep transcripts here` };
|
|
67
|
+
if (!homeDir || !sessionId) return { state: "unknown", dir: "", reason: "no home directory or session id to check" };
|
|
68
|
+
const dir = join(homeDir, TRANSCRIPT_ROOT, projectSlug(a.projectDir));
|
|
69
|
+
let entries;
|
|
70
|
+
try { entries = readdirSync(dir); } catch { return { state: "unknown", dir, reason: "the project's transcript directory is absent or unreadable" }; }
|
|
71
|
+
const names = (Array.isArray(entries) ? entries : []).map((e) => String(e && e.name ? e.name : e));
|
|
72
|
+
const transcripts = names.filter((n) => n.toLowerCase().endsWith(".jsonl"));
|
|
73
|
+
if (transcripts.some((n) => n.toLowerCase() === `${sessionId}.jsonl`)) {
|
|
74
|
+
return { state: "present", dir, reason: "the transcript is on disk" };
|
|
75
|
+
}
|
|
76
|
+
// An EMPTY directory proves nothing: it is equally what a seat looks like the
|
|
77
|
+
// moment before its first session writes, and rotating on it would be a guess.
|
|
78
|
+
if (transcripts.length === 0) return { state: "unknown", dir, reason: "the project's transcript directory holds no transcripts at all" };
|
|
79
|
+
return {
|
|
80
|
+
state: "missing",
|
|
81
|
+
dir,
|
|
82
|
+
reason: `${transcripts.length} transcript(s) on disk for this project and none of them is ${sessionId}`,
|
|
83
|
+
};
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
export default { TRANSCRIPT_ROOT, projectSlug, resumeTargetState };
|