pi-goal-list-loop-audit 0.34.14 → 0.34.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -235,7 +235,27 @@ No external watchdog plugin needed. It also recovers **stranded audits**
235
235
  (v0.29.1): a goal stuck in `auditing` with no auditor session alive re-runs
236
236
  the stored claim after 90s instead of black-holing. Storm protection: the
237
237
  send→pause→notify path rearms once per cycle and loud-stops after a 6-error
238
- brake streak, so a broken provider can't spin forever.
238
+ brake streak, so a broken provider can't spin forever. A confirmed queued
239
+ continuation with no turn is reported by the queue-stuck probe; it does not
240
+ inject terminal input.
241
+
242
+ ## Session replacement and stale handles
243
+
244
+ Recovery crosses pi's lifecycle boundary. On `session_shutdown`, glla records
245
+ `.pi-glla/session-handoff.json` for a supervising, non-quit replacement,
246
+ ledgering the reason and stopping every session-owned timer. On the fresh
247
+ `session_start`, the new context consumes only fresh, same-process debt and
248
+ continues the saved goal or loop. Timers resolve a fresh context when they fire;
249
+ they do not retain an old `ctx` or `pi` handle.
250
+
251
+ A user quit is not continuation consent: `reason: "quit"` is ledgered as
252
+ `session_handoff_suppressed`, leaves no handoff debt, and does not receive
253
+ same-pid rebind consent. An orphan with no fresh lifecycle event is reported
254
+ honestly because an invalidated extension cannot repair its own pi host; if the
255
+ warning says no replacement arrived, restart pi normally and let the saved
256
+ `.pi-glla/` state restore. There are no automatic `/reload` keystrokes and no
257
+ mux dependency. The legacy `autoReloadOnStale` and `autoRecovery` fields remain
258
+ only as deprecated settings-file compatibility fields; they are ignored.
239
259
 
240
260
  **User aborts mean STOP** (v0.29.4): Esc-aborting a turn stands the chain
241
261
  down with a named notify (`/goal resume` to continue) — it does NOT count
package/docs/DESIGN.md CHANGED
@@ -90,6 +90,26 @@ architectural decisions that changed the SHAPE of the system:
90
90
  Confirm path; it never audits work — the isolated auditor is the only
91
91
  verifier.
92
92
 
93
+ ## Addendum v0.34.16 (lifecycle handoff)
94
+
95
+ - **Recovery crosses pi's lifecycle, never the terminal**: `session_shutdown`
96
+ persists fresh same-process continuation debt in
97
+ `.pi-glla/session-handoff.json`, records the shutdown reason, and clears all
98
+ session-owned timers. A fresh `session_start` consumes matching debt and
99
+ continues from its new context. No stale callback is allowed to use the old
100
+ `ctx` or `pi` reference.
101
+ - **Quit is not implicit resume consent**: a shutdown with `reason: "quit"`
102
+ removes any handoff debt, records `session_handoff_suppressed`, and marks
103
+ the owner sidecar so the next same-pid startup is not mistaken for a
104
+ replacement rebind. The global `autoResume` policy remains independent and
105
+ explicit.
106
+ - **True orphans stay honest**: once pi invalidates an extension without a
107
+ fresh lifecycle event, the extension cannot repair its host. glla stops
108
+ stale work, preserves the artifact, and tells the user to restart pi only
109
+ when no replacement arrives. `autoReloadOnStale` and `autoRecovery` remain
110
+ deprecated deserialization compatibility fields; they do not select a
111
+ transport.
112
+
93
113
  ## Addendum v0.4.0 (completion)
94
114
 
95
115
  - **Auditor compaction enabled** (flaw #3 — the last open one). Safety:
@@ -176,6 +176,8 @@ export interface Goal {
176
176
  * At 3 the goal pauses loudly — a broken auditor model must not spin a
177
177
  * silent retry-forever loop. Cleared on any real auditor run. */
178
178
  auditInfraStreak?: number;
179
+ /** v0.34.15: persisted error-brake rung — survives /reload so the 6-brake park can engage. */
180
+ errorBrakeStreak?: number;
179
181
  /** v0.28.26: the completion claim captured when an audit attempt is
180
182
  * quota-blocked. The quota retry re-runs the AUDITOR directly with this
181
183
  * stored claim instead of re-engaging the agent — re-engaging produced a
@@ -215,10 +215,11 @@ export function buildStatusText(state: State, audit?: AuditDisplayProgress | nul
215
215
  }
216
216
  if (g.status === "active") {
217
217
  // v0.28.1 (S1/S2): a stale-handle interrupt keeps the goal ACTIVE.
218
- // v0.29.11: the fresh session HOLDS it (hold-everything restore gate;
219
- // autoresume=on resumes for you) — name the verb, don't promise auto.
218
+ // v0.34.16: a fresh session_start owns the handoff. A cold boot still
219
+ // follows the global autoResume setting, so the widget names the actual
220
+ // lifecycle rather than promising terminal keystroke recovery.
220
221
  if (g.interruptedAt) {
221
- return `glla: ${paint(theme, "error", "⚠ interrupted — stale handle · /reload → /glla resume")}${heldSuffix}`;
222
+ return `glla: ${paint(theme, "error", "⚠ interrupted — stale handle · fresh session_start resumes")}${heldSuffix}`;
222
223
  }
223
224
  // v0.24.7: list policy gets its own wording — a queue item is not a goal.
224
225
  // v0.28.11 (U10): goal policy joins it — "list 29" read as a command
@@ -68,18 +68,12 @@ export interface Settings {
68
68
  /** Consecutive stuck interventions before a loop stops (default 5,
69
69
  * 10 under aggressiveMode). */
70
70
  stuckMaxInterventions?: number;
71
- /** v0.29.13: on a stale-handle terminal, inject /reload into our own
72
- * tmux pane (keystroke self-heal; pi walls ctx.reload() behind
73
- * assertActive). Default true; only acts when TMUX and TMUX_PANE are
74
- * set — pi outside tmux just gets the manual warning. */
71
+ /** @deprecated v0.34.16: retained so older settings files deserialize, but
72
+ * ignored. Recovery now uses session_shutdown/session_start handoff and
73
+ * never injects terminal keystrokes. */
75
74
  autoReloadOnStale?: boolean;
76
- /** v0.34.13: auto-recovery ladder — when a wedge is detected that only a
77
- * /reload cures (unanswered continuation, send-retry storm), inject the
78
- * /reload ITSELF via the v0.29.13 tmux/WezTerm transport and resume the
79
- * goal/loop after the rebuild (sidecar marker — autoresume=off is a
80
- * restore-time setting, not a recovery veto). Default true. The one
81
- * class it does NOT cross: pi restart (transcript-writer dead) — that
82
- * stays a loud stop for the human. */
75
+ /** @deprecated v0.34.16: retained for settings-file compatibility, but
76
+ * ignored. Lifecycle handoff is always enabled. */
83
77
  autoRecovery?: boolean;
84
78
  /** v0.26.1: consecutive heartbeat refires without a real turn before
85
79
  * the goal pauses / loop stops (default 5; 0 = never escalate). */
@@ -214,8 +208,6 @@ export const SETTINGS_KEYS: Array<keyof Settings> = [
214
208
  "quotaRetryMinutes",
215
209
  "stuckMaxInterventions",
216
210
  "stallEscalationRefires",
217
- "autoReloadOnStale",
218
- "autoRecovery",
219
211
  "stallShortWords",
220
212
  "stallSimilarityThreshold",
221
213
  "postaudit",
@@ -15,7 +15,6 @@
15
15
  */
16
16
 
17
17
  import * as fs from "node:fs";
18
- import { exec } from "node:child_process";
19
18
  import * as os from "node:os";
20
19
  import * as path from "node:path";
21
20
 
@@ -221,18 +220,16 @@ const GOAL_EVENT_ENTRY = "goal-event";
221
220
  // not on ExtensionContext, so continuation sends need it at module scope.
222
221
  let extensionApi: ExtensionAPI | null = null;
223
222
  // v0.26.7: pi invalidates the extension runtime on session replacement
224
- // (newSession/fork/switchSession/reload — and the compaction path reaches
225
- // it via teardownCurrent in pi 0.82.x). Once stale, every sendMessage
223
+ // (newSession/fork/switchSession/reload). Once stale, every sendMessage
226
224
  // throws FOREVER in this process — retrying for hours is the hegemon
227
225
  // failure shape. Detect the stale signature once and go terminally loud.
228
226
  let extensionApiStale = false;
229
227
  // v0.32.0: CRITICAL — goStaleTerminal must gate on its OWN flag, not
230
228
  // extensionApiStale: probeExtensionApiStale() sets extensionApiStale on
231
229
  // detection, so the heartbeat's `probe → goStaleTerminal` sequence always
232
- // found the flag already true and returned silently — orphan-stale recovery
233
- // (ledger, loop stop, interruptedAt, warn, AUTO-RELOAD SELF-HEAL) was dead
234
- // code since v0.29.11. Field proof: hegemon sat stale for days and the
235
- // wezterm self-heal never fired.
230
+ // found the flag already true and returned silently. The terminal orphan
231
+ // path must still ledger the stale handle, stop stale work, and preserve the
232
+ // interrupt marker so a later fresh lifecycle can restore it.
236
233
  let staleTerminalDone = false;
237
234
 
238
235
  /** v0.26.7: a stale api is terminal for this process — go loudly with
@@ -241,25 +238,14 @@ let staleTerminalDone = false;
241
238
  * pausing — the restore gate only auto-resumes ACTIVE goals, so pausing
242
239
  * here stranded goals until manual /goal resume (hegemon/sraaal shape).
243
240
  * sendContinuation's extensionApiStale guard already stops further sends
244
- * in this doomed process; the next fresh session auto-resumes. */
245
- /** v0.30.0: rebind-first session-replacement survival. pi's sanctioned
246
- * pattern (docs/extensions.md lifecycle + the stale error text itself):
247
- * session_shutdown → cleanup, session_start → re-establish with the NEW
248
- * ctx. glla used to treat every stale handle as terminal ("run /reload"),
249
- * but three replacement shapes need three responses:
250
- * (a) switch (resume/new/fork): pi rebinds THIS module to the new
251
- * session — session_start delivers a fresh ctx. No user action, no
252
- * warning; reset the stale flag via a re-probe and continue.
253
- * (b) /reload: pi re-imports the extension modules — a SUCCESSOR
254
- * instance owns this cwd in the same process. The old module stands
255
- * down silently (owner-file check) instead of screaming + injecting
256
- * /reload (v0.29.22's injection is right for orphans, wrong here).
257
- * (c) orphan: the session died with NO replacement (hegemon 2026-07-31:
258
- * handle dead ~06:03, zero ledger events for 5h). Only a rebuild
259
- * revives extension function — goStaleTerminal's warning + self-heal
260
- * stays for this case ONLY.
261
- * session_shutdown is now ledgered with pi's reason, so the next
262
- * unexplained disposal is attributable from the ledger alone. */
241
+ * in this doomed process; the next fresh session can restore the work. */
242
+ /** v0.34.16: lifecycle-first session-replacement survival. pi's
243
+ * sanctioned pattern (docs/extensions.md lifecycle + the stale error text):
244
+ * session_shutdown → persist handoff debt + stop old timers,
245
+ * session_start → re-establish with the NEW ctx and consume the debt.
246
+ * A successor module may still stand down via the owner file. An orphan with
247
+ * no replacement is reported honestly: an invalid extension cannot repair its
248
+ * own pi host, so glla never injects terminal keystrokes. */
263
249
  const SESSION_REBIND_GRACE_MS = 60_000;
264
250
  let sessionReplacementUntil = 0;
265
251
  const instanceStartedAt = Date.now();
@@ -301,12 +287,7 @@ function absorbStaleIfSuperseded(ctx: ExtensionContext): boolean {
301
287
  appendLedger(ctx.cwd, "zombie_stood_down", { owner: owner.instanceId });
302
288
  zombieStoodDown = true;
303
289
  extensionApiStale = true; // silence the send paths WITHOUT the terminal theatre
304
- clearLoopTimer();
305
- if (continuationTimer) { clearTimeout(continuationTimer); continuationTimer = null; }
306
- // v0.32.0: the superseded module's heartbeat + UI ticker were IMMORTAL
307
- // (clearInterval appeared nowhere) — N /reloads = N×2 zombie tickers.
308
- if (heartbeatTimer) { clearInterval(heartbeatTimer); heartbeatTimer = null; }
309
- if (uiTicker) { clearInterval(uiTicker); uiTicker = null; }
290
+ clearSessionOwnedTimers();
310
291
  return true;
311
292
  }
312
293
  return false;
@@ -317,10 +298,10 @@ function goStaleTerminal(ctx: ExtensionContext, where: string): void {
317
298
  staleTerminalDone = true;
318
299
  extensionApiStale = true;
319
300
  appendLedger(ctx.cwd, "extension_api_stale", { where, kind: isLoopActive() ? "loop" : "goal" });
320
- const guidance = "pi invalidated this session's extension handle (session replacement — the session was disposed and this process's sends can never land). Run /reload — extensions rebuild IN PLACE, no pi restart needed — then /glla resume (autoresume=on resumes for you). Restart pi only if /reload itself fails.";
301
+ const guidance = "pi invalidated this session's extension handle without delivering a replacement session. glla stopped stale sends and kept the work safe in .pi-glla/. A fresh session_start will resume it; if pi does not create one, restart pi normally and glla will restore the saved work.";
321
302
  // v0.32.0: kill the continuation re-arm too — otherwise an orphaned goal
322
303
  // keeps spinning a flat 50ms retry below every watchdog.
323
- if (continuationTimer) { clearTimeout(continuationTimer); continuationTimer = null; continuationScheduledFor = null; }
304
+ clearSessionOwnedTimers();
324
305
  if (isLoopActive()) {
325
306
  clearLoopTimer();
326
307
  state.loop = { ...state.loop!, active: false, stopReason: `extension api stale: ${guidance}` };
@@ -329,107 +310,83 @@ function goStaleTerminal(ctx: ExtensionContext, where: string): void {
329
310
  updateGoal({ interruptedAt: nowIso(), interruptedReason: `extension api stale (${where})` }, ctx);
330
311
  }
331
312
  ctx.ui.notify(`glla: ${guidance}`, "warning");
332
- notifyExternal(ctx, `glla: extension api stale — run /reload, then /glla resume. (${where})`);
333
- attemptAutoReload(ctx, where);
313
+ notifyExternal(ctx, `glla: extension api stale — waiting for a fresh session_start; restart pi normally only if no replacement arrives. (${where})`);
334
314
  }
335
315
 
336
- /** v0.29.13: automatic recovery — the zombie handle can't call ctx.reload()
337
- * (pi walls EVERY runtime method behind assertActive), but fs/child_process
338
- * are extension-side and still work. Inject /reload as keystrokes into our
339
- * own terminal pane: pi rebuilds the extension runtime in place, the fresh
340
- * instance loads .pi-glla state and holds (autoresume=on resumes for you).
341
- * v0.29.22: transport-generalized — tmux OR WezTerm. Field: this rig runs
342
- * WezTerm (TERM_PROGRAM=WezTerm, WEZTERM_PANE set, no TMUX), so the
343
- * tmux-only gate failed silently 100% of the time — auto_reload_injected
344
- * never fired fleet-wide, and every stale handle fell back to the manual
345
- * warning (user: "stopping and told to reload is common"). v0.29.22 also
346
- * fires from the entry-probe path (the most common stale discovery).
347
- * Opt out: autoReloadOnStale=false. Best-effort: the manual warning
348
- * already fired, so failures cost nothing. */
349
- function attemptAutoReload(ctx: ExtensionContext, where: string): boolean {
316
+ /** v0.34.16: lifecycle handoff replaces terminal keystroke injection. A
317
+ * stale extension cannot call pi, so recovery must cross the lifecycle
318
+ * boundary: session_shutdown records durable resume debt, clears every timer
319
+ * that could retain the old context, and session_start consumes the debt from
320
+ * a fresh context. */
321
+ const SESSION_HANDOFF_FILE = "session-handoff.json";
322
+ const SESSION_HANDOFF_FRESH_MS = 300_000;
323
+ function sessionHandoffPath(cwd: string): string {
324
+ return path.join(piGlaDir(cwd), SESSION_HANDOFF_FILE);
325
+ }
326
+ function writeSessionHandoff(ctx: ExtensionContext, reason: string): boolean {
327
+ if (!isSupervising()) return false;
328
+ // A user quit is an explicit stop, not a replacement boundary. Do not
329
+ // leave debt that could silently resume the work on a later startup;
330
+ // global autoResume may still apply by its own explicit policy.
331
+ if (reason.trim().toLowerCase() === "quit") {
332
+ try { fs.rmSync(sessionHandoffPath(ctx.cwd), { force: true }); } catch { /* advisory cleanup */ }
333
+ appendLedger(ctx.cwd, "session_handoff_suppressed", { reason });
334
+ return false;
335
+ }
350
336
  try {
351
- if (loadSettings(ctx.cwd).autoReloadOnStale === false) return false;
352
- const tmuxPane = process.env.TMUX_PANE;
353
- const wezPane = process.env.WEZTERM_PANE;
354
- let transport: "tmux" | "wezterm";
355
- let cmd: string;
356
- let pane: string;
357
- if (process.env.TMUX && tmuxPane && /^%\d+$/.test(tmuxPane)) {
358
- transport = "tmux";
359
- pane = tmuxPane;
360
- // v0.29.22: NO leading Escape — a late zombie probe can fire AFTER a
361
- // fresh instance already resumed and started a turn, and Escape
362
- // would abort that turn without consent (the consent line). /reload
363
- // alone lands at a dead prompt and queues harmlessly mid-turn (pi
364
- // refuses the reload itself if a response is streaming).
365
- cmd = `tmux send-keys -t ${pane} -l '/reload' && tmux send-keys -t ${pane} Enter`;
366
- } else if (wezPane && /^\d+$/.test(wezPane)) {
367
- transport = "wezterm";
368
- pane = wezPane;
369
- // wezterm cli send-text --no-paste delivers bytes as keystrokes.
370
- // /reload + CR only — no Escape (see above). Literal CR byte inside
371
- // single quotes — no bash-isms (exec uses /bin/sh).
372
- cmd = `wezterm cli send-text --pane-id ${pane} --no-paste '/reload\r'`;
373
- } else {
374
- appendLedger(ctx.cwd, "auto_reload_skipped", { where, reason: "no supported multiplexer (tmux/WezTerm) in env" });
375
- return false;
376
- }
377
- appendLedger(ctx.cwd, "auto_reload_injected", { where, transport, pane });
378
- ctx.ui.notify("glla: injecting /reload into this pane — extensions rebuild in place and the fresh instance holds state. /glla resume after (automatic with autoresume=on).", "info");
379
- exec(cmd, () => { /* best-effort */ });
337
+ fs.mkdirSync(piGlaDir(ctx.cwd), { recursive: true });
338
+ fs.writeFileSync(sessionHandoffPath(ctx.cwd), JSON.stringify({ pid: process.pid, at: new Date().toISOString(), reason }));
339
+ appendLedger(ctx.cwd, "session_handoff_pending", { reason, pid: process.pid });
380
340
  return true;
381
341
  } catch {
382
- /* never let recovery take the warning path down */
342
+ appendLedger(ctx.cwd, "session_handoff_write_failed", { reason });
383
343
  return false;
384
344
  }
385
345
  }
386
-
387
- /** v0.34.13: one rung of the auto-recovery ladder. Returns true when a
388
- * /reload was injected (caller should NOT also pause/alert the manual
389
- * cure). Writes the sidecar resume marker FIRST so the fresh instance
390
- * resumes the goal/loop even with autoresume=off. Throttled — a second
391
- * wedge inside the window is the pi-restart class and must reach the
392
- * user as a loud stop, not another reload. */
393
- let lastAutoRecoveryAt = 0;
394
- function attemptAutoRecovery(ctx: ExtensionContext, where: string): boolean {
395
- if (loadSettings(ctx.cwd).autoRecovery === false) return false;
396
- const now = Date.now();
397
- if (now - lastAutoRecoveryAt < AUTO_RECOVERY_THROTTLE_MS) return false;
398
- // Inject FIRST — the throttle stamp + marker must reflect a real
399
- // recovery, not a no-transport skip (else the watchdog would mislabel
400
- // the next wedge as the pi-restart class).
401
- if (!attemptAutoReload(ctx, where)) return false;
402
- lastAutoRecoveryAt = now;
346
+ function consumeSessionHandoff(cwd: string): boolean {
403
347
  try {
404
- fs.writeFileSync(
405
- path.join(piGlaDir(ctx.cwd), RECOVERY_RESUME_MARKER),
406
- JSON.stringify({ at: new Date(now).toISOString(), where }),
407
- );
408
- } catch { /* best-effort — the reload still helps; restore just holds */ }
409
- appendLedger(ctx.cwd, "auto_recovery_reload", { where });
410
- ctx.ui.notify(`glla auto-recovery: ${where} — the ${isLoopActive() ? "loop" : "goal/list item"} resumes ITSELF after the rebuild (no /glla resume needed; the sidecar marker carries the consent). If this recurs within ${Math.round(AUTO_RECOVERY_THROTTLE_MS / 60_000)}m it's the pi-restart class and you'll get a loud stop.`, "warning");
411
- return true;
348
+ const p = sessionHandoffPath(cwd);
349
+ if (!fs.existsSync(p)) return false;
350
+ const raw = fs.readFileSync(p, "utf-8");
351
+ fs.unlinkSync(p);
352
+ const data = JSON.parse(raw) as { pid?: number; at?: string; reason?: string };
353
+ const at = Date.parse(data.at ?? "");
354
+ return data.pid === process.pid && data.reason?.trim().toLowerCase() !== "quit" && !Number.isNaN(at) && Date.now() - at < SESSION_HANDOFF_FRESH_MS;
355
+ } catch {
356
+ return false;
357
+ }
412
358
  }
413
359
 
414
360
  /** v0.34.14: /reload rebind detector. The extension runs INSIDE pi, so
415
361
  * process.pid IS pi's pid: an instance that boots and finds its OWN pid
416
- * already in the owner file is a /reload rebuild of a live session, not a
417
- * cold boot. Rebinds always resume active goals/loops — holding mid-work
362
+ * already in the owner file is normally a same-process rebuild, not a cold
363
+ * boot. A non-quit rebind resumes active goals/loops — holding mid-work
418
364
  * after an in-place rebuild is pure friction (user directive: keep going
419
365
  * unless we must stop; "the list is not continuing" after /reload,
420
- * hellhunter 2026-08-01). Cold boots (new pid) still honor autoresume=off.
421
- * Sidecar, not the ledger: read-before-write must be atomic-ish and the
422
- * ledger is append-only. */
366
+ * hellhunter 2026-08-01). An explicit quit is stamped in the sidecar and
367
+ * does not receive this implicit consent; cold boots (new pid) still honor
368
+ * autoresume=off. Sidecar, not the ledger: read-before-write must be
369
+ * atomic-ish and the ledger is append-only. */
423
370
  const SESSION_OWNER_FILE = "session-owner.json";
371
+ function markSessionOwnerShutdown(cwd: string, reason: string): void {
372
+ try {
373
+ const p = path.join(piGlaDir(cwd), SESSION_OWNER_FILE);
374
+ const owner = JSON.parse(fs.readFileSync(p, "utf-8")) as { pid?: number; at?: string };
375
+ if (owner.pid === process.pid) {
376
+ fs.writeFileSync(p, JSON.stringify({ ...owner, shutdownReason: reason, shutdownAt: new Date().toISOString() }));
377
+ }
378
+ } catch { /* advisory sidecar — lifecycle cleanup must not throw */ }
379
+ }
424
380
  function claimSessionOwnerAndDetectRebind(cwd: string): boolean {
425
381
  try {
426
382
  const p = path.join(piGlaDir(cwd), SESSION_OWNER_FILE);
427
- let prevPid: number | null = null;
383
+ let previous: { pid?: number; shutdownReason?: string } = {};
428
384
  try {
429
- prevPid = (JSON.parse(fs.readFileSync(p, "utf-8")) as { pid?: number }).pid ?? null;
385
+ previous = JSON.parse(fs.readFileSync(p, "utf-8")) as { pid?: number; shutdownReason?: string };
430
386
  } catch { /* absent or corrupt — first boot */ }
431
387
  fs.writeFileSync(p, JSON.stringify({ pid: process.pid, at: new Date().toISOString() }));
432
- return prevPid !== null && prevPid === process.pid;
388
+ const quit = previous.shutdownReason?.trim().toLowerCase() === "quit";
389
+ return previous.pid === process.pid && !quit;
433
390
  } catch {
434
391
  return false;
435
392
  }
@@ -486,6 +443,10 @@ function probeExtensionApiStale(): boolean {
486
443
  * promise from sync archiveCurrentGoal turned it into an uncaughtException
487
444
  * and pi EXITED mid-audit. Probe first, catch anyway, ledger the skip. */
488
445
  function safeSteerUser(ctx: ExtensionContext, text: string): boolean {
446
+ if (sessionHandoffPending) {
447
+ appendLedger(ctx.cwd, "steer_skipped_handoff", { chars: text.length });
448
+ return false;
449
+ }
489
450
  if (probeExtensionApiStale()) {
490
451
  appendLedger(ctx.cwd, "steer_skipped_stale", { chars: text.length });
491
452
  return false;
@@ -505,11 +466,15 @@ function safeSteerUser(ctx: ExtensionContext, text: string): boolean {
505
466
  * and must NOT claim work started (S3's "created — starting now" lie). */
506
467
  function warnIfStaleAtEntry(ctx: ExtensionContext, what: string): boolean {
507
468
  if (!probeExtensionApiStale()) return false;
508
- // v0.30.0: a successor may already own this session (e.g. /reload
509
- // re-imported the modules) — the user's command belongs to the fresh
510
- // instance; say so softly instead of demanding a reload.
469
+ if (sessionHandoffPending) {
470
+ ctx.ui.notify(`glla: this session is handing off to a fresh pi context — ${what} will be handled after session_start.`, "info");
471
+ return true;
472
+ }
473
+ // v0.30.0: a successor may already own this session (e.g. a module
474
+ // re-import) — the user's command belongs to the fresh instance; say so
475
+ // softly instead of claiming the old handle can recover it.
511
476
  // v0.32.0: the rebind window means a fresh instance is COMING, not here —
512
- // the old message claimed "handled there" while nothing owned the session.
477
+ // the message names that handoff rather than pretending a send landed.
513
478
  if (Date.now() < sessionReplacementUntil) {
514
479
  ctx.ui.notify(`glla: this session is rebinding after /reload — ${what} will be handled by the refreshed instance; retry in a moment if it doesn't.`, "info");
515
480
  return true;
@@ -520,14 +485,12 @@ function warnIfStaleAtEntry(ctx: ExtensionContext, what: string): boolean {
520
485
  }
521
486
  appendLedger(ctx.cwd, "extension_api_stale", { where: `entry probe (${what})` });
522
487
  ctx.ui.notify(
523
- `glla: this session's extension handle is stale (pi session replacement) — ${what} can't send continuations in this process. State is safe in .pi-glla/ — run /reload (extensions rebuild in place), then /glla resume. Restart pi only if /reload fails.`,
488
+ `glla: this session's extension handle is stale (pi session replacement) — ${what} can't send continuations in this process. State is safe in .pi-glla/. A fresh session_start will resume it; if pi does not create one, restart pi normally and restore the saved work.`,
524
489
  "warning",
525
490
  );
526
- // v0.29.22: deliberately NO self-heal from the entry probe — it fires
527
- // when the user is ACTIVELY typing a /glla command, and injected
528
- // keystrokes would race their input. User-present cases keep the manual
529
- // warning; the self-heal stays on the autonomous paths (heartbeat
530
- // probe, send paths) where no one is at the keyboard.
491
+ // Entry probes never mutate the terminal. The only recovery boundary is
492
+ // pi's own session lifecycle; user-present commands keep an honest warning
493
+ // and the durable state remains available to the fresh session.
531
494
  return true;
532
495
  }
533
496
 
@@ -616,6 +579,11 @@ function resolveCarryover(ctx: ExtensionContext, trigger: "goal" | "loop" | "lis
616
579
  // pi replaces sessions (newSession/fork/reload) and stale ctx throws on use,
617
580
  // so timers must never capture a ctx — they read lastCtx at fire time.
618
581
  let lastCtx: ExtensionContext | null = null;
582
+ // v0.34.16: shutdown sets this before pi invalidates the old context. Any
583
+ // timer or late event that reaches the old module must stand down until the
584
+ // fresh session_start rebinds it.
585
+ let sessionHandoffPending = false;
586
+ const sessionTimeouts = new Set<NodeJS.Timeout>();
619
587
  // v0.23.8: the session that OWNS the loop (its sessionManager). Subagent
620
588
  // sessions (pi-subagents binds extensions there too) fire our handlers
621
589
  // with their own ctx — they must never take over lastCtx (a headless
@@ -718,15 +686,9 @@ const CONTINUATION_UNANSWERED_THROTTLE_MS = 300_000;
718
686
  // cycle. Sending 2.5s AFTER agent_end lets teardown settle; the send lands
719
687
  // and the next turn starts immediately. 2.5s per turn beats 60s per turn.
720
688
  const EAGER_CONTINUATION_SETTLE_MS = Number(process.env.GLLA_EAGER_SETTLE_MS ?? 2_500);
721
- // v0.34.13: auto-recovery ladder ("keep going unless we MUST stop — a
722
- // question, or done" — user directive 2026-08-01). A wedge that only a
723
- // /reload cures should /reload ITSELF: inject the keystrokes (v0.29.13
724
- // transport) with a sidecar marker so the fresh instance RESUMES even when
725
- // autoresume=off (the consent came from autoRecovery at recovery time, not
726
- // the restore-time setting). Throttled to one attempt per window: a
727
- // recurrence inside the window is the transcript-writer-dead class, which
728
- // only a pi restart cures — that one stays a loud stop for the human.
729
- const AUTO_RECOVERY_THROTTLE_MS = 600_000;
689
+ // v0.34.16: retain the old recovery marker for one compatibility window so
690
+ // an in-flight v0.34.15 reload can still resume once. New recovery debt uses
691
+ // the session lifecycle handoff below and never injects terminal keystrokes.
730
692
  const RECOVERY_RESUME_MARKER = "recovery-resume.json";
731
693
  const RECOVERY_RESUME_FRESH_MS = 300_000;
732
694
  // v0.29.19: dead-turn caps (agent_end exemption path). 6 consecutive
@@ -849,7 +811,6 @@ const COMPACTION_GRACE_MS = 3 * 60_000;
849
811
  // budget now spans ~5.5m) and escalate the brake cooldown per consecutive
850
812
  // brake (1m, 2m, 4m, 8m, 16m cap). A successful turn resets both.
851
813
  const ERROR_RETRY_LADDER_MS = [5_000, 15_000, 45_000, 90_000, 180_000];
852
- let errorBrakeStreak = 0;
853
814
  const SEND_REARM_LEDGER_MILESTONES_MS = [2 * 60_000, 5 * 60_000, 10 * 60_000];
854
815
  // v0.28.29: escalation is TIME-based and ACTIVITY-gated. A busy session is
855
816
  // NORMAL — the user conversing, or one long subagent turn — and the old
@@ -905,14 +866,10 @@ function escalateSendRearmStorm(ctx: ExtensionContext, kind: "continuation" | "l
905
866
  const silent = Math.round(SEND_REARM_ESCALATE_SILENT_MS / 60000);
906
867
  appendLedger(ctx.cwd, "send_rearm_escalated", { kind, afterMinutes: mins, silentMinutes: silent });
907
868
  if (kind === "loop" && isLoopActive()) {
908
- if (attemptAutoRecovery(ctx, "send-retry storm")) {
909
- appendLedger(ctx.cwd, "send_rearm_escalated_suppressed", { reason: "auto-recovery reload" });
910
- return;
911
- }
912
869
  clearLoopTimer();
913
- state.loop = { ...state.loop!, active: false, stopReason: `send-retry storm: ${mins}m of re-arms with no session activity for ${silent}m — the session is wedged. Press Escape to cancel the stuck run (pi's own rate-limit retry holds it; pi prints "escape to cancel"), then /loop resume — the loop holds on restore. If still wedged: /reload rebuilds extensions in place, then /loop resume again. Restart pi only if /reload itself fails.` };
870
+ state.loop = { ...state.loop!, active: false, stopReason: `send-retry storm: ${mins}m of re-arms with no session activity for ${silent}m — the session is wedged. Press Escape to cancel the stuck run (pi's own rate-limit retry holds it; pi prints "escape to cancel"), then /loop resume — the loop holds on restore. A fresh session_start rebinds the loop; restart pi normally only if no replacement arrives.` };
914
871
  persistState(ctx);
915
- ctx.ui.notify(`Loop stopped: send-retry storm (${mins}m, session silent ${silent}m). Escape cancels the stuck run, then /loop resume (the loop holds on restore). /reload if it persists; restart pi only if /reload fails.`, "warning");
872
+ ctx.ui.notify(`Loop stopped: send-retry storm (${mins}m, session silent ${silent}m). Escape cancels the stuck run, then /loop resume (the loop holds on restore). A fresh session_start rebinds it; restart pi normally only if no replacement arrives.`, "warning");
916
873
  notifyExternal(ctx, "Loop stopped: send-retry storm.");
917
874
  return;
918
875
  }
@@ -932,21 +889,13 @@ function escalateSendRearmStorm(ctx: ExtensionContext, kind: "continuation" | "l
932
889
  return;
933
890
  }
934
891
  if (state.goal && state.goal.status === "active") {
935
- // v0.34.13: keep going unless we MUST stop — try the auto-recovery
936
- // /reload before spending the user's attention on a pause. A reload
937
- // that fails to cure throttles the next attempt, and the pause below
938
- // fires then as today.
939
- if (attemptAutoRecovery(ctx, "send-retry storm")) {
940
- appendLedger(ctx.cwd, "send_rearm_escalated_suppressed", { reason: "auto-recovery reload" });
941
- return;
942
- }
943
892
  updateGoal({
944
893
  status: "paused",
945
894
  pauseKind: "error",
946
895
  pauseReason: `send-retry storm: ${mins}m of re-arms with no session activity for ${silent}m — the session never went idle for the continuation`,
947
- pauseSuggestedAction: "The session produced no events while the send retried (wedged queue — often pi's own rate-limit retry holding the run; pi prints 'escape to cancel'). Press Escape, then /goal resume. If still wedged: /reload rebuilds extensions in place, then /goal resume again. Restart pi only if /reload fails.",
896
+ pauseSuggestedAction: "The session produced no events while the send retried (wedged queue — often pi's own rate-limit retry holding the run; pi prints 'escape to cancel'). Press Escape, then /goal resume. A fresh session_start rebinds the goal; restart pi normally only if no replacement arrives.",
948
897
  }, ctx);
949
- ctx.ui.notify(`${goalNoun()} paused: send-retry storm (${mins}m, session silent ${silent}m). Escape cancels the stuck run, then /goal resume. /reload if it persists; restart pi only if /reload fails.`, "warning");
898
+ ctx.ui.notify(`${goalNoun()} paused: send-retry storm (${mins}m, session silent ${silent}m). Escape cancels the stuck run, then /goal resume. A fresh session_start rebinds it; restart pi normally only if no replacement arrives.`, "warning");
950
899
  notifyExternal(ctx, `${goalNoun()} paused: send-retry storm.`);
951
900
  }
952
901
  }
@@ -955,17 +904,11 @@ function escalateStallNow(ctx: ExtensionContext, threshold: number): boolean {
955
904
  if (!shouldEscalateStall(consecutiveStalls, threshold)) return false;
956
905
  consecutiveStalls = 0;
957
906
  appendLedger(ctx.cwd, "stall_escalated", { threshold, kind: isLoopActive() ? "loop" : "goal" });
958
- // v0.34.13: recovery before stop — usually throttled (the 2.5min
959
- // watchdog already tried), in which case the stop proceeds as today.
960
- if (attemptAutoRecovery(ctx, "stall escalation")) {
961
- appendLedger(ctx.cwd, "stall_escalated_suppressed", { reason: "auto-recovery reload", threshold });
962
- return true;
963
- }
964
907
  if (isLoopActive()) {
965
908
  clearLoopTimer();
966
- state.loop = { ...state.loop!, active: false, stopReason: `stalled: ${threshold} continuation refires landed no turn — the session is not continuing (wedged message queue or stale API). Press Escape to cancel any stuck run, then /loop resume — the loop holds on restore. If the handle is stale, /reload rebuilds extensions in place; restart pi only if /reload fails.` };
909
+ state.loop = { ...state.loop!, active: false, stopReason: `stalled: ${threshold} continuation refires landed no turn — the session is not continuing (wedged message queue or stale API). Press Escape to cancel any stuck run, then /loop resume — the loop holds on restore. A fresh session_start rebinds the loop or goal; restart pi normally only if no replacement arrives.` };
967
910
  persistState(ctx);
968
- ctx.ui.notify(`Loop stopped: ${threshold} refires produced no turn — the continuation is not landing. Escape cancels a stuck run, then /loop resume (the loop holds on restore). /reload if stale; restart pi only if /reload fails.`, "warning");
911
+ ctx.ui.notify(`Loop stopped: ${threshold} refires produced no turn — the continuation is not landing. Escape cancels a stuck run, then /loop resume (the loop holds on restore). A fresh session_start rebinds it; restart pi normally only if no replacement arrives.`, "warning");
969
912
  notifyExternal(ctx, "Loop stopped: stalled (continuation not landing).");
970
913
  return true;
971
914
  }
@@ -974,9 +917,9 @@ function escalateStallNow(ctx: ExtensionContext, threshold: number): boolean {
974
917
  status: "paused",
975
918
  pauseKind: "error",
976
919
  pauseReason: `stalled: ${threshold} continuation refires landed no turn`,
977
- pauseSuggestedAction: "The continuation chain is broken in this process (wedged message queue or stale API). Press Escape to cancel any stuck run, then /goal resume. If the handle is stale, /reload rebuilds extensions in place — restart pi only if /reload fails.",
920
+ pauseSuggestedAction: "The continuation chain is broken in this process (wedged message queue or stale API). Press Escape to cancel any stuck run, then /goal resume. A fresh session_start rebinds the goal; restart pi normally only if no replacement arrives.",
978
921
  }, ctx);
979
- ctx.ui.notify(`${goalNoun()} paused: ${threshold} refires produced no turn. Escape cancels a stuck run, then /goal resume. /reload if stale; restart pi only if /reload fails.`, "warning");
922
+ ctx.ui.notify(`${goalNoun()} paused: ${threshold} refires produced no turn. Escape cancels a stuck run, then /goal resume. A fresh session_start rebinds it; restart pi normally only if no replacement arrives.`, "warning");
980
923
  notifyExternal(ctx, `${goalNoun()} paused: stalled (continuation not landing).`);
981
924
  return true;
982
925
  }
@@ -1070,17 +1013,10 @@ function heartbeatTick(): void {
1070
1013
  ) {
1071
1014
  lastUnansweredAlertAt = Date.now();
1072
1015
  appendLedger(ctx.cwd, "continuation_unanswered", { silentMs: Date.now() - lastContinuationSentAt });
1073
- // v0.34.13: recover, don't just report. Only the failure modes reach
1074
- // the user: throttled (a reload already failed to cure → the
1075
- // pi-restart class) or recovery unavailable (setting off / no pane).
1076
- if (!attemptAutoRecovery(ctx, "continuation unanswered")) {
1077
- const mins = Math.round((Date.now() - lastContinuationSentAt) / 60_000);
1078
- const msg = lastAutoRecoveryAt > 0
1079
- ? `glla: auto-recovery /reload did NOT unstick this session — still no turn ${mins}m after the continuation. This class kills pi's transcript writer; only a pi RESTART cures it: restart pi in this tab, then /glla resume (the goal holds in .pi-glla state).`
1080
- : `glla: pi accepted the continuation ${mins}m ago but NO turn has started — no tool calls, no tokens, transcript frozen (the turn trigger is wedged). Cure: /reload — autoresume re-fires the ${isLoopActive() ? "loop" : "goal/list item"}. (auto-recovery is off or no tmux/WezTerm pane — /glla settings autoRecovery=on enables the automatic form.)`;
1081
- ctx.ui.notify(msg, "warning");
1082
- notifyExternal(ctx, msg);
1083
- }
1016
+ const mins = Math.round((Date.now() - lastContinuationSentAt) / 60_000);
1017
+ const msg = `glla: pi accepted the continuation ${mins}m ago but NO turn has started — no tool calls, no tokens, transcript frozen (the turn trigger is wedged). Re-sends do not unstick it. A fresh session_start will rebind the ${isLoopActive() ? "loop" : "goal/list item"}; if no replacement arrives, restart pi normally and restore the saved work.`;
1018
+ ctx.ui.notify(msg, "warning");
1019
+ notifyExternal(ctx, msg);
1084
1020
  }
1085
1021
  // v0.29.1: stranded-audit recovery. A goal left in "auditing" with NO
1086
1022
  // in-flight audit means the auditor's result never landed (wedged queue
@@ -1242,11 +1178,37 @@ function clearContinuationTimer(): void {
1242
1178
  continuationScheduledFor = null;
1243
1179
  }
1244
1180
 
1181
+ function scheduleSessionTimeout(callback: () => void, delayMs: number): NodeJS.Timeout {
1182
+ let timer: NodeJS.Timeout;
1183
+ timer = setTimeout(() => {
1184
+ sessionTimeouts.delete(timer);
1185
+ callback();
1186
+ }, delayMs);
1187
+ sessionTimeouts.add(timer);
1188
+ timer.unref?.();
1189
+ return timer;
1190
+ }
1191
+
1192
+ function clearSessionOwnedTimers(): void {
1193
+ sessionHandoffPending = true;
1194
+ clearContinuationTimer();
1195
+ clearLoopTimer();
1196
+ if (queueStuckProbe) { clearTimeout(queueStuckProbe); queueStuckProbe = null; }
1197
+ if (heartbeatTimer) { clearInterval(heartbeatTimer); heartbeatTimer = null; }
1198
+ if (uiTicker) { clearInterval(uiTicker); uiTicker = null; }
1199
+ for (const timer of sessionTimeouts) clearTimeout(timer);
1200
+ sessionTimeouts.clear();
1201
+ cancelQuotaRetry();
1202
+ lastCtx = null;
1203
+ ownerSession = null;
1204
+ }
1205
+
1245
1206
  function isActionableGoal(): boolean {
1246
1207
  return !!state.goal && state.goal.status === "active" && state.goal.autoContinue;
1247
1208
  }
1248
1209
 
1249
1210
  function freshCtx(): ExtensionContext | null {
1211
+ if (sessionHandoffPending) return null;
1250
1212
  // A captured ctx throws "stale" after session replacement. Probe cheaply;
1251
1213
  // on stale, drop it and wait for the next event to hand us a fresh one.
1252
1214
  if (!lastCtx) return null;
@@ -1259,7 +1221,39 @@ function freshCtx(): ExtensionContext | null {
1259
1221
  }
1260
1222
  }
1261
1223
 
1224
+ // v0.34.15 (hegemon 2026-08-01): pi ACCEPTED the continuation — footer showed
1225
+ // "1 queued" — but the turn trigger was dead, so the message sat queued while
1226
+ // pi idled. The 0.34.11 watchdog gates on "pi reported NO pending" and the
1227
+ // stall ladder takes ~10 minutes; a send that lands queued-without-a-turn is
1228
+ // a CONFIRMED dead trigger (hegemon law), so probe once, ~45s after every
1229
+ // landed send, and report it without terminal input. A consumed message (even
1230
+ // an instant-429 turn consumes it) or any real activity disarms the probe.
1231
+ function queueStuckProbeMs(): number {
1232
+ return Number(process.env.GLLA_QUEUE_STUCK_MS ?? 45_000);
1233
+ }
1234
+ let queueStuckProbe: ReturnType<typeof setTimeout> | null = null;
1235
+ function armQueueStuckProbe(sentAt: number): void {
1236
+ if (queueStuckProbe) clearTimeout(queueStuckProbe);
1237
+ queueStuckProbe = scheduleSessionTimeout(() => {
1238
+ queueStuckProbe = null;
1239
+ try {
1240
+ const ctx = freshCtx();
1241
+ if (!ctx) return; // no fresh lifecycle context — do not touch a stale one
1242
+ if (!isSupervising()) return; // paused/completed meanwhile
1243
+ if (lastContinuationSentAt !== sentAt) return; // a newer send armed its own probe
1244
+ if (lastRealActivityAt > sentAt) return; // the turn started and worked
1245
+ if (!ctx.isIdle()) return; // a turn is running — healthy
1246
+ if (!ctx.hasPendingMessages()) return; // consumed — even an instant 429 consumes
1247
+ appendLedger(ctx.cwd, "queue_stuck_detected", { waitedMs: Date.now() - sentAt });
1248
+ const msg = `${goalNoun()}: the continuation is QUEUED but pi won't start a turn — the turn trigger is dead (re-sends only queue). glla will resume from a fresh session_start; if no replacement arrives, restart pi normally and restore the saved work.`;
1249
+ ctx.ui.notify(msg, "warning");
1250
+ notifyExternal(ctx, msg);
1251
+ } catch { /* stale ctx — the live instance owns the probe now */ }
1252
+ }, queueStuckProbeMs());
1253
+ }
1254
+
1262
1255
  function scheduleContinuation(ctx: ExtensionContext, force = false, delayMs?: number): void {
1256
+ if (sessionHandoffPending) return;
1263
1257
  abortedStandDown = false; // v0.29.5: any explicit schedule ends the stand-down
1264
1258
  if (!isActionableGoal()) return;
1265
1259
  rememberCtx(ctx);
@@ -1273,11 +1267,11 @@ function scheduleContinuation(ctx: ExtensionContext, force = false, delayMs?: nu
1273
1267
  return;
1274
1268
  }
1275
1269
  continuationScheduledFor = goalId;
1276
- continuationTimer = setTimeout(() => sendContinuation(goalId), delay);
1277
- continuationTimer.unref?.();
1270
+ continuationTimer = scheduleSessionTimeout(() => sendContinuation(goalId), delay);
1278
1271
  }
1279
1272
 
1280
1273
  function sendContinuation(goalId: string): void {
1274
+ if (sessionHandoffPending) return;
1281
1275
  continuationTimer = null;
1282
1276
  continuationScheduledFor = null;
1283
1277
  if (!isActionableGoal()) return;
@@ -1288,16 +1282,14 @@ function sendContinuation(goalId: string): void {
1288
1282
  if (probeExtensionApiStale()) return;
1289
1283
  // No live ctx — retry shortly; the next session event will refresh it.
1290
1284
  continuationScheduledFor = goalId;
1291
- continuationTimer = setTimeout(() => sendContinuation(goalId), BACKOFF_IDLE_RETRY_MS);
1292
- continuationTimer.unref?.();
1285
+ continuationTimer = scheduleSessionTimeout(() => sendContinuation(goalId), BACKOFF_IDLE_RETRY_MS);
1293
1286
  return;
1294
1287
  }
1295
1288
  if (!ctx.isIdle() || ctx.hasPendingMessages()) {
1296
1289
  accountSendRearm(ctx, "continuation");
1297
1290
  continuationScheduledFor = goalId;
1298
1291
  // v0.28.29: backing-off cadence (was flat 50ms — 6,000 spins in 5m).
1299
- continuationTimer = setTimeout(() => sendContinuation(goalId), sendRearmDelayMs(continuationRearmStreak));
1300
- continuationTimer.unref?.();
1292
+ continuationTimer = scheduleSessionTimeout(() => sendContinuation(goalId), sendRearmDelayMs(continuationRearmStreak));
1301
1293
  return;
1302
1294
  }
1303
1295
  if (!extensionApi || extensionApiStale) return;
@@ -1315,6 +1307,7 @@ function sendContinuation(goalId: string): void {
1315
1307
  continuationRearmStreak = 0; continuationRearmSince = 0; // v0.28.5 (E3): a landed send clears the storm
1316
1308
  appendLedger(ctx.cwd, "goal_continuation_sent", { goalId });
1317
1309
  lastContinuationSentAt = Date.now();
1310
+ armQueueStuckProbe(lastContinuationSentAt);
1318
1311
  } catch (err) {
1319
1312
  appendLedger(ctx.cwd, "goal_continuation_send_failed", { goalId, error: err instanceof Error ? err.message : String(err) });
1320
1313
  // v0.26.7: stale runtime = terminal (sends can never land); anything
@@ -1328,7 +1321,7 @@ function sendContinuation(goalId: string): void {
1328
1321
  // what closes the turn: complete_goal if done, pause_goal if blocked, a tool
1329
1322
  // call otherwise. display: true — the user should see the warning too.
1330
1323
  function sendStallEscalation(ctx: ExtensionContext, nudges: number): void {
1331
- if (!extensionApi || extensionApiStale) return;
1324
+ if (sessionHandoffPending || !extensionApi || extensionApiStale) return;
1332
1325
  const remaining = HEARTBEAT_MAX_NUDGES - nudges;
1333
1326
  const text = [
1334
1327
  `[STALL WARNING ${nudges}/${HEARTBEAT_MAX_NUDGES}] The last turn produced no tool calls.`,
@@ -1350,7 +1343,7 @@ function sendStallEscalation(ctx: ExtensionContext, nudges: number): void {
1350
1343
  // sendContinuation (stale api = terminal), independent of goal state —
1351
1344
  // plain sessions truncate too.
1352
1345
  function sendLengthContinue(ctx: ExtensionContext, consecutive: number): void {
1353
- if (!extensionApi || extensionApiStale) return;
1346
+ if (sessionHandoffPending || !extensionApi || extensionApiStale) return;
1354
1347
  try {
1355
1348
  extensionApi.sendMessage({
1356
1349
  customType: GOAL_EVENT_ENTRY,
@@ -1881,7 +1874,7 @@ function fireReviewer(
1881
1874
  // to still count as "proposed" in the report + notify. Now the
1882
1875
  // failure is LOUD and the proposal goes uncounted.
1883
1876
  ctx.ui.notify(
1884
- `Postaudit /goal proposal NOT delivered: ${err instanceof Error ? err.message : String(err)} — the follow-up never reached the session. Run /reload if the session was just replaced (extensions rebuild in place), then retry.`,
1877
+ `Postaudit /goal proposal NOT delivered: ${err instanceof Error ? err.message : String(err)} — the follow-up never reached the session. Wait for a fresh session_start, then retry.`,
1885
1878
  "warning",
1886
1879
  );
1887
1880
  return false;
@@ -2017,7 +2010,7 @@ async function startDrafting(ctx: ExtensionContext, target: "goal" | "list" | "l
2017
2010
  if (isStaleApiError(err)) {
2018
2011
  extensionApiStale = true;
2019
2012
  appendLedger(ctx.cwd, "extension_api_stale", { where: "startDrafting seed" });
2020
- ctx.ui.notify("glla: can't start the drafting interview — this session's extension handle is stale (pi session replacement). Run /reload (extensions rebuild in place), then re-run the command. Restart pi only if /reload fails.", "warning");
2013
+ ctx.ui.notify("glla: can't start the drafting interview — this session's extension handle is stale (pi session replacement). A fresh session_start will rebind it; if no replacement arrives, restart pi normally, then re-run the command.", "warning");
2021
2014
  } else {
2022
2015
  ctx.ui.notify(`glla: couldn't start the drafting interview (${err instanceof Error ? err.message : String(err)}) — try again.`, "warning");
2023
2016
  }
@@ -2146,7 +2139,7 @@ async function cmdSet(args: string, ctx: ExtensionContext, skipDraft = false): P
2146
2139
  // v0.28.1 (S3): the goal is persisted — mark the interrupt so the next
2147
2140
  // fresh session LOADS it (held by default since v0.28.21), and tell the truth instead of "starting now".
2148
2141
  updateGoal({ interruptedAt: nowIso(), interruptedReason: "created in a stale session" }, ctx);
2149
- ctx.ui.notify(`Goal saved: ${shortObj(goal.objective)} — safe in .pi-glla/, but this stale process can't send continuations. Run /reload (extensions rebuild in place, state survives), then /goal resume. Restart pi only if /reload itself fails.`, "warning");
2142
+ ctx.ui.notify(`Goal saved: ${shortObj(goal.objective)} — safe in .pi-glla/, but this stale process can't send continuations. A fresh session_start will resume it; if no replacement arrives, restart pi normally, then /goal resume if autoresume is off.`, "warning");
2150
2143
  return;
2151
2144
  }
2152
2145
  ctx.ui.notify(`Goal started: ${shortObj(goal.objective)} — the auditor will verify on completion.`, "info");
@@ -2326,8 +2319,11 @@ async function showDecisionPrompt(ctx: ExtensionContext): Promise<boolean> {
2326
2319
  * disabled (/glla decisionpopup=off), or when one is already open. */
2327
2320
  function maybeDecisionPopup(ctx: ExtensionContext): void {
2328
2321
  if (!ctx.hasUI || loadSettings(ctx.cwd).decisionPopup === false) return;
2329
- setTimeout(() => {
2330
- void showDecisionPrompt(ctx).catch(() => {});
2322
+ const cwd = ctx.cwd;
2323
+ scheduleSessionTimeout(() => {
2324
+ const fresh = freshCtx();
2325
+ if (!fresh || fresh.cwd !== cwd) return;
2326
+ void showDecisionPrompt(fresh).catch(() => {});
2331
2327
  }, 600);
2332
2328
  }
2333
2329
 
@@ -2874,7 +2870,7 @@ function loopPrompt(loop: LoopState, regressionNote: string, strategyNote: strin
2874
2870
  }
2875
2871
 
2876
2872
  function scheduleLoopTick(ctx: ExtensionContext): void {
2877
- if (!isLoopActive()) return;
2873
+ if (sessionHandoffPending || !isLoopActive()) return;
2878
2874
  rememberCtx(ctx);
2879
2875
  clearLoopTimer();
2880
2876
  let delay = 0;
@@ -2883,11 +2879,11 @@ function scheduleLoopTick(ctx: ExtensionContext): void {
2883
2879
  } catch {
2884
2880
  return;
2885
2881
  }
2886
- loopTimer = setTimeout(() => sendLoopTurn(), delay);
2887
- loopTimer.unref?.();
2882
+ loopTimer = scheduleSessionTimeout(() => sendLoopTurn(), delay);
2888
2883
  }
2889
2884
 
2890
2885
  function sendLoopTurn(): void {
2886
+ if (sessionHandoffPending) return;
2891
2887
  loopTimer = null;
2892
2888
  if (!isLoopActive() || !extensionApi) return;
2893
2889
  const ctx = freshCtx();
@@ -2899,8 +2895,7 @@ function sendLoopTurn(): void {
2899
2895
  if (probeExtensionApiStale()) return;
2900
2896
  loopRearmStreak++;
2901
2897
  } else accountSendRearm(ctx, "loop");
2902
- loopTimer = setTimeout(() => sendLoopTurn(), sendRearmDelayMs(loopRearmStreak)); // v0.28.29: backing-off cadence
2903
- loopTimer.unref?.();
2898
+ loopTimer = scheduleSessionTimeout(() => sendLoopTurn(), sendRearmDelayMs(loopRearmStreak)); // v0.28.29: backing-off cadence
2904
2899
  return;
2905
2900
  }
2906
2901
  const loop = state.loop!;
@@ -2997,6 +2992,7 @@ function sendLoopTurn(): void {
2997
2992
  loopRearmStreak = 0; loopRearmSince = 0; // v0.28.5 (E3): a landed turn clears the storm
2998
2993
  appendLedger(ctx.cwd, "loop_turn_sent", { iteration: loop.iteration });
2999
2994
  lastContinuationSentAt = Date.now();
2995
+ armQueueStuckProbe(lastContinuationSentAt);
3000
2996
  } catch (err) {
3001
2997
  // stale API — next agent_end reschedules (but if none comes, the
3002
2998
  // heartbeat's stall escalation stops the spin — v0.26.1).
@@ -4212,7 +4208,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4212
4208
  // refused; the dialog simply can't render in a doomed process.
4213
4209
  extensionApiStale = true;
4214
4210
  appendLedger(liveCtx.cwd, "extension_api_stale", { where: "batch confirm" });
4215
- return { content: [{ type: "text", text: "The Confirm dialog could not render: pi invalidated this session's extension handle (session replacement). This is NOT a rejection — do NOT refine or re-propose. Tell the user to restart pi, then re-run the drafting flow." }], details: {} };
4211
+ return { content: [{ type: "text", text: "The Confirm dialog could not render: pi invalidated this session's extension handle (session replacement). This is NOT a rejection — do NOT refine or re-propose. Wait for a fresh session_start, then re-run the drafting flow." }], details: {} };
4216
4212
  }
4217
4213
  batchConfirmed = c === "yes";
4218
4214
  }
@@ -4256,7 +4252,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4256
4252
  // v0.28.1 (T1): a stale dialog is NOT "Draft rejected by the user".
4257
4253
  extensionApiStale = true;
4258
4254
  appendLedger(liveCtx.cwd, "extension_api_stale", { where: "draft confirm" });
4259
- return { content: [{ type: "text", text: "The Confirm dialog could not render: pi invalidated this session's extension handle (session replacement). This is NOT a rejection — do NOT refine or re-propose. Tell the user to restart pi, then re-run the drafting flow." }], details: {} };
4255
+ return { content: [{ type: "text", text: "The Confirm dialog could not render: pi invalidated this session's extension handle (session replacement). This is NOT a rejection — do NOT refine or re-propose. Wait for a fresh session_start, then re-run the drafting flow." }], details: {} };
4260
4256
  }
4261
4257
  confirmed = c === "yes";
4262
4258
  }
@@ -6206,7 +6202,7 @@ export default function (pi: ExtensionAPI): void {
6206
6202
  // EVERY post-grace tick until agent_start discharges it).
6207
6203
  postCompactResumeOwed = true;
6208
6204
  postCompactResyncPending = true;
6209
- const settle = setTimeout(() => {
6205
+ scheduleSessionTimeout(() => {
6210
6206
  const c = freshCtx();
6211
6207
  if (!c) return;
6212
6208
  try {
@@ -6219,7 +6215,6 @@ export default function (pi: ExtensionAPI): void {
6219
6215
  /* settle race — the 60s heartbeat covers it */
6220
6216
  }
6221
6217
  }, 2000);
6222
- settle.unref?.();
6223
6218
  // v0.29.21: a SECOND settle at grace expiry. The 2s settle almost
6224
6219
  // always loses (pi is mid-compact / mid-resumed-turn then), and the
6225
6220
  // heartbeat's first post-grace tick lands up to one interval late —
@@ -6228,7 +6223,7 @@ export default function (pi: ExtensionAPI): void {
6228
6223
  // 195.8k after two output-limit turns, zero rearms after the compact
6229
6224
  // event, recovery only at 04:34:48 via the post-grace heartbeat;
6230
6225
  // ~4 min that read as a stoppage). Refire the moment the grace ends.
6231
- const graceSettle = setTimeout(() => {
6226
+ scheduleSessionTimeout(() => {
6232
6227
  const c = freshCtx();
6233
6228
  if (!c) return;
6234
6229
  try {
@@ -6241,7 +6236,6 @@ export default function (pi: ExtensionAPI): void {
6241
6236
  /* settle race — the 60s heartbeat covers it */
6242
6237
  }
6243
6238
  }, COMPACTION_GRACE_MS + 2_000);
6244
- graceSettle.unref?.();
6245
6239
  });
6246
6240
 
6247
6241
  pi.on("message_start", async (event: any, _ctx: ExtensionContext) => {
@@ -6316,15 +6310,24 @@ export default function (pi: ExtensionAPI): void {
6316
6310
  // tells the stale probe that a rebind (session_start) is imminent.
6317
6311
  const shutdownReason = typeof event?.reason === "string" ? event.reason : "unknown";
6318
6312
  appendLedger(ctx.cwd, "session_shutdown", { reason: shutdownReason });
6313
+ markSessionOwnerShutdown(ctx.cwd, shutdownReason);
6314
+ writeSessionHandoff(ctx, shutdownReason);
6319
6315
  sessionReplacementUntil = Date.now() + SESSION_REBIND_GRACE_MS;
6316
+ clearSessionOwnedTimers();
6317
+ registeredCtx = null;
6318
+ toolHealNotified = false;
6320
6319
  });
6321
6320
 
6322
6321
  pi.on("session_start", async (event: any, ctx: ExtensionContext) => {
6323
- rememberCtx(ctx);
6324
6322
  // v0.23.8: subagent sessions (pi-subagents binds extensions there too)
6325
6323
  // are workers — never run the restore gate or reschedule the loop from
6326
6324
  // a foreign session.
6327
6325
  if (isForeignCtx(ctx)) return;
6326
+ extensionApi = pi;
6327
+ sessionHandoffPending = false;
6328
+ rememberCtx(ctx);
6329
+ startHeartbeat();
6330
+ startUITicker();
6328
6331
  // v0.30.0: rebind bookkeeping — claim ownership, close any replacement
6329
6332
  // window, and reset a stale flag left over from the PREVIOUS session's
6330
6333
  // invalidation. pi rebinds THIS module to the new session (switch) or
@@ -6344,10 +6347,12 @@ export default function (pi: ExtensionAPI): void {
6344
6347
  const stillStale = probeExtensionApiStale();
6345
6348
  appendLedger(ctx.cwd, "stale_flag_reset_on_rebind", { reason: startReason, stillStale });
6346
6349
  if (stillStale) {
6347
- ctx.ui.notify("glla: session rebound but the extension handle is still stale — run /reload (extensions rebuild in place), then /glla resume.", "warning");
6350
+ ctx.ui.notify("glla: session rebound but the extension handle is still stale — waiting for another fresh session_start; restart pi normally only if no replacement arrives, then /glla resume.", "warning");
6348
6351
  }
6349
6352
  }
6350
6353
  state = readState(ctx.cwd);
6354
+ const handoffResume = consumeSessionHandoff(ctx.cwd);
6355
+ if (handoffResume) appendLedger(ctx.cwd, "session_handoff_resumed", { pid: process.pid, reason: startReason });
6351
6356
  // v0.28.14: snapshot carryover BEFORE any restore logic mutates state —
6352
6357
  // a paused goal, waiting list items, or a loop that was live/held when
6353
6358
  // the last session ended. Resolved once at the first NEW activation.
@@ -6445,16 +6450,17 @@ export default function (pi: ExtensionAPI): void {
6445
6450
  persistState(ctx);
6446
6451
  appendLedger(ctx.cwd, "audit_loop_target_migrated", { from: "audit-every-iteration", to: "fix-first" });
6447
6452
  }
6448
- // v0.34.13: an auto-recovery /reload carries its own resume consent —
6449
- // the sidecar marker overrides autoresume=off for THIS restore only.
6453
+ // v0.34.15 compatibility: consume one legacy recovery marker if an older
6454
+ // process wrote it before this lifecycle-first build landed.
6450
6455
  const recoveryResume = consumeRecoveryResume(ctx.cwd);
6451
- // v0.34.14: a /reload rebind (same pi pid) ALWAYS resumes — the session
6452
- // is live mid-work; holding is the "list is not continuing" bug.
6456
+ // v0.34.16: a same-process lifecycle handoff is explicit continuation
6457
+ // debt, so it resumes independently of the cold-boot autoResume setting.
6458
+ // A same-pid owner rebind is the second same-process signal.
6453
6459
  const rebindResume = claimSessionOwnerAndDetectRebind(ctx.cwd);
6454
6460
  if (rebindResume) appendLedger(ctx.cwd, "rebind_resume", { pid: process.pid });
6455
6461
  if (isLoopActive()) {
6456
6462
  const l = state.loop!;
6457
- if (autoResume || recoveryResume || rebindResume) {
6463
+ if (autoResume || recoveryResume || rebindResume || handoffResume) {
6458
6464
  ctx.ui.notify(
6459
6465
  `Resuming loop (iteration ${l.iteration}/${l.maxIterations > 0 ? l.maxIterations : "∞"}, best ${l.bestValue ?? "n/a"}, stall ${l.stallCount}/${l.plateauWindow}): ${l.target.slice(0, 60)}`,
6460
6466
  "info",
@@ -6475,7 +6481,7 @@ export default function (pi: ExtensionAPI): void {
6475
6481
  // "load it but not auto start it"). Interrupted goals hold like
6476
6482
  // everything else; autoresume=on (unattended rigs) still auto-resumes
6477
6483
  // them, and the marker is cleared only on that promised auto-resume.
6478
- if (autoResume || recoveryResume || rebindResume) {
6484
+ if (autoResume || recoveryResume || rebindResume || handoffResume) {
6479
6485
  // v0.28.1 (S2): clear the stale-handle interrupt marker — this IS
6480
6486
  // the auto-resume the marker promised.
6481
6487
  if (wasInterrupted) updateGoal({ interruptedAt: undefined, interruptedReason: undefined }, ctx);
@@ -6731,12 +6737,22 @@ export default function (pi: ExtensionAPI): void {
6731
6737
  // 60-second provider hiccup waiting on a manual /goal resume.
6732
6738
  const detail = text.trim() ? ` (last: ${text.trim().replace(/\s+/g, " ").slice(0, 160)})` : "";
6733
6739
  const reason = `5 consecutive errors${detail}`;
6740
+ // v0.34.15: the streak now lives ON THE GOAL so it survives the
6741
+ // auto-recovery /reloads that used to zero the module counter —
6742
+ // hegemon 2026-08-01: a hard-exhausted MiniMax plan churned 1-minute
6743
+ // probes for an hour because every reload reset the ladder to rung 1
6744
+ // and the 6-brake park (v0.29.9) could never engage.
6745
+ const brakeStreak = state.goal!.errorBrakeStreak ?? 0;
6746
+ // v0.34.15: a quota/rate-limit wall is NOT a flake — the card must
6747
+ // say "resuming won't help; switch /model or wait out the window"
6748
+ // (the raw 429 text was in `detail` but nobody parses JSON on a card).
6749
+ const quotaWall = /rate.?limit|usage limit|quota|insufficient|credits/i.test(detail);
6734
6750
  // v0.29.1: brake-cycle CAP. The v0.28.25 ladder slows the thrash
6735
6751
  // (1m→16m) but never STOPS it — junk-runner/hellhunter/pully each
6736
6752
  // burned 4+ pause↔retry cycles against provider windows that last
6737
6753
  // hours. After 6 consecutive brakes: park. v0.29.9: the park keeps
6738
6754
  // probing at the top of each hour (clock-aligned window resets).
6739
- if (errorBrakeStreak >= 6) {
6755
+ if (brakeStreak >= 6) {
6740
6756
  // v0.29.9: park — but keep probing at the top of each hour
6741
6757
  // (user: "simply adding an hourly retry … just to pick up work
6742
6758
  // faster assuming the retry expired"). Coding-plan rate-limit
@@ -6750,18 +6766,20 @@ export default function (pi: ExtensionAPI): void {
6750
6766
  status: "paused",
6751
6767
  pauseKind: "error",
6752
6768
  pauseReason: `${reason} — 6 error-brakes in a row; the provider has been erroring for an extended window`,
6753
- pauseSuggestedAction: "Probing at the top of each hour — rate-limit windows typically expire on clock-hour boundaries. /goal resume retries now.",
6769
+ pauseSuggestedAction: quotaWall
6770
+ ? "Provider quota/rate-limit wall — resuming won't help until the window resets. Hourly top-of-hour probes will pick work back up; switch /model to a different provider to continue immediately."
6771
+ : "Probing at the top of each hour — rate-limit windows typically expire on clock-hour boundaries. /goal resume retries now.",
6754
6772
  }, ctx);
6755
- ctx.ui.notify(`${goalNoun()} parked: ${reason} — 6 brakes in a row. Hourly top-of-hour probes will pick work back up when the window opens; /goal resume retries now.`, "warning");
6773
+ ctx.ui.notify(`${goalNoun()} parked: ${reason} — 6 brakes in a row. ${quotaWall ? "Quota/rate-limit wall — switching /model continues immediately; otherwise hourly" : "Hourly"} top-of-hour probes will pick work back up when the window opens.`, "warning");
6756
6774
  notifyExternal(ctx, `${goalNoun()} parked: provider erroring across 6 error-brake cycles — hourly top-of-hour probes scheduled.`);
6757
- appendLedger(ctx.cwd, "error_brake_capped", { streak: errorBrakeStreak, reason });
6775
+ appendLedger(ctx.cwd, "error_brake_capped", { streak: brakeStreak, reason });
6758
6776
  const probeMs = msUntilNextHourBoundary(Date.now());
6759
6777
  scheduleQuotaRetry(ctx, probeMs / 1000, reason, () => {
6760
6778
  // Re-check: only probe if STILL parked by the error-brake cap —
6761
6779
  // a user pause/resume/cancel meanwhile is never stomped.
6762
6780
  if (state.goal && state.goal.status === "paused" && state.goal.pauseKind === "error"
6763
6781
  && (state.goal.pauseReason ?? "").includes("error-brakes in a row")) {
6764
- appendLedger(ctx.cwd, "hourly_rate_probe", { goalId: state.goal.id, streak: errorBrakeStreak });
6782
+ appendLedger(ctx.cwd, "hourly_rate_probe", { goalId: state.goal.id, streak: state.goal.errorBrakeStreak ?? 0 });
6765
6783
  updateGoal({ status: "active" }, ctx);
6766
6784
  appendLedger(ctx.cwd, "goal_resumed", { via: "hourly-rate-probe" });
6767
6785
  ctx.ui.notify("Hourly probe: resuming (rate-limit windows typically expire at the top of the hour).", "info");
@@ -6772,17 +6790,19 @@ export default function (pi: ExtensionAPI): void {
6772
6790
  }
6773
6791
  // v0.28.25: the cooldown escalates per CONSECUTIVE brake — a fleet-wide
6774
6792
  // 403 window is not cleared by re-braking every 60 seconds.
6775
- const cooldownMs = 60_000 * 2 ** Math.min(errorBrakeStreak, 4);
6793
+ const cooldownMs = 60_000 * 2 ** Math.min(brakeStreak, 4);
6776
6794
  const cooldownMin = Math.round(cooldownMs / 60_000);
6777
- errorBrakeStreak++;
6778
6795
  updateGoal({
6779
6796
  status: "paused",
6780
6797
  pauseKind: "wait",
6781
6798
  pauseResumeAt: new Date(Date.now() + cooldownMs).toISOString(),
6782
6799
  pauseReason: reason,
6783
- pauseSuggestedAction: `Transient provider flake? The goal auto-resumes once in ${cooldownMin}m if still paused for this reason — or /goal resume now.`,
6800
+ errorBrakeStreak: brakeStreak + 1,
6801
+ pauseSuggestedAction: quotaWall
6802
+ ? `Provider quota/rate-limit wall — resuming won't help until the window resets. Switch /model to a different provider to continue now, or let the probe auto-resume in ${cooldownMin}m.`
6803
+ : `Transient provider flake? The goal auto-resumes once in ${cooldownMin}m if still paused for this reason — or /goal resume now.`,
6784
6804
  }, ctx);
6785
- ctx.ui.notify(`Goal paused: ${reason}.`, "warning");
6805
+ ctx.ui.notify(`Goal paused: ${reason}.${quotaWall ? " Quota/rate-limit wall — resuming won't help until the window resets; switch /model to continue now." : ""}`, "warning");
6786
6806
  notifyExternal(ctx, `Goal paused: ${reason}.`);
6787
6807
  appendLedger(ctx.cwd, "goal_paused", { reason });
6788
6808
  scheduleQuotaRetry(ctx, cooldownMs / 1000, reason, () => {
@@ -6833,7 +6853,8 @@ export default function (pi: ExtensionAPI): void {
6833
6853
  } else {
6834
6854
  consecutiveErrorIterations = 0;
6835
6855
  consecutiveAbortIterations = 0;
6836
- errorBrakeStreak = 0; // v0.28.25: a healthy turn clears the brake cooldown
6856
+ // v0.28.25/v0.34.15: a healthy turn clears the (now persisted) brake streak
6857
+ if ((state.goal?.errorBrakeStreak ?? 0) > 0) updateGoal({ errorBrakeStreak: undefined }, ctx);
6837
6858
  }
6838
6859
 
6839
6860
  // No wall-clock cap by design: a goal ends via completion, explicit
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.34.14",
3
+ "version": "0.34.16",
4
4
  "description": "Mission control for autonomous pi: interview-drafted goals, an audited task queue, and forever-loops (metric, spec, project-audit) that run for hours. An isolated extension-less auditor re-verifies every completion with raw evidence; confirmed drafts, decision pauses and consent gates keep you in charge.",
5
5
  "license": "MIT",
6
6
  "author": "dracon",
@@ -53,6 +53,7 @@
53
53
  "interruptedAt": { "type": "string" },
54
54
  "interruptedReason": { "type": "string" },
55
55
  "auditInfraStreak": { "type": "number" },
56
+ "errorBrakeStreak": { "type": "number" },
56
57
  "pendingCompletion": { "type": "object" },
57
58
  "createdVia": { "type": "string" },
58
59
  "activePath": { "type": "string" },