pi-goal-list-loop-audit 0.34.15 → 0.34.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -235,7 +235,27 @@ No external watchdog plugin needed. It also recovers **stranded audits**
235
235
  (v0.29.1): a goal stuck in `auditing` with no auditor session alive re-runs
236
236
  the stored claim after 90s instead of black-holing. Storm protection: the
237
237
  send→pause→notify path rearms once per cycle and loud-stops after a 6-error
238
- brake streak, so a broken provider can't spin forever.
238
+ brake streak, so a broken provider can't spin forever. A confirmed queued
239
+ continuation with no turn is reported by the queue-stuck probe; it does not
240
+ inject terminal input.
241
+
242
+ ## Session replacement and stale handles
243
+
244
+ Recovery crosses pi's lifecycle boundary. On `session_shutdown`, glla records
245
+ `.pi-glla/session-handoff.json` for a supervising, non-quit replacement,
246
+ ledgering the reason and stopping every session-owned timer. On the fresh
247
+ `session_start`, the new context consumes only fresh, same-process debt and
248
+ continues the saved goal or loop. Timers resolve a fresh context when they fire;
249
+ they do not retain an old `ctx` or `pi` handle.
250
+
251
+ A user quit is not continuation consent: `reason: "quit"` is ledgered as
252
+ `session_handoff_suppressed`, leaves no handoff debt, and does not receive
253
+ same-pid rebind consent. An orphan with no fresh lifecycle event is reported
254
+ honestly because an invalidated extension cannot repair its own pi host; if the
255
+ warning says no replacement arrived, restart pi normally and let the saved
256
+ `.pi-glla/` state restore. There are no automatic `/reload` keystrokes and no
257
+ mux dependency. The legacy `autoReloadOnStale` and `autoRecovery` fields remain
258
+ only as deprecated settings-file compatibility fields; they are ignored.
239
259
 
240
260
  **User aborts mean STOP** (v0.29.4): Esc-aborting a turn stands the chain
241
261
  down with a named notify (`/goal resume` to continue) — it does NOT count
package/docs/DESIGN.md CHANGED
@@ -90,6 +90,26 @@ architectural decisions that changed the SHAPE of the system:
90
90
  Confirm path; it never audits work — the isolated auditor is the only
91
91
  verifier.
92
92
 
93
+ ## Addendum v0.34.16 (lifecycle handoff)
94
+
95
+ - **Recovery crosses pi's lifecycle, never the terminal**: `session_shutdown`
96
+ persists fresh same-process continuation debt in
97
+ `.pi-glla/session-handoff.json`, records the shutdown reason, and clears all
98
+ session-owned timers. A fresh `session_start` consumes matching debt and
99
+ continues from its new context. No stale callback is allowed to use the old
100
+ `ctx` or `pi` reference.
101
+ - **Quit is not implicit resume consent**: a shutdown with `reason: "quit"`
102
+ removes any handoff debt, records `session_handoff_suppressed`, and marks
103
+ the owner sidecar so the next same-pid startup is not mistaken for a
104
+ replacement rebind. The global `autoResume` policy remains independent and
105
+ explicit.
106
+ - **True orphans stay honest**: once pi invalidates an extension without a
107
+ fresh lifecycle event, the extension cannot repair its host. glla stops
108
+ stale work, preserves the artifact, and tells the user to restart pi only
109
+ when no replacement arrives. `autoReloadOnStale` and `autoRecovery` remain
110
+ deprecated deserialization compatibility fields; they do not select a
111
+ transport.
112
+
93
113
  ## Addendum v0.4.0 (completion)
94
114
 
95
115
  - **Auditor compaction enabled** (flaw #3 — the last open one). Safety:
@@ -215,10 +215,11 @@ export function buildStatusText(state: State, audit?: AuditDisplayProgress | nul
215
215
  }
216
216
  if (g.status === "active") {
217
217
  // v0.28.1 (S1/S2): a stale-handle interrupt keeps the goal ACTIVE.
218
- // v0.29.11: the fresh session HOLDS it (hold-everything restore gate;
219
- // autoresume=on resumes for you) — name the verb, don't promise auto.
218
+ // v0.34.16: a fresh session_start owns the handoff. A cold boot still
219
+ // follows the global autoResume setting, so the widget names the actual
220
+ // lifecycle rather than promising terminal keystroke recovery.
220
221
  if (g.interruptedAt) {
221
- return `glla: ${paint(theme, "error", "⚠ interrupted — stale handle · /reload → /glla resume")}${heldSuffix}`;
222
+ return `glla: ${paint(theme, "error", "⚠ interrupted — stale handle · fresh session_start resumes")}${heldSuffix}`;
222
223
  }
223
224
  // v0.24.7: list policy gets its own wording — a queue item is not a goal.
224
225
  // v0.28.11 (U10): goal policy joins it — "list 29" read as a command
@@ -68,18 +68,12 @@ export interface Settings {
68
68
  /** Consecutive stuck interventions before a loop stops (default 5,
69
69
  * 10 under aggressiveMode). */
70
70
  stuckMaxInterventions?: number;
71
- /** v0.29.13: on a stale-handle terminal, inject /reload into our own
72
- * tmux pane (keystroke self-heal; pi walls ctx.reload() behind
73
- * assertActive). Default true; only acts when TMUX and TMUX_PANE are
74
- * set — pi outside tmux just gets the manual warning. */
71
+ /** @deprecated v0.34.16: retained so older settings files deserialize, but
72
+ * ignored. Recovery now uses session_shutdown/session_start handoff and
73
+ * never injects terminal keystrokes. */
75
74
  autoReloadOnStale?: boolean;
76
- /** v0.34.13: auto-recovery ladder — when a wedge is detected that only a
77
- * /reload cures (unanswered continuation, send-retry storm), inject the
78
- * /reload ITSELF via the v0.29.13 tmux/WezTerm transport and resume the
79
- * goal/loop after the rebuild (sidecar marker — autoresume=off is a
80
- * restore-time setting, not a recovery veto). Default true. The one
81
- * class it does NOT cross: pi restart (transcript-writer dead) — that
82
- * stays a loud stop for the human. */
75
+ /** @deprecated v0.34.16: retained for settings-file compatibility, but
76
+ * ignored. Lifecycle handoff is always enabled. */
83
77
  autoRecovery?: boolean;
84
78
  /** v0.26.1: consecutive heartbeat refires without a real turn before
85
79
  * the goal pauses / loop stops (default 5; 0 = never escalate). */
@@ -214,8 +208,6 @@ export const SETTINGS_KEYS: Array<keyof Settings> = [
214
208
  "quotaRetryMinutes",
215
209
  "stuckMaxInterventions",
216
210
  "stallEscalationRefires",
217
- "autoReloadOnStale",
218
- "autoRecovery",
219
211
  "stallShortWords",
220
212
  "stallSimilarityThreshold",
221
213
  "postaudit",
@@ -15,7 +15,6 @@
15
15
  */
16
16
 
17
17
  import * as fs from "node:fs";
18
- import { exec } from "node:child_process";
19
18
  import * as os from "node:os";
20
19
  import * as path from "node:path";
21
20
 
@@ -221,18 +220,16 @@ const GOAL_EVENT_ENTRY = "goal-event";
221
220
  // not on ExtensionContext, so continuation sends need it at module scope.
222
221
  let extensionApi: ExtensionAPI | null = null;
223
222
  // v0.26.7: pi invalidates the extension runtime on session replacement
224
- // (newSession/fork/switchSession/reload — and the compaction path reaches
225
- // it via teardownCurrent in pi 0.82.x). Once stale, every sendMessage
223
+ // (newSession/fork/switchSession/reload). Once stale, every sendMessage
226
224
  // throws FOREVER in this process — retrying for hours is the hegemon
227
225
  // failure shape. Detect the stale signature once and go terminally loud.
228
226
  let extensionApiStale = false;
229
227
  // v0.32.0: CRITICAL — goStaleTerminal must gate on its OWN flag, not
230
228
  // extensionApiStale: probeExtensionApiStale() sets extensionApiStale on
231
229
  // detection, so the heartbeat's `probe → goStaleTerminal` sequence always
232
- // found the flag already true and returned silently — orphan-stale recovery
233
- // (ledger, loop stop, interruptedAt, warn, AUTO-RELOAD SELF-HEAL) was dead
234
- // code since v0.29.11. Field proof: hegemon sat stale for days and the
235
- // wezterm self-heal never fired.
230
+ // found the flag already true and returned silently. The terminal orphan
231
+ // path must still ledger the stale handle, stop stale work, and preserve the
232
+ // interrupt marker so a later fresh lifecycle can restore it.
236
233
  let staleTerminalDone = false;
237
234
 
238
235
  /** v0.26.7: a stale api is terminal for this process — go loudly with
@@ -241,25 +238,14 @@ let staleTerminalDone = false;
241
238
  * pausing — the restore gate only auto-resumes ACTIVE goals, so pausing
242
239
  * here stranded goals until manual /goal resume (hegemon/sraaal shape).
243
240
  * sendContinuation's extensionApiStale guard already stops further sends
244
- * in this doomed process; the next fresh session auto-resumes. */
245
- /** v0.30.0: rebind-first session-replacement survival. pi's sanctioned
246
- * pattern (docs/extensions.md lifecycle + the stale error text itself):
247
- * session_shutdown → cleanup, session_start → re-establish with the NEW
248
- * ctx. glla used to treat every stale handle as terminal ("run /reload"),
249
- * but three replacement shapes need three responses:
250
- * (a) switch (resume/new/fork): pi rebinds THIS module to the new
251
- * session — session_start delivers a fresh ctx. No user action, no
252
- * warning; reset the stale flag via a re-probe and continue.
253
- * (b) /reload: pi re-imports the extension modules — a SUCCESSOR
254
- * instance owns this cwd in the same process. The old module stands
255
- * down silently (owner-file check) instead of screaming + injecting
256
- * /reload (v0.29.22's injection is right for orphans, wrong here).
257
- * (c) orphan: the session died with NO replacement (hegemon 2026-07-31:
258
- * handle dead ~06:03, zero ledger events for 5h). Only a rebuild
259
- * revives extension function — goStaleTerminal's warning + self-heal
260
- * stays for this case ONLY.
261
- * session_shutdown is now ledgered with pi's reason, so the next
262
- * unexplained disposal is attributable from the ledger alone. */
241
+ * in this doomed process; the next fresh session can restore the work. */
242
+ /** v0.34.16: lifecycle-first session-replacement survival. pi's
243
+ * sanctioned pattern (docs/extensions.md lifecycle + the stale error text):
244
+ * session_shutdown → persist handoff debt + stop old timers,
245
+ * session_start → re-establish with the NEW ctx and consume the debt.
246
+ * A successor module may still stand down via the owner file. An orphan with
247
+ * no replacement is reported honestly: an invalid extension cannot repair its
248
+ * own pi host, so glla never injects terminal keystrokes. */
263
249
  const SESSION_REBIND_GRACE_MS = 60_000;
264
250
  let sessionReplacementUntil = 0;
265
251
  const instanceStartedAt = Date.now();
@@ -301,12 +287,7 @@ function absorbStaleIfSuperseded(ctx: ExtensionContext): boolean {
301
287
  appendLedger(ctx.cwd, "zombie_stood_down", { owner: owner.instanceId });
302
288
  zombieStoodDown = true;
303
289
  extensionApiStale = true; // silence the send paths WITHOUT the terminal theatre
304
- clearLoopTimer();
305
- if (continuationTimer) { clearTimeout(continuationTimer); continuationTimer = null; }
306
- // v0.32.0: the superseded module's heartbeat + UI ticker were IMMORTAL
307
- // (clearInterval appeared nowhere) — N /reloads = N×2 zombie tickers.
308
- if (heartbeatTimer) { clearInterval(heartbeatTimer); heartbeatTimer = null; }
309
- if (uiTicker) { clearInterval(uiTicker); uiTicker = null; }
290
+ clearSessionOwnedTimers();
310
291
  return true;
311
292
  }
312
293
  return false;
@@ -317,10 +298,10 @@ function goStaleTerminal(ctx: ExtensionContext, where: string): void {
317
298
  staleTerminalDone = true;
318
299
  extensionApiStale = true;
319
300
  appendLedger(ctx.cwd, "extension_api_stale", { where, kind: isLoopActive() ? "loop" : "goal" });
320
- const guidance = "pi invalidated this session's extension handle (session replacement — the session was disposed and this process's sends can never land). Run /reload — extensions rebuild IN PLACE, no pi restart needed — then /glla resume (autoresume=on resumes for you). Restart pi only if /reload itself fails.";
301
+ const guidance = "pi invalidated this session's extension handle without delivering a replacement session. glla stopped stale sends and kept the work safe in .pi-glla/. A fresh session_start will resume it; if pi does not create one, restart pi normally and glla will restore the saved work.";
321
302
  // v0.32.0: kill the continuation re-arm too — otherwise an orphaned goal
322
303
  // keeps spinning a flat 50ms retry below every watchdog.
323
- if (continuationTimer) { clearTimeout(continuationTimer); continuationTimer = null; continuationScheduledFor = null; }
304
+ clearSessionOwnedTimers();
324
305
  if (isLoopActive()) {
325
306
  clearLoopTimer();
326
307
  state.loop = { ...state.loop!, active: false, stopReason: `extension api stale: ${guidance}` };
@@ -329,107 +310,83 @@ function goStaleTerminal(ctx: ExtensionContext, where: string): void {
329
310
  updateGoal({ interruptedAt: nowIso(), interruptedReason: `extension api stale (${where})` }, ctx);
330
311
  }
331
312
  ctx.ui.notify(`glla: ${guidance}`, "warning");
332
- notifyExternal(ctx, `glla: extension api stale — run /reload, then /glla resume. (${where})`);
333
- attemptAutoReload(ctx, where);
313
+ notifyExternal(ctx, `glla: extension api stale — waiting for a fresh session_start; restart pi normally only if no replacement arrives. (${where})`);
334
314
  }
335
315
 
336
- /** v0.29.13: automatic recovery — the zombie handle can't call ctx.reload()
337
- * (pi walls EVERY runtime method behind assertActive), but fs/child_process
338
- * are extension-side and still work. Inject /reload as keystrokes into our
339
- * own terminal pane: pi rebuilds the extension runtime in place, the fresh
340
- * instance loads .pi-glla state and holds (autoresume=on resumes for you).
341
- * v0.29.22: transport-generalized — tmux OR WezTerm. Field: this rig runs
342
- * WezTerm (TERM_PROGRAM=WezTerm, WEZTERM_PANE set, no TMUX), so the
343
- * tmux-only gate failed silently 100% of the time — auto_reload_injected
344
- * never fired fleet-wide, and every stale handle fell back to the manual
345
- * warning (user: "stopping and told to reload is common"). v0.29.22 also
346
- * fires from the entry-probe path (the most common stale discovery).
347
- * Opt out: autoReloadOnStale=false. Best-effort: the manual warning
348
- * already fired, so failures cost nothing. */
349
- function attemptAutoReload(ctx: ExtensionContext, where: string): boolean {
316
+ /** v0.34.16: lifecycle handoff replaces terminal keystroke injection. A
317
+ * stale extension cannot call pi, so recovery must cross the lifecycle
318
+ * boundary: session_shutdown records durable resume debt, clears every timer
319
+ * that could retain the old context, and session_start consumes the debt from
320
+ * a fresh context. */
321
+ const SESSION_HANDOFF_FILE = "session-handoff.json";
322
+ const SESSION_HANDOFF_FRESH_MS = 300_000;
323
+ function sessionHandoffPath(cwd: string): string {
324
+ return path.join(piGlaDir(cwd), SESSION_HANDOFF_FILE);
325
+ }
326
+ function writeSessionHandoff(ctx: ExtensionContext, reason: string): boolean {
327
+ if (!isSupervising()) return false;
328
+ // A user quit is an explicit stop, not a replacement boundary. Do not
329
+ // leave debt that could silently resume the work on a later startup;
330
+ // global autoResume may still apply by its own explicit policy.
331
+ if (reason.trim().toLowerCase() === "quit") {
332
+ try { fs.rmSync(sessionHandoffPath(ctx.cwd), { force: true }); } catch { /* advisory cleanup */ }
333
+ appendLedger(ctx.cwd, "session_handoff_suppressed", { reason });
334
+ return false;
335
+ }
350
336
  try {
351
- if (loadSettings(ctx.cwd).autoReloadOnStale === false) return false;
352
- const tmuxPane = process.env.TMUX_PANE;
353
- const wezPane = process.env.WEZTERM_PANE;
354
- let transport: "tmux" | "wezterm";
355
- let cmd: string;
356
- let pane: string;
357
- if (process.env.TMUX && tmuxPane && /^%\d+$/.test(tmuxPane)) {
358
- transport = "tmux";
359
- pane = tmuxPane;
360
- // v0.29.22: NO leading Escape — a late zombie probe can fire AFTER a
361
- // fresh instance already resumed and started a turn, and Escape
362
- // would abort that turn without consent (the consent line). /reload
363
- // alone lands at a dead prompt and queues harmlessly mid-turn (pi
364
- // refuses the reload itself if a response is streaming).
365
- cmd = `tmux send-keys -t ${pane} -l '/reload' && tmux send-keys -t ${pane} Enter`;
366
- } else if (wezPane && /^\d+$/.test(wezPane)) {
367
- transport = "wezterm";
368
- pane = wezPane;
369
- // wezterm cli send-text --no-paste delivers bytes as keystrokes.
370
- // /reload + CR only — no Escape (see above). Literal CR byte inside
371
- // single quotes — no bash-isms (exec uses /bin/sh).
372
- cmd = `wezterm cli send-text --pane-id ${pane} --no-paste '/reload\r'`;
373
- } else {
374
- appendLedger(ctx.cwd, "auto_reload_skipped", { where, reason: "no supported multiplexer (tmux/WezTerm) in env" });
375
- return false;
376
- }
377
- appendLedger(ctx.cwd, "auto_reload_injected", { where, transport, pane });
378
- ctx.ui.notify("glla: injecting /reload into this pane — extensions rebuild in place and the fresh instance holds state. /glla resume after (automatic with autoresume=on).", "info");
379
- exec(cmd, () => { /* best-effort */ });
337
+ fs.mkdirSync(piGlaDir(ctx.cwd), { recursive: true });
338
+ fs.writeFileSync(sessionHandoffPath(ctx.cwd), JSON.stringify({ pid: process.pid, at: new Date().toISOString(), reason }));
339
+ appendLedger(ctx.cwd, "session_handoff_pending", { reason, pid: process.pid });
380
340
  return true;
381
341
  } catch {
382
- /* never let recovery take the warning path down */
342
+ appendLedger(ctx.cwd, "session_handoff_write_failed", { reason });
383
343
  return false;
384
344
  }
385
345
  }
386
-
387
- /** v0.34.13: one rung of the auto-recovery ladder. Returns true when a
388
- * /reload was injected (caller should NOT also pause/alert the manual
389
- * cure). Writes the sidecar resume marker FIRST so the fresh instance
390
- * resumes the goal/loop even with autoresume=off. Throttled — a second
391
- * wedge inside the window is the pi-restart class and must reach the
392
- * user as a loud stop, not another reload. */
393
- let lastAutoRecoveryAt = 0;
394
- function attemptAutoRecovery(ctx: ExtensionContext, where: string): boolean {
395
- if (loadSettings(ctx.cwd).autoRecovery === false) return false;
396
- const now = Date.now();
397
- if (now - lastAutoRecoveryAt < AUTO_RECOVERY_THROTTLE_MS) return false;
398
- // Inject FIRST — the throttle stamp + marker must reflect a real
399
- // recovery, not a no-transport skip (else the watchdog would mislabel
400
- // the next wedge as the pi-restart class).
401
- if (!attemptAutoReload(ctx, where)) return false;
402
- lastAutoRecoveryAt = now;
346
+ function consumeSessionHandoff(cwd: string): boolean {
403
347
  try {
404
- fs.writeFileSync(
405
- path.join(piGlaDir(ctx.cwd), RECOVERY_RESUME_MARKER),
406
- JSON.stringify({ at: new Date(now).toISOString(), where }),
407
- );
408
- } catch { /* best-effort — the reload still helps; restore just holds */ }
409
- appendLedger(ctx.cwd, "auto_recovery_reload", { where });
410
- ctx.ui.notify(`glla auto-recovery: ${where} — the ${isLoopActive() ? "loop" : "goal/list item"} resumes ITSELF after the rebuild (no /glla resume needed; the sidecar marker carries the consent). If this recurs within ${Math.round(AUTO_RECOVERY_THROTTLE_MS / 60_000)}m it's the pi-restart class and you'll get a loud stop.`, "warning");
411
- return true;
348
+ const p = sessionHandoffPath(cwd);
349
+ if (!fs.existsSync(p)) return false;
350
+ const raw = fs.readFileSync(p, "utf-8");
351
+ fs.unlinkSync(p);
352
+ const data = JSON.parse(raw) as { pid?: number; at?: string; reason?: string };
353
+ const at = Date.parse(data.at ?? "");
354
+ return data.pid === process.pid && data.reason?.trim().toLowerCase() !== "quit" && !Number.isNaN(at) && Date.now() - at < SESSION_HANDOFF_FRESH_MS;
355
+ } catch {
356
+ return false;
357
+ }
412
358
  }
413
359
 
414
360
  /** v0.34.14: /reload rebind detector. The extension runs INSIDE pi, so
415
361
  * process.pid IS pi's pid: an instance that boots and finds its OWN pid
416
- * already in the owner file is a /reload rebuild of a live session, not a
417
- * cold boot. Rebinds always resume active goals/loops — holding mid-work
362
+ * already in the owner file is normally a same-process rebuild, not a cold
363
+ * boot. A non-quit rebind resumes active goals/loops — holding mid-work
418
364
  * after an in-place rebuild is pure friction (user directive: keep going
419
365
  * unless we must stop; "the list is not continuing" after /reload,
420
- * hellhunter 2026-08-01). Cold boots (new pid) still honor autoresume=off.
421
- * Sidecar, not the ledger: read-before-write must be atomic-ish and the
422
- * ledger is append-only. */
366
+ * hellhunter 2026-08-01). An explicit quit is stamped in the sidecar and
367
+ * does not receive this implicit consent; cold boots (new pid) still honor
368
+ * autoresume=off. Sidecar, not the ledger: read-before-write must be
369
+ * atomic-ish and the ledger is append-only. */
423
370
  const SESSION_OWNER_FILE = "session-owner.json";
371
+ function markSessionOwnerShutdown(cwd: string, reason: string): void {
372
+ try {
373
+ const p = path.join(piGlaDir(cwd), SESSION_OWNER_FILE);
374
+ const owner = JSON.parse(fs.readFileSync(p, "utf-8")) as { pid?: number; at?: string };
375
+ if (owner.pid === process.pid) {
376
+ fs.writeFileSync(p, JSON.stringify({ ...owner, shutdownReason: reason, shutdownAt: new Date().toISOString() }));
377
+ }
378
+ } catch { /* advisory sidecar — lifecycle cleanup must not throw */ }
379
+ }
424
380
  function claimSessionOwnerAndDetectRebind(cwd: string): boolean {
425
381
  try {
426
382
  const p = path.join(piGlaDir(cwd), SESSION_OWNER_FILE);
427
- let prevPid: number | null = null;
383
+ let previous: { pid?: number; shutdownReason?: string } = {};
428
384
  try {
429
- prevPid = (JSON.parse(fs.readFileSync(p, "utf-8")) as { pid?: number }).pid ?? null;
385
+ previous = JSON.parse(fs.readFileSync(p, "utf-8")) as { pid?: number; shutdownReason?: string };
430
386
  } catch { /* absent or corrupt — first boot */ }
431
387
  fs.writeFileSync(p, JSON.stringify({ pid: process.pid, at: new Date().toISOString() }));
432
- return prevPid !== null && prevPid === process.pid;
388
+ const quit = previous.shutdownReason?.trim().toLowerCase() === "quit";
389
+ return previous.pid === process.pid && !quit;
433
390
  } catch {
434
391
  return false;
435
392
  }
@@ -486,6 +443,10 @@ function probeExtensionApiStale(): boolean {
486
443
  * promise from sync archiveCurrentGoal turned it into an uncaughtException
487
444
  * and pi EXITED mid-audit. Probe first, catch anyway, ledger the skip. */
488
445
  function safeSteerUser(ctx: ExtensionContext, text: string): boolean {
446
+ if (sessionHandoffPending) {
447
+ appendLedger(ctx.cwd, "steer_skipped_handoff", { chars: text.length });
448
+ return false;
449
+ }
489
450
  if (probeExtensionApiStale()) {
490
451
  appendLedger(ctx.cwd, "steer_skipped_stale", { chars: text.length });
491
452
  return false;
@@ -505,11 +466,15 @@ function safeSteerUser(ctx: ExtensionContext, text: string): boolean {
505
466
  * and must NOT claim work started (S3's "created — starting now" lie). */
506
467
  function warnIfStaleAtEntry(ctx: ExtensionContext, what: string): boolean {
507
468
  if (!probeExtensionApiStale()) return false;
508
- // v0.30.0: a successor may already own this session (e.g. /reload
509
- // re-imported the modules) — the user's command belongs to the fresh
510
- // instance; say so softly instead of demanding a reload.
469
+ if (sessionHandoffPending) {
470
+ ctx.ui.notify(`glla: this session is handing off to a fresh pi context — ${what} will be handled after session_start.`, "info");
471
+ return true;
472
+ }
473
+ // v0.30.0: a successor may already own this session (e.g. a module
474
+ // re-import) — the user's command belongs to the fresh instance; say so
475
+ // softly instead of claiming the old handle can recover it.
511
476
  // v0.32.0: the rebind window means a fresh instance is COMING, not here —
512
- // the old message claimed "handled there" while nothing owned the session.
477
+ // the message names that handoff rather than pretending a send landed.
513
478
  if (Date.now() < sessionReplacementUntil) {
514
479
  ctx.ui.notify(`glla: this session is rebinding after /reload — ${what} will be handled by the refreshed instance; retry in a moment if it doesn't.`, "info");
515
480
  return true;
@@ -520,14 +485,12 @@ function warnIfStaleAtEntry(ctx: ExtensionContext, what: string): boolean {
520
485
  }
521
486
  appendLedger(ctx.cwd, "extension_api_stale", { where: `entry probe (${what})` });
522
487
  ctx.ui.notify(
523
- `glla: this session's extension handle is stale (pi session replacement) — ${what} can't send continuations in this process. State is safe in .pi-glla/ — run /reload (extensions rebuild in place), then /glla resume. Restart pi only if /reload fails.`,
488
+ `glla: this session's extension handle is stale (pi session replacement) — ${what} can't send continuations in this process. State is safe in .pi-glla/. A fresh session_start will resume it; if pi does not create one, restart pi normally and restore the saved work.`,
524
489
  "warning",
525
490
  );
526
- // v0.29.22: deliberately NO self-heal from the entry probe — it fires
527
- // when the user is ACTIVELY typing a /glla command, and injected
528
- // keystrokes would race their input. User-present cases keep the manual
529
- // warning; the self-heal stays on the autonomous paths (heartbeat
530
- // probe, send paths) where no one is at the keyboard.
491
+ // Entry probes never mutate the terminal. The only recovery boundary is
492
+ // pi's own session lifecycle; user-present commands keep an honest warning
493
+ // and the durable state remains available to the fresh session.
531
494
  return true;
532
495
  }
533
496
 
@@ -616,6 +579,11 @@ function resolveCarryover(ctx: ExtensionContext, trigger: "goal" | "loop" | "lis
616
579
  // pi replaces sessions (newSession/fork/reload) and stale ctx throws on use,
617
580
  // so timers must never capture a ctx — they read lastCtx at fire time.
618
581
  let lastCtx: ExtensionContext | null = null;
582
+ // v0.34.16: shutdown sets this before pi invalidates the old context. Any
583
+ // timer or late event that reaches the old module must stand down until the
584
+ // fresh session_start rebinds it.
585
+ let sessionHandoffPending = false;
586
+ const sessionTimeouts = new Set<NodeJS.Timeout>();
619
587
  // v0.23.8: the session that OWNS the loop (its sessionManager). Subagent
620
588
  // sessions (pi-subagents binds extensions there too) fire our handlers
621
589
  // with their own ctx — they must never take over lastCtx (a headless
@@ -718,15 +686,9 @@ const CONTINUATION_UNANSWERED_THROTTLE_MS = 300_000;
718
686
  // cycle. Sending 2.5s AFTER agent_end lets teardown settle; the send lands
719
687
  // and the next turn starts immediately. 2.5s per turn beats 60s per turn.
720
688
  const EAGER_CONTINUATION_SETTLE_MS = Number(process.env.GLLA_EAGER_SETTLE_MS ?? 2_500);
721
- // v0.34.13: auto-recovery ladder ("keep going unless we MUST stop — a
722
- // question, or done" — user directive 2026-08-01). A wedge that only a
723
- // /reload cures should /reload ITSELF: inject the keystrokes (v0.29.13
724
- // transport) with a sidecar marker so the fresh instance RESUMES even when
725
- // autoresume=off (the consent came from autoRecovery at recovery time, not
726
- // the restore-time setting). Throttled to one attempt per window: a
727
- // recurrence inside the window is the transcript-writer-dead class, which
728
- // only a pi restart cures — that one stays a loud stop for the human.
729
- const AUTO_RECOVERY_THROTTLE_MS = 600_000;
689
+ // v0.34.16: retain the old recovery marker for one compatibility window so
690
+ // an in-flight v0.34.15 reload can still resume once. New recovery debt uses
691
+ // the session lifecycle handoff below and never injects terminal keystrokes.
730
692
  const RECOVERY_RESUME_MARKER = "recovery-resume.json";
731
693
  const RECOVERY_RESUME_FRESH_MS = 300_000;
732
694
  // v0.29.19: dead-turn caps (agent_end exemption path). 6 consecutive
@@ -904,14 +866,10 @@ function escalateSendRearmStorm(ctx: ExtensionContext, kind: "continuation" | "l
904
866
  const silent = Math.round(SEND_REARM_ESCALATE_SILENT_MS / 60000);
905
867
  appendLedger(ctx.cwd, "send_rearm_escalated", { kind, afterMinutes: mins, silentMinutes: silent });
906
868
  if (kind === "loop" && isLoopActive()) {
907
- if (attemptAutoRecovery(ctx, "send-retry storm")) {
908
- appendLedger(ctx.cwd, "send_rearm_escalated_suppressed", { reason: "auto-recovery reload" });
909
- return;
910
- }
911
869
  clearLoopTimer();
912
- state.loop = { ...state.loop!, active: false, stopReason: `send-retry storm: ${mins}m of re-arms with no session activity for ${silent}m — the session is wedged. Press Escape to cancel the stuck run (pi's own rate-limit retry holds it; pi prints "escape to cancel"), then /loop resume — the loop holds on restore. If still wedged: /reload rebuilds extensions in place, then /loop resume again. Restart pi only if /reload itself fails.` };
870
+ state.loop = { ...state.loop!, active: false, stopReason: `send-retry storm: ${mins}m of re-arms with no session activity for ${silent}m — the session is wedged. Press Escape to cancel the stuck run (pi's own rate-limit retry holds it; pi prints "escape to cancel"), then /loop resume — the loop holds on restore. A fresh session_start rebinds the loop; restart pi normally only if no replacement arrives.` };
913
871
  persistState(ctx);
914
- ctx.ui.notify(`Loop stopped: send-retry storm (${mins}m, session silent ${silent}m). Escape cancels the stuck run, then /loop resume (the loop holds on restore). /reload if it persists; restart pi only if /reload fails.`, "warning");
872
+ ctx.ui.notify(`Loop stopped: send-retry storm (${mins}m, session silent ${silent}m). Escape cancels the stuck run, then /loop resume (the loop holds on restore). A fresh session_start rebinds it; restart pi normally only if no replacement arrives.`, "warning");
915
873
  notifyExternal(ctx, "Loop stopped: send-retry storm.");
916
874
  return;
917
875
  }
@@ -931,21 +889,13 @@ function escalateSendRearmStorm(ctx: ExtensionContext, kind: "continuation" | "l
931
889
  return;
932
890
  }
933
891
  if (state.goal && state.goal.status === "active") {
934
- // v0.34.13: keep going unless we MUST stop — try the auto-recovery
935
- // /reload before spending the user's attention on a pause. A reload
936
- // that fails to cure throttles the next attempt, and the pause below
937
- // fires then as today.
938
- if (attemptAutoRecovery(ctx, "send-retry storm")) {
939
- appendLedger(ctx.cwd, "send_rearm_escalated_suppressed", { reason: "auto-recovery reload" });
940
- return;
941
- }
942
892
  updateGoal({
943
893
  status: "paused",
944
894
  pauseKind: "error",
945
895
  pauseReason: `send-retry storm: ${mins}m of re-arms with no session activity for ${silent}m — the session never went idle for the continuation`,
946
- pauseSuggestedAction: "The session produced no events while the send retried (wedged queue — often pi's own rate-limit retry holding the run; pi prints 'escape to cancel'). Press Escape, then /goal resume. If still wedged: /reload rebuilds extensions in place, then /goal resume again. Restart pi only if /reload fails.",
896
+ pauseSuggestedAction: "The session produced no events while the send retried (wedged queue — often pi's own rate-limit retry holding the run; pi prints 'escape to cancel'). Press Escape, then /goal resume. A fresh session_start rebinds the goal; restart pi normally only if no replacement arrives.",
947
897
  }, ctx);
948
- ctx.ui.notify(`${goalNoun()} paused: send-retry storm (${mins}m, session silent ${silent}m). Escape cancels the stuck run, then /goal resume. /reload if it persists; restart pi only if /reload fails.`, "warning");
898
+ ctx.ui.notify(`${goalNoun()} paused: send-retry storm (${mins}m, session silent ${silent}m). Escape cancels the stuck run, then /goal resume. A fresh session_start rebinds it; restart pi normally only if no replacement arrives.`, "warning");
949
899
  notifyExternal(ctx, `${goalNoun()} paused: send-retry storm.`);
950
900
  }
951
901
  }
@@ -954,17 +904,11 @@ function escalateStallNow(ctx: ExtensionContext, threshold: number): boolean {
954
904
  if (!shouldEscalateStall(consecutiveStalls, threshold)) return false;
955
905
  consecutiveStalls = 0;
956
906
  appendLedger(ctx.cwd, "stall_escalated", { threshold, kind: isLoopActive() ? "loop" : "goal" });
957
- // v0.34.13: recovery before stop — usually throttled (the 2.5min
958
- // watchdog already tried), in which case the stop proceeds as today.
959
- if (attemptAutoRecovery(ctx, "stall escalation")) {
960
- appendLedger(ctx.cwd, "stall_escalated_suppressed", { reason: "auto-recovery reload", threshold });
961
- return true;
962
- }
963
907
  if (isLoopActive()) {
964
908
  clearLoopTimer();
965
- state.loop = { ...state.loop!, active: false, stopReason: `stalled: ${threshold} continuation refires landed no turn — the session is not continuing (wedged message queue or stale API). Press Escape to cancel any stuck run, then /loop resume — the loop holds on restore. If the handle is stale, /reload rebuilds extensions in place; restart pi only if /reload fails.` };
909
+ state.loop = { ...state.loop!, active: false, stopReason: `stalled: ${threshold} continuation refires landed no turn — the session is not continuing (wedged message queue or stale API). Press Escape to cancel any stuck run, then /loop resume — the loop holds on restore. A fresh session_start rebinds the loop or goal; restart pi normally only if no replacement arrives.` };
966
910
  persistState(ctx);
967
- ctx.ui.notify(`Loop stopped: ${threshold} refires produced no turn — the continuation is not landing. Escape cancels a stuck run, then /loop resume (the loop holds on restore). /reload if stale; restart pi only if /reload fails.`, "warning");
911
+ ctx.ui.notify(`Loop stopped: ${threshold} refires produced no turn — the continuation is not landing. Escape cancels a stuck run, then /loop resume (the loop holds on restore). A fresh session_start rebinds it; restart pi normally only if no replacement arrives.`, "warning");
968
912
  notifyExternal(ctx, "Loop stopped: stalled (continuation not landing).");
969
913
  return true;
970
914
  }
@@ -973,9 +917,9 @@ function escalateStallNow(ctx: ExtensionContext, threshold: number): boolean {
973
917
  status: "paused",
974
918
  pauseKind: "error",
975
919
  pauseReason: `stalled: ${threshold} continuation refires landed no turn`,
976
- pauseSuggestedAction: "The continuation chain is broken in this process (wedged message queue or stale API). Press Escape to cancel any stuck run, then /goal resume. If the handle is stale, /reload rebuilds extensions in place — restart pi only if /reload fails.",
920
+ pauseSuggestedAction: "The continuation chain is broken in this process (wedged message queue or stale API). Press Escape to cancel any stuck run, then /goal resume. A fresh session_start rebinds the goal; restart pi normally only if no replacement arrives.",
977
921
  }, ctx);
978
- ctx.ui.notify(`${goalNoun()} paused: ${threshold} refires produced no turn. Escape cancels a stuck run, then /goal resume. /reload if stale; restart pi only if /reload fails.`, "warning");
922
+ ctx.ui.notify(`${goalNoun()} paused: ${threshold} refires produced no turn. Escape cancels a stuck run, then /goal resume. A fresh session_start rebinds it; restart pi normally only if no replacement arrives.`, "warning");
979
923
  notifyExternal(ctx, `${goalNoun()} paused: stalled (continuation not landing).`);
980
924
  return true;
981
925
  }
@@ -1069,17 +1013,10 @@ function heartbeatTick(): void {
1069
1013
  ) {
1070
1014
  lastUnansweredAlertAt = Date.now();
1071
1015
  appendLedger(ctx.cwd, "continuation_unanswered", { silentMs: Date.now() - lastContinuationSentAt });
1072
- // v0.34.13: recover, don't just report. Only the failure modes reach
1073
- // the user: throttled (a reload already failed to cure → the
1074
- // pi-restart class) or recovery unavailable (setting off / no pane).
1075
- if (!attemptAutoRecovery(ctx, "continuation unanswered")) {
1076
- const mins = Math.round((Date.now() - lastContinuationSentAt) / 60_000);
1077
- const msg = lastAutoRecoveryAt > 0
1078
- ? `glla: auto-recovery /reload did NOT unstick this session — still no turn ${mins}m after the continuation. This class kills pi's transcript writer; only a pi RESTART cures it: restart pi in this tab, then /glla resume (the goal holds in .pi-glla state).`
1079
- : `glla: pi accepted the continuation ${mins}m ago but NO turn has started — no tool calls, no tokens, transcript frozen (the turn trigger is wedged). Cure: /reload — autoresume re-fires the ${isLoopActive() ? "loop" : "goal/list item"}. (auto-recovery is off or no tmux/WezTerm pane — /glla settings autoRecovery=on enables the automatic form.)`;
1080
- ctx.ui.notify(msg, "warning");
1081
- notifyExternal(ctx, msg);
1082
- }
1016
+ const mins = Math.round((Date.now() - lastContinuationSentAt) / 60_000);
1017
+ const msg = `glla: pi accepted the continuation ${mins}m ago but NO turn has started — no tool calls, no tokens, transcript frozen (the turn trigger is wedged). Re-sends do not unstick it. A fresh session_start will rebind the ${isLoopActive() ? "loop" : "goal/list item"}; if no replacement arrives, restart pi normally and restore the saved work.`;
1018
+ ctx.ui.notify(msg, "warning");
1019
+ notifyExternal(ctx, msg);
1083
1020
  }
1084
1021
  // v0.29.1: stranded-audit recovery. A goal left in "auditing" with NO
1085
1022
  // in-flight audit means the auditor's result never landed (wedged queue
@@ -1241,11 +1178,37 @@ function clearContinuationTimer(): void {
1241
1178
  continuationScheduledFor = null;
1242
1179
  }
1243
1180
 
1181
+ function scheduleSessionTimeout(callback: () => void, delayMs: number): NodeJS.Timeout {
1182
+ let timer: NodeJS.Timeout;
1183
+ timer = setTimeout(() => {
1184
+ sessionTimeouts.delete(timer);
1185
+ callback();
1186
+ }, delayMs);
1187
+ sessionTimeouts.add(timer);
1188
+ timer.unref?.();
1189
+ return timer;
1190
+ }
1191
+
1192
+ function clearSessionOwnedTimers(): void {
1193
+ sessionHandoffPending = true;
1194
+ clearContinuationTimer();
1195
+ clearLoopTimer();
1196
+ if (queueStuckProbe) { clearTimeout(queueStuckProbe); queueStuckProbe = null; }
1197
+ if (heartbeatTimer) { clearInterval(heartbeatTimer); heartbeatTimer = null; }
1198
+ if (uiTicker) { clearInterval(uiTicker); uiTicker = null; }
1199
+ for (const timer of sessionTimeouts) clearTimeout(timer);
1200
+ sessionTimeouts.clear();
1201
+ cancelQuotaRetry();
1202
+ lastCtx = null;
1203
+ ownerSession = null;
1204
+ }
1205
+
1244
1206
  function isActionableGoal(): boolean {
1245
1207
  return !!state.goal && state.goal.status === "active" && state.goal.autoContinue;
1246
1208
  }
1247
1209
 
1248
1210
  function freshCtx(): ExtensionContext | null {
1211
+ if (sessionHandoffPending) return null;
1249
1212
  // A captured ctx throws "stale" after session replacement. Probe cheaply;
1250
1213
  // on stale, drop it and wait for the next event to hand us a fresh one.
1251
1214
  if (!lastCtx) return null;
@@ -1263,35 +1226,34 @@ function freshCtx(): ExtensionContext | null {
1263
1226
  // pi idled. The 0.34.11 watchdog gates on "pi reported NO pending" and the
1264
1227
  // stall ladder takes ~10 minutes; a send that lands queued-without-a-turn is
1265
1228
  // a CONFIRMED dead trigger (hegemon law), so probe once, ~45s after every
1266
- // landed send, and go straight to auto-recovery. A consumed message (even an
1267
- // instant-429 turn consumes it) or any real activity disarms the probe.
1229
+ // landed send, and report it without terminal input. A consumed message (even
1230
+ // an instant-429 turn consumes it) or any real activity disarms the probe.
1268
1231
  function queueStuckProbeMs(): number {
1269
1232
  return Number(process.env.GLLA_QUEUE_STUCK_MS ?? 45_000);
1270
1233
  }
1271
1234
  let queueStuckProbe: ReturnType<typeof setTimeout> | null = null;
1272
- function armQueueStuckProbe(ctx: ExtensionContext, sentAt: number): void {
1235
+ function armQueueStuckProbe(sentAt: number): void {
1273
1236
  if (queueStuckProbe) clearTimeout(queueStuckProbe);
1274
- queueStuckProbe = setTimeout(() => {
1237
+ queueStuckProbe = scheduleSessionTimeout(() => {
1275
1238
  queueStuckProbe = null;
1276
1239
  try {
1277
- if (isForeignCtx(ctx)) return; // stale instance — the live one probes
1240
+ const ctx = freshCtx();
1241
+ if (!ctx) return; // no fresh lifecycle context — do not touch a stale one
1278
1242
  if (!isSupervising()) return; // paused/completed meanwhile
1279
1243
  if (lastContinuationSentAt !== sentAt) return; // a newer send armed its own probe
1280
1244
  if (lastRealActivityAt > sentAt) return; // the turn started and worked
1281
1245
  if (!ctx.isIdle()) return; // a turn is running — healthy
1282
1246
  if (!ctx.hasPendingMessages()) return; // consumed — even an instant 429 consumes
1283
1247
  appendLedger(ctx.cwd, "queue_stuck_detected", { waitedMs: Date.now() - sentAt });
1284
- if (!attemptAutoRecovery(ctx, "queue-stuck continuation")) {
1285
- const msg = `${goalNoun()}: the continuation is QUEUED but pi won't start a turn — the turn trigger is dead (re-sends only queue). Cure: /reload — the goal resumes itself after the rebuild (autoresume off? /glla resume).`;
1286
- ctx.ui.notify(msg, "warning");
1287
- notifyExternal(ctx, msg);
1288
- }
1248
+ const msg = `${goalNoun()}: the continuation is QUEUED but pi won't start a turn — the turn trigger is dead (re-sends only queue). glla will resume from a fresh session_start; if no replacement arrives, restart pi normally and restore the saved work.`;
1249
+ ctx.ui.notify(msg, "warning");
1250
+ notifyExternal(ctx, msg);
1289
1251
  } catch { /* stale ctx — the live instance owns the probe now */ }
1290
1252
  }, queueStuckProbeMs());
1291
- queueStuckProbe.unref?.();
1292
1253
  }
1293
1254
 
1294
1255
  function scheduleContinuation(ctx: ExtensionContext, force = false, delayMs?: number): void {
1256
+ if (sessionHandoffPending) return;
1295
1257
  abortedStandDown = false; // v0.29.5: any explicit schedule ends the stand-down
1296
1258
  if (!isActionableGoal()) return;
1297
1259
  rememberCtx(ctx);
@@ -1305,11 +1267,11 @@ function scheduleContinuation(ctx: ExtensionContext, force = false, delayMs?: nu
1305
1267
  return;
1306
1268
  }
1307
1269
  continuationScheduledFor = goalId;
1308
- continuationTimer = setTimeout(() => sendContinuation(goalId), delay);
1309
- continuationTimer.unref?.();
1270
+ continuationTimer = scheduleSessionTimeout(() => sendContinuation(goalId), delay);
1310
1271
  }
1311
1272
 
1312
1273
  function sendContinuation(goalId: string): void {
1274
+ if (sessionHandoffPending) return;
1313
1275
  continuationTimer = null;
1314
1276
  continuationScheduledFor = null;
1315
1277
  if (!isActionableGoal()) return;
@@ -1320,16 +1282,14 @@ function sendContinuation(goalId: string): void {
1320
1282
  if (probeExtensionApiStale()) return;
1321
1283
  // No live ctx — retry shortly; the next session event will refresh it.
1322
1284
  continuationScheduledFor = goalId;
1323
- continuationTimer = setTimeout(() => sendContinuation(goalId), BACKOFF_IDLE_RETRY_MS);
1324
- continuationTimer.unref?.();
1285
+ continuationTimer = scheduleSessionTimeout(() => sendContinuation(goalId), BACKOFF_IDLE_RETRY_MS);
1325
1286
  return;
1326
1287
  }
1327
1288
  if (!ctx.isIdle() || ctx.hasPendingMessages()) {
1328
1289
  accountSendRearm(ctx, "continuation");
1329
1290
  continuationScheduledFor = goalId;
1330
1291
  // v0.28.29: backing-off cadence (was flat 50ms — 6,000 spins in 5m).
1331
- continuationTimer = setTimeout(() => sendContinuation(goalId), sendRearmDelayMs(continuationRearmStreak));
1332
- continuationTimer.unref?.();
1292
+ continuationTimer = scheduleSessionTimeout(() => sendContinuation(goalId), sendRearmDelayMs(continuationRearmStreak));
1333
1293
  return;
1334
1294
  }
1335
1295
  if (!extensionApi || extensionApiStale) return;
@@ -1347,7 +1307,7 @@ function sendContinuation(goalId: string): void {
1347
1307
  continuationRearmStreak = 0; continuationRearmSince = 0; // v0.28.5 (E3): a landed send clears the storm
1348
1308
  appendLedger(ctx.cwd, "goal_continuation_sent", { goalId });
1349
1309
  lastContinuationSentAt = Date.now();
1350
- armQueueStuckProbe(ctx, lastContinuationSentAt);
1310
+ armQueueStuckProbe(lastContinuationSentAt);
1351
1311
  } catch (err) {
1352
1312
  appendLedger(ctx.cwd, "goal_continuation_send_failed", { goalId, error: err instanceof Error ? err.message : String(err) });
1353
1313
  // v0.26.7: stale runtime = terminal (sends can never land); anything
@@ -1361,7 +1321,7 @@ function sendContinuation(goalId: string): void {
1361
1321
  // what closes the turn: complete_goal if done, pause_goal if blocked, a tool
1362
1322
  // call otherwise. display: true — the user should see the warning too.
1363
1323
  function sendStallEscalation(ctx: ExtensionContext, nudges: number): void {
1364
- if (!extensionApi || extensionApiStale) return;
1324
+ if (sessionHandoffPending || !extensionApi || extensionApiStale) return;
1365
1325
  const remaining = HEARTBEAT_MAX_NUDGES - nudges;
1366
1326
  const text = [
1367
1327
  `[STALL WARNING ${nudges}/${HEARTBEAT_MAX_NUDGES}] The last turn produced no tool calls.`,
@@ -1383,7 +1343,7 @@ function sendStallEscalation(ctx: ExtensionContext, nudges: number): void {
1383
1343
  // sendContinuation (stale api = terminal), independent of goal state —
1384
1344
  // plain sessions truncate too.
1385
1345
  function sendLengthContinue(ctx: ExtensionContext, consecutive: number): void {
1386
- if (!extensionApi || extensionApiStale) return;
1346
+ if (sessionHandoffPending || !extensionApi || extensionApiStale) return;
1387
1347
  try {
1388
1348
  extensionApi.sendMessage({
1389
1349
  customType: GOAL_EVENT_ENTRY,
@@ -1914,7 +1874,7 @@ function fireReviewer(
1914
1874
  // to still count as "proposed" in the report + notify. Now the
1915
1875
  // failure is LOUD and the proposal goes uncounted.
1916
1876
  ctx.ui.notify(
1917
- `Postaudit /goal proposal NOT delivered: ${err instanceof Error ? err.message : String(err)} — the follow-up never reached the session. Run /reload if the session was just replaced (extensions rebuild in place), then retry.`,
1877
+ `Postaudit /goal proposal NOT delivered: ${err instanceof Error ? err.message : String(err)} — the follow-up never reached the session. Wait for a fresh session_start, then retry.`,
1918
1878
  "warning",
1919
1879
  );
1920
1880
  return false;
@@ -2050,7 +2010,7 @@ async function startDrafting(ctx: ExtensionContext, target: "goal" | "list" | "l
2050
2010
  if (isStaleApiError(err)) {
2051
2011
  extensionApiStale = true;
2052
2012
  appendLedger(ctx.cwd, "extension_api_stale", { where: "startDrafting seed" });
2053
- ctx.ui.notify("glla: can't start the drafting interview — this session's extension handle is stale (pi session replacement). Run /reload (extensions rebuild in place), then re-run the command. Restart pi only if /reload fails.", "warning");
2013
+ ctx.ui.notify("glla: can't start the drafting interview — this session's extension handle is stale (pi session replacement). A fresh session_start will rebind it; if no replacement arrives, restart pi normally, then re-run the command.", "warning");
2054
2014
  } else {
2055
2015
  ctx.ui.notify(`glla: couldn't start the drafting interview (${err instanceof Error ? err.message : String(err)}) — try again.`, "warning");
2056
2016
  }
@@ -2179,7 +2139,7 @@ async function cmdSet(args: string, ctx: ExtensionContext, skipDraft = false): P
2179
2139
  // v0.28.1 (S3): the goal is persisted — mark the interrupt so the next
2180
2140
  // fresh session LOADS it (held by default since v0.28.21), and tell the truth instead of "starting now".
2181
2141
  updateGoal({ interruptedAt: nowIso(), interruptedReason: "created in a stale session" }, ctx);
2182
- ctx.ui.notify(`Goal saved: ${shortObj(goal.objective)} — safe in .pi-glla/, but this stale process can't send continuations. Run /reload (extensions rebuild in place, state survives), then /goal resume. Restart pi only if /reload itself fails.`, "warning");
2142
+ ctx.ui.notify(`Goal saved: ${shortObj(goal.objective)} — safe in .pi-glla/, but this stale process can't send continuations. A fresh session_start will resume it; if no replacement arrives, restart pi normally, then /goal resume if autoresume is off.`, "warning");
2183
2143
  return;
2184
2144
  }
2185
2145
  ctx.ui.notify(`Goal started: ${shortObj(goal.objective)} — the auditor will verify on completion.`, "info");
@@ -2359,8 +2319,11 @@ async function showDecisionPrompt(ctx: ExtensionContext): Promise<boolean> {
2359
2319
  * disabled (/glla decisionpopup=off), or when one is already open. */
2360
2320
  function maybeDecisionPopup(ctx: ExtensionContext): void {
2361
2321
  if (!ctx.hasUI || loadSettings(ctx.cwd).decisionPopup === false) return;
2362
- setTimeout(() => {
2363
- void showDecisionPrompt(ctx).catch(() => {});
2322
+ const cwd = ctx.cwd;
2323
+ scheduleSessionTimeout(() => {
2324
+ const fresh = freshCtx();
2325
+ if (!fresh || fresh.cwd !== cwd) return;
2326
+ void showDecisionPrompt(fresh).catch(() => {});
2364
2327
  }, 600);
2365
2328
  }
2366
2329
 
@@ -2907,7 +2870,7 @@ function loopPrompt(loop: LoopState, regressionNote: string, strategyNote: strin
2907
2870
  }
2908
2871
 
2909
2872
  function scheduleLoopTick(ctx: ExtensionContext): void {
2910
- if (!isLoopActive()) return;
2873
+ if (sessionHandoffPending || !isLoopActive()) return;
2911
2874
  rememberCtx(ctx);
2912
2875
  clearLoopTimer();
2913
2876
  let delay = 0;
@@ -2916,11 +2879,11 @@ function scheduleLoopTick(ctx: ExtensionContext): void {
2916
2879
  } catch {
2917
2880
  return;
2918
2881
  }
2919
- loopTimer = setTimeout(() => sendLoopTurn(), delay);
2920
- loopTimer.unref?.();
2882
+ loopTimer = scheduleSessionTimeout(() => sendLoopTurn(), delay);
2921
2883
  }
2922
2884
 
2923
2885
  function sendLoopTurn(): void {
2886
+ if (sessionHandoffPending) return;
2924
2887
  loopTimer = null;
2925
2888
  if (!isLoopActive() || !extensionApi) return;
2926
2889
  const ctx = freshCtx();
@@ -2932,8 +2895,7 @@ function sendLoopTurn(): void {
2932
2895
  if (probeExtensionApiStale()) return;
2933
2896
  loopRearmStreak++;
2934
2897
  } else accountSendRearm(ctx, "loop");
2935
- loopTimer = setTimeout(() => sendLoopTurn(), sendRearmDelayMs(loopRearmStreak)); // v0.28.29: backing-off cadence
2936
- loopTimer.unref?.();
2898
+ loopTimer = scheduleSessionTimeout(() => sendLoopTurn(), sendRearmDelayMs(loopRearmStreak)); // v0.28.29: backing-off cadence
2937
2899
  return;
2938
2900
  }
2939
2901
  const loop = state.loop!;
@@ -3030,7 +2992,7 @@ function sendLoopTurn(): void {
3030
2992
  loopRearmStreak = 0; loopRearmSince = 0; // v0.28.5 (E3): a landed turn clears the storm
3031
2993
  appendLedger(ctx.cwd, "loop_turn_sent", { iteration: loop.iteration });
3032
2994
  lastContinuationSentAt = Date.now();
3033
- armQueueStuckProbe(ctx, lastContinuationSentAt);
2995
+ armQueueStuckProbe(lastContinuationSentAt);
3034
2996
  } catch (err) {
3035
2997
  // stale API — next agent_end reschedules (but if none comes, the
3036
2998
  // heartbeat's stall escalation stops the spin — v0.26.1).
@@ -4246,7 +4208,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4246
4208
  // refused; the dialog simply can't render in a doomed process.
4247
4209
  extensionApiStale = true;
4248
4210
  appendLedger(liveCtx.cwd, "extension_api_stale", { where: "batch confirm" });
4249
- return { content: [{ type: "text", text: "The Confirm dialog could not render: pi invalidated this session's extension handle (session replacement). This is NOT a rejection — do NOT refine or re-propose. Tell the user to restart pi, then re-run the drafting flow." }], details: {} };
4211
+ return { content: [{ type: "text", text: "The Confirm dialog could not render: pi invalidated this session's extension handle (session replacement). This is NOT a rejection — do NOT refine or re-propose. Wait for a fresh session_start, then re-run the drafting flow." }], details: {} };
4250
4212
  }
4251
4213
  batchConfirmed = c === "yes";
4252
4214
  }
@@ -4290,7 +4252,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4290
4252
  // v0.28.1 (T1): a stale dialog is NOT "Draft rejected by the user".
4291
4253
  extensionApiStale = true;
4292
4254
  appendLedger(liveCtx.cwd, "extension_api_stale", { where: "draft confirm" });
4293
- return { content: [{ type: "text", text: "The Confirm dialog could not render: pi invalidated this session's extension handle (session replacement). This is NOT a rejection — do NOT refine or re-propose. Tell the user to restart pi, then re-run the drafting flow." }], details: {} };
4255
+ return { content: [{ type: "text", text: "The Confirm dialog could not render: pi invalidated this session's extension handle (session replacement). This is NOT a rejection — do NOT refine or re-propose. Wait for a fresh session_start, then re-run the drafting flow." }], details: {} };
4294
4256
  }
4295
4257
  confirmed = c === "yes";
4296
4258
  }
@@ -6240,7 +6202,7 @@ export default function (pi: ExtensionAPI): void {
6240
6202
  // EVERY post-grace tick until agent_start discharges it).
6241
6203
  postCompactResumeOwed = true;
6242
6204
  postCompactResyncPending = true;
6243
- const settle = setTimeout(() => {
6205
+ scheduleSessionTimeout(() => {
6244
6206
  const c = freshCtx();
6245
6207
  if (!c) return;
6246
6208
  try {
@@ -6253,7 +6215,6 @@ export default function (pi: ExtensionAPI): void {
6253
6215
  /* settle race — the 60s heartbeat covers it */
6254
6216
  }
6255
6217
  }, 2000);
6256
- settle.unref?.();
6257
6218
  // v0.29.21: a SECOND settle at grace expiry. The 2s settle almost
6258
6219
  // always loses (pi is mid-compact / mid-resumed-turn then), and the
6259
6220
  // heartbeat's first post-grace tick lands up to one interval late —
@@ -6262,7 +6223,7 @@ export default function (pi: ExtensionAPI): void {
6262
6223
  // 195.8k after two output-limit turns, zero rearms after the compact
6263
6224
  // event, recovery only at 04:34:48 via the post-grace heartbeat;
6264
6225
  // ~4 min that read as a stoppage). Refire the moment the grace ends.
6265
- const graceSettle = setTimeout(() => {
6226
+ scheduleSessionTimeout(() => {
6266
6227
  const c = freshCtx();
6267
6228
  if (!c) return;
6268
6229
  try {
@@ -6275,7 +6236,6 @@ export default function (pi: ExtensionAPI): void {
6275
6236
  /* settle race — the 60s heartbeat covers it */
6276
6237
  }
6277
6238
  }, COMPACTION_GRACE_MS + 2_000);
6278
- graceSettle.unref?.();
6279
6239
  });
6280
6240
 
6281
6241
  pi.on("message_start", async (event: any, _ctx: ExtensionContext) => {
@@ -6350,15 +6310,24 @@ export default function (pi: ExtensionAPI): void {
6350
6310
  // tells the stale probe that a rebind (session_start) is imminent.
6351
6311
  const shutdownReason = typeof event?.reason === "string" ? event.reason : "unknown";
6352
6312
  appendLedger(ctx.cwd, "session_shutdown", { reason: shutdownReason });
6313
+ markSessionOwnerShutdown(ctx.cwd, shutdownReason);
6314
+ writeSessionHandoff(ctx, shutdownReason);
6353
6315
  sessionReplacementUntil = Date.now() + SESSION_REBIND_GRACE_MS;
6316
+ clearSessionOwnedTimers();
6317
+ registeredCtx = null;
6318
+ toolHealNotified = false;
6354
6319
  });
6355
6320
 
6356
6321
  pi.on("session_start", async (event: any, ctx: ExtensionContext) => {
6357
- rememberCtx(ctx);
6358
6322
  // v0.23.8: subagent sessions (pi-subagents binds extensions there too)
6359
6323
  // are workers — never run the restore gate or reschedule the loop from
6360
6324
  // a foreign session.
6361
6325
  if (isForeignCtx(ctx)) return;
6326
+ extensionApi = pi;
6327
+ sessionHandoffPending = false;
6328
+ rememberCtx(ctx);
6329
+ startHeartbeat();
6330
+ startUITicker();
6362
6331
  // v0.30.0: rebind bookkeeping — claim ownership, close any replacement
6363
6332
  // window, and reset a stale flag left over from the PREVIOUS session's
6364
6333
  // invalidation. pi rebinds THIS module to the new session (switch) or
@@ -6378,10 +6347,12 @@ export default function (pi: ExtensionAPI): void {
6378
6347
  const stillStale = probeExtensionApiStale();
6379
6348
  appendLedger(ctx.cwd, "stale_flag_reset_on_rebind", { reason: startReason, stillStale });
6380
6349
  if (stillStale) {
6381
- ctx.ui.notify("glla: session rebound but the extension handle is still stale — run /reload (extensions rebuild in place), then /glla resume.", "warning");
6350
+ ctx.ui.notify("glla: session rebound but the extension handle is still stale — waiting for another fresh session_start; restart pi normally only if no replacement arrives, then /glla resume.", "warning");
6382
6351
  }
6383
6352
  }
6384
6353
  state = readState(ctx.cwd);
6354
+ const handoffResume = consumeSessionHandoff(ctx.cwd);
6355
+ if (handoffResume) appendLedger(ctx.cwd, "session_handoff_resumed", { pid: process.pid, reason: startReason });
6385
6356
  // v0.28.14: snapshot carryover BEFORE any restore logic mutates state —
6386
6357
  // a paused goal, waiting list items, or a loop that was live/held when
6387
6358
  // the last session ended. Resolved once at the first NEW activation.
@@ -6479,16 +6450,17 @@ export default function (pi: ExtensionAPI): void {
6479
6450
  persistState(ctx);
6480
6451
  appendLedger(ctx.cwd, "audit_loop_target_migrated", { from: "audit-every-iteration", to: "fix-first" });
6481
6452
  }
6482
- // v0.34.13: an auto-recovery /reload carries its own resume consent —
6483
- // the sidecar marker overrides autoresume=off for THIS restore only.
6453
+ // v0.34.15 compatibility: consume one legacy recovery marker if an older
6454
+ // process wrote it before this lifecycle-first build landed.
6484
6455
  const recoveryResume = consumeRecoveryResume(ctx.cwd);
6485
- // v0.34.14: a /reload rebind (same pi pid) ALWAYS resumes — the session
6486
- // is live mid-work; holding is the "list is not continuing" bug.
6456
+ // v0.34.16: a same-process lifecycle handoff is explicit continuation
6457
+ // debt, so it resumes independently of the cold-boot autoResume setting.
6458
+ // A same-pid owner rebind is the second same-process signal.
6487
6459
  const rebindResume = claimSessionOwnerAndDetectRebind(ctx.cwd);
6488
6460
  if (rebindResume) appendLedger(ctx.cwd, "rebind_resume", { pid: process.pid });
6489
6461
  if (isLoopActive()) {
6490
6462
  const l = state.loop!;
6491
- if (autoResume || recoveryResume || rebindResume) {
6463
+ if (autoResume || recoveryResume || rebindResume || handoffResume) {
6492
6464
  ctx.ui.notify(
6493
6465
  `Resuming loop (iteration ${l.iteration}/${l.maxIterations > 0 ? l.maxIterations : "∞"}, best ${l.bestValue ?? "n/a"}, stall ${l.stallCount}/${l.plateauWindow}): ${l.target.slice(0, 60)}`,
6494
6466
  "info",
@@ -6509,7 +6481,7 @@ export default function (pi: ExtensionAPI): void {
6509
6481
  // "load it but not auto start it"). Interrupted goals hold like
6510
6482
  // everything else; autoresume=on (unattended rigs) still auto-resumes
6511
6483
  // them, and the marker is cleared only on that promised auto-resume.
6512
- if (autoResume || recoveryResume || rebindResume) {
6484
+ if (autoResume || recoveryResume || rebindResume || handoffResume) {
6513
6485
  // v0.28.1 (S2): clear the stale-handle interrupt marker — this IS
6514
6486
  // the auto-resume the marker promised.
6515
6487
  if (wasInterrupted) updateGoal({ interruptedAt: undefined, interruptedReason: undefined }, ctx);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.34.15",
3
+ "version": "0.34.16",
4
4
  "description": "Mission control for autonomous pi: interview-drafted goals, an audited task queue, and forever-loops (metric, spec, project-audit) that run for hours. An isolated extension-less auditor re-verifies every completion with raw evidence; confirmed drafts, decision pauses and consent gates keep you in charge.",
5
5
  "license": "MIT",
6
6
  "author": "dracon",