pi-goal-list-loop-audit 0.34.12 → 0.34.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -73,6 +73,14 @@ export interface Settings {
73
73
  * assertActive). Default true; only acts when TMUX and TMUX_PANE are
74
74
  * set — pi outside tmux just gets the manual warning. */
75
75
  autoReloadOnStale?: boolean;
76
+ /** v0.34.13: auto-recovery ladder — when a wedge is detected that only a
77
+ * /reload cures (unanswered continuation, send-retry storm), inject the
78
+ * /reload ITSELF via the v0.29.13 tmux/WezTerm transport and resume the
79
+ * goal/loop after the rebuild (sidecar marker — autoresume=off is a
80
+ * restore-time setting, not a recovery veto). Default true. The one
81
+ * class it does NOT cross: pi restart (transcript-writer dead) — that
82
+ * stays a loud stop for the human. */
83
+ autoRecovery?: boolean;
76
84
  /** v0.26.1: consecutive heartbeat refires without a real turn before
77
85
  * the goal pauses / loop stops (default 5; 0 = never escalate). */
78
86
  stallEscalationRefires?: number;
@@ -207,6 +215,7 @@ export const SETTINGS_KEYS: Array<keyof Settings> = [
207
215
  "stuckMaxInterventions",
208
216
  "stallEscalationRefires",
209
217
  "autoReloadOnStale",
218
+ "autoRecovery",
210
219
  "stallShortWords",
211
220
  "stallSimilarityThreshold",
212
221
  "postaudit",
@@ -346,9 +346,9 @@ function goStaleTerminal(ctx: ExtensionContext, where: string): void {
346
346
  * fires from the entry-probe path (the most common stale discovery).
347
347
  * Opt out: autoReloadOnStale=false. Best-effort: the manual warning
348
348
  * already fired, so failures cost nothing. */
349
- function attemptAutoReload(ctx: ExtensionContext, where: string): void {
349
+ function attemptAutoReload(ctx: ExtensionContext, where: string): boolean {
350
350
  try {
351
- if (loadSettings(ctx.cwd).autoReloadOnStale === false) return;
351
+ if (loadSettings(ctx.cwd).autoReloadOnStale === false) return false;
352
352
  const tmuxPane = process.env.TMUX_PANE;
353
353
  const wezPane = process.env.WEZTERM_PANE;
354
354
  let transport: "tmux" | "wezterm";
@@ -372,13 +372,82 @@ function attemptAutoReload(ctx: ExtensionContext, where: string): void {
372
372
  cmd = `wezterm cli send-text --pane-id ${pane} --no-paste '/reload\r'`;
373
373
  } else {
374
374
  appendLedger(ctx.cwd, "auto_reload_skipped", { where, reason: "no supported multiplexer (tmux/WezTerm) in env" });
375
- return;
375
+ return false;
376
376
  }
377
377
  appendLedger(ctx.cwd, "auto_reload_injected", { where, transport, pane });
378
378
  ctx.ui.notify("glla: injecting /reload into this pane — extensions rebuild in place and the fresh instance holds state. /glla resume after (automatic with autoresume=on).", "info");
379
379
  exec(cmd, () => { /* best-effort */ });
380
+ return true;
380
381
  } catch {
381
382
  /* never let recovery take the warning path down */
383
+ return false;
384
+ }
385
+ }
386
+
387
+ /** v0.34.13: one rung of the auto-recovery ladder. Returns true when a
388
+ * /reload was injected (caller should NOT also pause/alert the manual
389
+ * cure). Writes the sidecar resume marker FIRST so the fresh instance
390
+ * resumes the goal/loop even with autoresume=off. Throttled — a second
391
+ * wedge inside the window is the pi-restart class and must reach the
392
+ * user as a loud stop, not another reload. */
393
+ let lastAutoRecoveryAt = 0;
394
+ function attemptAutoRecovery(ctx: ExtensionContext, where: string): boolean {
395
+ if (loadSettings(ctx.cwd).autoRecovery === false) return false;
396
+ const now = Date.now();
397
+ if (now - lastAutoRecoveryAt < AUTO_RECOVERY_THROTTLE_MS) return false;
398
+ // Inject FIRST — the throttle stamp + marker must reflect a real
399
+ // recovery, not a no-transport skip (else the watchdog would mislabel
400
+ // the next wedge as the pi-restart class).
401
+ if (!attemptAutoReload(ctx, where)) return false;
402
+ lastAutoRecoveryAt = now;
403
+ try {
404
+ fs.writeFileSync(
405
+ path.join(piGlaDir(ctx.cwd), RECOVERY_RESUME_MARKER),
406
+ JSON.stringify({ at: new Date(now).toISOString(), where }),
407
+ );
408
+ } catch { /* best-effort — the reload still helps; restore just holds */ }
409
+ appendLedger(ctx.cwd, "auto_recovery_reload", { where });
410
+ ctx.ui.notify(`glla auto-recovery: ${where} — the ${isLoopActive() ? "loop" : "goal/list item"} resumes ITSELF after the rebuild (no /glla resume needed; the sidecar marker carries the consent). If this recurs within ${Math.round(AUTO_RECOVERY_THROTTLE_MS / 60_000)}m it's the pi-restart class and you'll get a loud stop.`, "warning");
411
+ return true;
412
+ }
413
+
414
+ /** v0.34.14: /reload rebind detector. The extension runs INSIDE pi, so
415
+ * process.pid IS pi's pid: an instance that boots and finds its OWN pid
416
+ * already in the owner file is a /reload rebuild of a live session, not a
417
+ * cold boot. Rebinds always resume active goals/loops — holding mid-work
418
+ * after an in-place rebuild is pure friction (user directive: keep going
419
+ * unless we must stop; "the list is not continuing" after /reload,
420
+ * hellhunter 2026-08-01). Cold boots (new pid) still honor autoresume=off.
421
+ * Sidecar, not the ledger: read-before-write must be atomic-ish and the
422
+ * ledger is append-only. */
423
+ const SESSION_OWNER_FILE = "session-owner.json";
424
+ function claimSessionOwnerAndDetectRebind(cwd: string): boolean {
425
+ try {
426
+ const p = path.join(piGlaDir(cwd), SESSION_OWNER_FILE);
427
+ let prevPid: number | null = null;
428
+ try {
429
+ prevPid = (JSON.parse(fs.readFileSync(p, "utf-8")) as { pid?: number }).pid ?? null;
430
+ } catch { /* absent or corrupt — first boot */ }
431
+ fs.writeFileSync(p, JSON.stringify({ pid: process.pid, at: new Date().toISOString() }));
432
+ return prevPid !== null && prevPid === process.pid;
433
+ } catch {
434
+ return false;
435
+ }
436
+ }
437
+
438
+ /** v0.34.13: consume the sidecar marker on session restore. Single-use,
439
+ * freshness-bounded — a stale marker from an abandoned recovery must not
440
+ * surprise-resume a later session. */
441
+ function consumeRecoveryResume(cwd: string): boolean {
442
+ try {
443
+ const p = path.join(piGlaDir(cwd), RECOVERY_RESUME_MARKER);
444
+ if (!fs.existsSync(p)) return false;
445
+ const raw = fs.readFileSync(p, "utf-8");
446
+ fs.unlinkSync(p);
447
+ const at = Date.parse((JSON.parse(raw) as { at?: string }).at ?? "");
448
+ return !Number.isNaN(at) && Date.now() - at < RECOVERY_RESUME_FRESH_MS;
449
+ } catch {
450
+ return false;
382
451
  }
383
452
  }
384
453
 
@@ -635,7 +704,7 @@ const ZOMBIE_RUN_ALERT_THROTTLE_MS = 10 * 60_000;
635
704
  // (sendMessage never threw; session reported idle) but started NO turn —
636
705
  // transcript frozen, tokens flat, 10+ minutes of refires into the void.
637
706
  // Same family as the post-compaction dropped trigger (v0.26.5), but the
638
- // pending-latch watchdog needs idle&&pending and pi reported no pending
707
+ // the pending latch needs idle&&pending and pi reported no pending
639
708
  // here, and the zombie watchdog needs busy — this shape falls between both
640
709
  // chairs. Disarm signal = real activity (agent_end/tool_call) AFTER the
641
710
  // last send; a landed turn — even a lazy text-only one — disarms it.
@@ -649,6 +718,17 @@ const CONTINUATION_UNANSWERED_THROTTLE_MS = 300_000;
649
718
  // cycle. Sending 2.5s AFTER agent_end lets teardown settle; the send lands
650
719
  // and the next turn starts immediately. 2.5s per turn beats 60s per turn.
651
720
  const EAGER_CONTINUATION_SETTLE_MS = Number(process.env.GLLA_EAGER_SETTLE_MS ?? 2_500);
721
+ // v0.34.13: auto-recovery ladder ("keep going unless we MUST stop — a
722
+ // question, or done" — user directive 2026-08-01). A wedge that only a
723
+ // /reload cures should /reload ITSELF: inject the keystrokes (v0.29.13
724
+ // transport) with a sidecar marker so the fresh instance RESUMES even when
725
+ // autoresume=off (the consent came from autoRecovery at recovery time, not
726
+ // the restore-time setting). Throttled to one attempt per window: a
727
+ // recurrence inside the window is the transcript-writer-dead class, which
728
+ // only a pi restart cures — that one stays a loud stop for the human.
729
+ const AUTO_RECOVERY_THROTTLE_MS = 600_000;
730
+ const RECOVERY_RESUME_MARKER = "recovery-resume.json";
731
+ const RECOVERY_RESUME_FRESH_MS = 300_000;
652
732
  // v0.29.19: dead-turn caps (agent_end exemption path). 6 consecutive
653
733
  // provider-error turns = a real outage, not bad luck — stop honestly.
654
734
  // 3 consecutive user aborts = the user means it (user aborts mean STOP).
@@ -825,6 +905,10 @@ function escalateSendRearmStorm(ctx: ExtensionContext, kind: "continuation" | "l
825
905
  const silent = Math.round(SEND_REARM_ESCALATE_SILENT_MS / 60000);
826
906
  appendLedger(ctx.cwd, "send_rearm_escalated", { kind, afterMinutes: mins, silentMinutes: silent });
827
907
  if (kind === "loop" && isLoopActive()) {
908
+ if (attemptAutoRecovery(ctx, "send-retry storm")) {
909
+ appendLedger(ctx.cwd, "send_rearm_escalated_suppressed", { reason: "auto-recovery reload" });
910
+ return;
911
+ }
828
912
  clearLoopTimer();
829
913
  state.loop = { ...state.loop!, active: false, stopReason: `send-retry storm: ${mins}m of re-arms with no session activity for ${silent}m — the session is wedged. Press Escape to cancel the stuck run (pi's own rate-limit retry holds it; pi prints "escape to cancel"), then /loop resume — the loop holds on restore. If still wedged: /reload rebuilds extensions in place, then /loop resume again. Restart pi only if /reload itself fails.` };
830
914
  persistState(ctx);
@@ -848,6 +932,14 @@ function escalateSendRearmStorm(ctx: ExtensionContext, kind: "continuation" | "l
848
932
  return;
849
933
  }
850
934
  if (state.goal && state.goal.status === "active") {
935
+ // v0.34.13: keep going unless we MUST stop — try the auto-recovery
936
+ // /reload before spending the user's attention on a pause. A reload
937
+ // that fails to cure throttles the next attempt, and the pause below
938
+ // fires then as today.
939
+ if (attemptAutoRecovery(ctx, "send-retry storm")) {
940
+ appendLedger(ctx.cwd, "send_rearm_escalated_suppressed", { reason: "auto-recovery reload" });
941
+ return;
942
+ }
851
943
  updateGoal({
852
944
  status: "paused",
853
945
  pauseKind: "error",
@@ -863,6 +955,12 @@ function escalateStallNow(ctx: ExtensionContext, threshold: number): boolean {
863
955
  if (!shouldEscalateStall(consecutiveStalls, threshold)) return false;
864
956
  consecutiveStalls = 0;
865
957
  appendLedger(ctx.cwd, "stall_escalated", { threshold, kind: isLoopActive() ? "loop" : "goal" });
958
+ // v0.34.13: recovery before stop — usually throttled (the 2.5min
959
+ // watchdog already tried), in which case the stop proceeds as today.
960
+ if (attemptAutoRecovery(ctx, "stall escalation")) {
961
+ appendLedger(ctx.cwd, "stall_escalated_suppressed", { reason: "auto-recovery reload", threshold });
962
+ return true;
963
+ }
866
964
  if (isLoopActive()) {
867
965
  clearLoopTimer();
868
966
  state.loop = { ...state.loop!, active: false, stopReason: `stalled: ${threshold} continuation refires landed no turn — the session is not continuing (wedged message queue or stale API). Press Escape to cancel any stuck run, then /loop resume — the loop holds on restore. If the handle is stale, /reload rebuilds extensions in place; restart pi only if /reload fails.` };
@@ -972,9 +1070,17 @@ function heartbeatTick(): void {
972
1070
  ) {
973
1071
  lastUnansweredAlertAt = Date.now();
974
1072
  appendLedger(ctx.cwd, "continuation_unanswered", { silentMs: Date.now() - lastContinuationSentAt });
975
- const msg = `glla: pi accepted the continuation ${Math.round((Date.now() - lastContinuationSentAt) / 60_000)}m ago but NO turn has started — no tool calls, no tokens, transcript frozen (the turn trigger is wedged; same pi failure family as the post-compaction blackhole). Re-sends don't unstick it. Cure: /reload — autoresume re-fires the ${isLoopActive() ? "loop" : "goal/list item"} automatically.`;
976
- ctx.ui.notify(msg, "warning");
977
- notifyExternal(ctx, msg);
1073
+ // v0.34.13: recover, don't just report. Only the failure modes reach
1074
+ // the user: throttled (a reload already failed to cure → the
1075
+ // pi-restart class) or recovery unavailable (setting off / no pane).
1076
+ if (!attemptAutoRecovery(ctx, "continuation unanswered")) {
1077
+ const mins = Math.round((Date.now() - lastContinuationSentAt) / 60_000);
1078
+ const msg = lastAutoRecoveryAt > 0
1079
+ ? `glla: auto-recovery /reload did NOT unstick this session — still no turn ${mins}m after the continuation. This class kills pi's transcript writer; only a pi RESTART cures it: restart pi in this tab, then /glla resume (the goal holds in .pi-glla state).`
1080
+ : `glla: pi accepted the continuation ${mins}m ago but NO turn has started — no tool calls, no tokens, transcript frozen (the turn trigger is wedged). Cure: /reload — autoresume re-fires the ${isLoopActive() ? "loop" : "goal/list item"}. (auto-recovery is off or no tmux/WezTerm pane — /glla settings autoRecovery=on enables the automatic form.)`;
1081
+ ctx.ui.notify(msg, "warning");
1082
+ notifyExternal(ctx, msg);
1083
+ }
978
1084
  }
979
1085
  // v0.29.1: stranded-audit recovery. A goal left in "auditing" with NO
980
1086
  // in-flight audit means the auditor's result never landed (wedged queue
@@ -3597,7 +3703,13 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3597
3703
  // logged as disapprovals.
3598
3704
  const auditorRan = result.output.trim().length > 0;
3599
3705
  // v0.28.5 (E2): a REAL auditor run clears the infra-error streak.
3600
- if (auditorRan && (state.goal.auditInfraStreak ?? 0) > 0) updateGoal({ auditInfraStreak: undefined }, ctx);
3706
+ // v0.34.14: …but only a CLEAN one. A STALLED run returns the partial
3707
+ // output it streamed before the abort — non-empty, so auditorRan is
3708
+ // true — while result.error still marks it an infrastructure failure.
3709
+ // Clearing the streak on those meant the 3-strike breaker at :3874
3710
+ // NEVER engaged: pully 2026-08-01 looped 10-min stall cycles for 4h
3711
+ // (the auditor hung on an ssh/sudo verification every attempt).
3712
+ if (auditorRan && !result.error && (state.goal.auditInfraStreak ?? 0) > 0) updateGoal({ auditInfraStreak: undefined }, ctx);
3601
3713
  const history = state.goal.auditHistory ?? [];
3602
3714
  if (auditorRan) {
3603
3715
  // v0.25.4: strip think-block leakage (MiniMax-M3 `</think>`
@@ -3796,11 +3908,11 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3796
3908
  auditHistory: history,
3797
3909
  auditInfraStreak: infraStreak,
3798
3910
  pauseKind: "error",
3799
- pauseReason: `auditor infrastructure failed ${infraStreak}× in a row — the auditor model is likely broken (last: ${result.error.slice(0, 120)})`,
3911
+ pauseReason: `auditor infrastructure failed ${infraStreak}× in a row — the auditor model is likely broken OR a verification command is hanging (ssh/sudo/long test runs stall the stream) (last: ${result.error.slice(0, 120)})`,
3800
3912
  pauseSuggestedAction: "Fix the auditor model (/glla model=provider/id) or restart pi, then /goal resume. Your work was NOT judged.",
3801
3913
  }, ctx);
3802
3914
  appendLedger(ctx.cwd, "goal_paused", { reason: `auditor infra streak ${infraStreak}: ${result.error.slice(0, 120)}` });
3803
- ctx.ui.notify(`${goalNoun()} paused: auditor infrastructure failed ${infraStreak}× in a row. Fix the auditor model (/glla model=...), then /goal resume.`, "warning");
3915
+ ctx.ui.notify(`${goalNoun()} paused: auditor infrastructure failed ${infraStreak}× in a row — model broken or a verification command hanging (ssh/sudo/long runs). Fix with /glla model=... or unblock the command, then /goal resume.`, "warning");
3804
3916
  notifyExternal(ctx, `${goalNoun()} paused: auditor infrastructure ${infraStreak}× — model likely broken.`);
3805
3917
  return {
3806
3918
  content: [{
@@ -6333,9 +6445,16 @@ export default function (pi: ExtensionAPI): void {
6333
6445
  persistState(ctx);
6334
6446
  appendLedger(ctx.cwd, "audit_loop_target_migrated", { from: "audit-every-iteration", to: "fix-first" });
6335
6447
  }
6448
+ // v0.34.13: an auto-recovery /reload carries its own resume consent —
6449
+ // the sidecar marker overrides autoresume=off for THIS restore only.
6450
+ const recoveryResume = consumeRecoveryResume(ctx.cwd);
6451
+ // v0.34.14: a /reload rebind (same pi pid) ALWAYS resumes — the session
6452
+ // is live mid-work; holding is the "list is not continuing" bug.
6453
+ const rebindResume = claimSessionOwnerAndDetectRebind(ctx.cwd);
6454
+ if (rebindResume) appendLedger(ctx.cwd, "rebind_resume", { pid: process.pid });
6336
6455
  if (isLoopActive()) {
6337
6456
  const l = state.loop!;
6338
- if (autoResume) {
6457
+ if (autoResume || recoveryResume || rebindResume) {
6339
6458
  ctx.ui.notify(
6340
6459
  `Resuming loop (iteration ${l.iteration}/${l.maxIterations > 0 ? l.maxIterations : "∞"}, best ${l.bestValue ?? "n/a"}, stall ${l.stallCount}/${l.plateauWindow}): ${l.target.slice(0, 60)}`,
6341
6460
  "info",
@@ -6356,7 +6475,7 @@ export default function (pi: ExtensionAPI): void {
6356
6475
  // "load it but not auto start it"). Interrupted goals hold like
6357
6476
  // everything else; autoresume=on (unattended rigs) still auto-resumes
6358
6477
  // them, and the marker is cleared only on that promised auto-resume.
6359
- if (autoResume) {
6478
+ if (autoResume || recoveryResume || rebindResume) {
6360
6479
  // v0.28.1 (S2): clear the stale-handle interrupt marker — this IS
6361
6480
  // the auto-resume the marker promised.
6362
6481
  if (wasInterrupted) updateGoal({ interruptedAt: undefined, interruptedReason: undefined }, ctx);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.34.12",
3
+ "version": "0.34.14",
4
4
  "description": "Mission control for autonomous pi: interview-drafted goals, an audited task queue, and forever-loops (metric, spec, project-audit) that run for hours. An isolated extension-less auditor re-verifies every completion with raw evidence; confirmed drafts, decision pauses and consent gates keep you in charge.",
5
5
  "license": "MIT",
6
6
  "author": "dracon",