pi-goal-list-loop-audit 0.29.18 → 0.29.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10,7 +10,7 @@
10
10
  * self-reports progress.
11
11
  */
12
12
 
13
- import { existsSync, statSync } from "node:fs";
13
+ import { existsSync, readFileSync, statSync } from "node:fs";
14
14
  import { join } from "node:path";
15
15
 
16
16
  export type LoopDirection = "min" | "max";
@@ -55,6 +55,22 @@ export interface LoopState {
55
55
  * (a real number that didn't improve); a broken measure says nothing
56
56
  * about movement and must stop the loop with its own loud reason. */
57
57
  consecutiveNullMeasures?: number;
58
+ /** v0.29.19: consecutive provider-error/user-abort turns. Exempt from
59
+ * stall/stuck/plateau accounting (the model never got a say — a dead
60
+ * turn is not evidence about the work); capped so an outage stops the
61
+ * loop with an honest reason instead of burning turns forever. Field
62
+ * 2026-07-31 (MiniMax token-plan 429 storm): hegemon false-plateau'd
63
+ * with 13 open findings, polis with 3+, hellhunter stuck-stopped at
64
+ * iter 93 — every counted turn was a dead 429 turn. */
65
+ consecutiveErrors?: number;
66
+ /** v0.29.19: audit plateau reprieves used so far (open findings remain
67
+ * = the well isn't dry, so the plateau stop stands down). Bounded by
68
+ * AUDIT_PLATEAU_MAX_REPRIEVES — the next plateau stop stands, honestly
69
+ * named. */
70
+ auditPlateauReprieves?: number;
71
+ /** v0.29.19: one-shot shove injected into the next iteration's prompt
72
+ * after a plateau reprieve; cleared on use. */
73
+ auditReprieveNote?: string;
58
74
  bestValue: number | null;
59
75
  lastValue: number | null;
60
76
  /** v0.29.10: audit loops (measure counts open findings) get a deferred
@@ -430,6 +446,24 @@ export function auditTarget(): string {
430
446
  return `Audit the project for real problems and fix them, iteration by iteration — FIX-FIRST: the open backlog comes down before new hunting (user design 2026-07-30: "audit to fix then audit then fix again" — not find-and-present). Every iteration: (1) FIX the highest-severity OPEN finding(s) in ${AUDIT_FINDINGS_REL} — real fixes, committed — then check the box: "- [x] … — fixed in <commit>". An iteration that closes nothing while OPEN findings remain is a wasted iteration: if the top findings are genuinely blocked, say what blocks them in one line and work the first unblocked one — "no new action this turn" is never an acceptable iteration while open boxes exist. (2) RE-AUDIT on cadence, not every iteration — run a fresh audit pass (spawn Explore subagents for breadth; hunting real issues: bugs, broken flows, regressions, drift between docs and code, dead code, security holes; not style nits, not speculative refactors) ONLY when no OPEN findings remain, when roughly ten iterations have passed since the last pass, or when your own fixes plausibly broke something. (3) Append every NEW finding as one checkbox line "- [ ] SEVERITY: short description (file:line)" to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed). (4) Honesty law: never fabricate findings to look busy; never mark a finding fixed without the fix commit existing. The orchestrator counts CLOSED findings every iteration (direction=max): discovery alone does not move the metric — landing fixes does. When a full audit pass surfaces nothing new AND no open findings remain, say so plainly — the plateau stop ends the loop when the well is dry.`;
431
447
  }
432
448
 
449
+ /** v0.29.19: how many times an audit loop's plateau stop stands down
450
+ * while open findings remain. The plateau after the last reprieve stops
451
+ * the loop with the honest "no closure despite K open findings" reason. */
452
+ export const AUDIT_PLATEAU_MAX_REPRIEVES = 2;
453
+
454
+ /** v0.29.19: orchestrator-side count of OPEN audit findings — the honest
455
+ * "is the well dry" signal for audit-loop plateau decisions. The plateau
456
+ * stop means "the well is dry"; with K open boxes it is objectively not. */
457
+ export function countOpenAuditFindings(cwd: string): number {
458
+ try {
459
+ const p = join(cwd, AUDIT_FINDINGS_REL);
460
+ if (!existsSync(p)) return 0;
461
+ return readFileSync(p, "utf-8").split("\n").filter((l) => /^- \[ \]/.test(l)).length;
462
+ } catch {
463
+ return 0;
464
+ }
465
+ }
466
+
433
467
  // ---- /goal audit-project (v0.29.8) ----
434
468
 
435
469
  /**
@@ -171,7 +171,11 @@ import {
171
171
  respecTarget,
172
172
  auditMeasureCmd,
173
173
  auditTarget,
174
+ AUDIT_PLATEAU_MAX_REPRIEVES,
175
+ countOpenAuditFindings,
176
+ AUDIT_FINDINGS_REL,
174
177
  projectAuditTarget,
178
+ type LoopTickOutcome,
175
179
  HELD_ON_RESTORE,
176
180
  type LoopState,
177
181
  } from "../goal-loop-forever.js";
@@ -267,6 +271,13 @@ export function __testOnlyResetStaleFlag(): void {
267
271
  extensionApiStale = false;
268
272
  }
269
273
 
274
+ /** Test-only: release the claimed session owner so a later test file can
275
+ * drive agent_end with its own sessionManager identity (ownerSession is
276
+ * process-wide module state; behavioral-orchestrator claims it first). */
277
+ export function __testOnlyResetOwnerSession(): void {
278
+ ownerSession = null;
279
+ }
280
+
270
281
  /** v0.28.1 (S3): side-effect-free staleness probe — getSessionName()
271
282
  * routes through pi's assertActive() and throws the stale signature iff
272
283
  * pi invalidated this factory handle (session replacement). A positive
@@ -461,6 +472,11 @@ let heartbeatTimer: NodeJS.Timeout | null = null;
461
472
 
462
473
  const ZOMBIE_RUN_SILENT_MS = 20 * 60_000;
463
474
  const ZOMBIE_RUN_ALERT_THROTTLE_MS = 10 * 60_000;
475
+ // v0.29.19: dead-turn caps (agent_end exemption path). 6 consecutive
476
+ // provider-error turns = a real outage, not bad luck — stop honestly.
477
+ // 3 consecutive user aborts = the user means it (user aborts mean STOP).
478
+ const LOOP_MAX_CONSECUTIVE_ERRORS = 6;
479
+ const LOOP_MAX_CONSECUTIVE_ABORTS = 3;
464
480
 
465
481
  function noteActivity(real = false): void {
466
482
  lastActivityAt = Date.now();
@@ -2338,9 +2354,13 @@ function sendLoopTurn(): void {
2338
2354
  }
2339
2355
  // v0.24.0: a stuck intervention REPLACES the pep talk — the rotating
2340
2356
  // directive names why the loop is stuck and what rung of the ladder it's on.
2341
- const interventionNote = (loop.consecutiveStuck ?? 0) > 0 && loop.lastStuckReason
2357
+ // v0.29.19: a plateau reprieve's one-shot shove takes priority over the
2358
+ // stuck directive (they can't both be meaningful in the same iteration).
2359
+ const reprieveNote = loop.auditReprieveNote ?? "";
2360
+ if (reprieveNote) loop.auditReprieveNote = undefined;
2361
+ const interventionNote = reprieveNote || ((loop.consecutiveStuck ?? 0) > 0 && loop.lastStuckReason
2342
2362
  ? loopInterventionDirective(loop.consecutiveStuck!, loop.lastStuckReason, loop.recentTexts ?? [])
2343
- : "";
2363
+ : "");
2344
2364
  // v0.24.0: identical prompts invite identical answers — rotate the base
2345
2365
  // instruction (metricless loops; metric loops already vary via values).
2346
2366
  const variantNote = metricless ? continueVariant(loop.iteration) : "";
@@ -2455,7 +2475,7 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
2455
2475
  loop.consecutiveStuck = 0;
2456
2476
  loop.lastStuckReason = undefined;
2457
2477
  }
2458
- const outcome = metricless ? applyMetriclessTick(loop, nowIso()) : applyMeasurement(loop, value, nowIso());
2478
+ let outcome: LoopTickOutcome = metricless ? applyMetriclessTick(loop, nowIso()) : applyMeasurement(loop, value, nowIso());
2459
2479
  persistState(ctx);
2460
2480
  appendLedger(ctx.cwd, "loop_measured", {
2461
2481
  iteration: loop.iteration,
@@ -2494,6 +2514,34 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
2494
2514
  return;
2495
2515
  }
2496
2516
  if (outcome.kind === "stop") {
2517
+ // v0.29.19: an audit loop's plateau is only honest when the well is
2518
+ // ACTUALLY dry. Plateauing with open findings means the agent fumbled
2519
+ // (or the provider ate) N turns — not "nothing left" (field: hegemon
2520
+ // stopped at best 74 with 13 OPEN boxes; polis at best 46 with 3+).
2521
+ // Stand the stop down with a strategy shove — bounded: the plateau
2522
+ // after the last reprieve stops with an honest blocked-named reason.
2523
+ if (loop.kind === "audit" && outcome.reason.startsWith("plateau —")) {
2524
+ const open = countOpenAuditFindings(ctx.cwd);
2525
+ if (open > 0) {
2526
+ const reprieves = (loop.auditPlateauReprieves ?? 0) + 1;
2527
+ if (reprieves <= AUDIT_PLATEAU_MAX_REPRIEVES) {
2528
+ loop.active = true;
2529
+ loop.stopReason = undefined;
2530
+ loop.stallCount = 0;
2531
+ loop.auditPlateauReprieves = reprieves;
2532
+ loop.auditReprieveNote = `PLATEAU REPRIEVE (${reprieves}/${AUDIT_PLATEAU_MAX_REPRIEVES}): ${open} finding(s) still OPEN in ${AUDIT_FINDINGS_REL} — the plateau stop does not fire while the well isn't dry. Stop hunting and stop narrating: pick the smallest OPEN finding and CLOSE it this iteration (fix commit + checked box). ${AUDIT_PLATEAU_MAX_REPRIEVES - reprieves} reprieve(s) remain.`;
2533
+ persistState(ctx);
2534
+ appendLedger(ctx.cwd, "audit_plateau_reprieve", { open, reprieves, best: loop.bestValue });
2535
+ ctx.ui.notify(`Audit loop plateau reprieve (${reprieves}/${AUDIT_PLATEAU_MAX_REPRIEVES}): ${open} open findings — the well isn't dry, continuing.`, "info");
2536
+ scheduleLoopTick(ctx);
2537
+ return;
2538
+ }
2539
+ const honest = `plateau — no closure in ${loop.plateauWindow}×${reprieves} iterations despite ${open} open findings (treat as blocked; /loop resume to push again)`;
2540
+ loop.stopReason = honest;
2541
+ persistState(ctx);
2542
+ outcome = { kind: "stop", reason: honest };
2543
+ }
2544
+ }
2497
2545
  await finishLoopGit(ctx, loop);
2498
2546
  ctx.ui.notify(`Loop stopped: ${outcome.reason}. ${loop.history.length} iterations recorded.`, "info");
2499
2547
  appendLedger(ctx.cwd, "loop_stopped", { reason: outcome.reason, iterations: loop.iteration, best: loop.bestValue });
@@ -2637,14 +2685,27 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
2637
2685
  return;
2638
2686
  }
2639
2687
  const stored = state.loop;
2640
- if (stored && !stored.active && stored.stopReason === HELD_ON_RESTORE) {
2688
+ // v0.29.20: plain plateau stops are resumable too — pre-gate plateaus
2689
+ // could be false (hegemon/polis stopped 2026-07-31 with open findings
2690
+ // on 429-dead turns), and an explicit resume is the user's call; the
2691
+ // v0.29.19 gate + re-armed counters make the resumed run honest.
2692
+ const RESUMABLE_STOP = (r?: string): boolean =>
2693
+ r === HELD_ON_RESTORE ||
2694
+ !!r?.startsWith("provider errors —") ||
2695
+ !!r?.startsWith("stopped by user —") ||
2696
+ !!r?.startsWith("plateau —") ||
2697
+ !!r?.startsWith("stuck —");
2698
+ if (stored && !stored.active && RESUMABLE_STOP(stored.stopReason)) {
2641
2699
  // v0.28.14: one-active-thing — a held loop must not resume over an
2642
2700
  // active goal/list-item (this was the last unguarded stacking path).
2643
2701
  if (state.goal && state.goal.status === "active") {
2644
2702
  ctx.ui.notify("A goal is active — the held loop stays held. /goal pause or /goal cancel it first, then /loop resume.", "warning");
2645
2703
  return;
2646
2704
  }
2647
- state.loop = { ...stored, active: true, stopReason: undefined };
2705
+ // An explicit resume re-arms the counters: fresh stall window,
2706
+ // cleared dead-turn/stuck streaks, reprieves restored — the user
2707
+ // saying "push again" wins over the ladder's memory (v0.29.19).
2708
+ state.loop = { ...stored, active: true, stopReason: undefined, consecutiveErrors: 0, consecutiveStuck: 0, lastStuckReason: undefined, stallCount: 0, auditPlateauReprieves: 0 };
2648
2709
  persistState(ctx);
2649
2710
  scheduleLoopTick(ctx);
2650
2711
  ctx.ui.notify(
@@ -5672,6 +5733,36 @@ export default function (pi: ExtensionAPI): void {
5672
5733
  // Loop 3 runs on the same heartbeat: measure after every agent turn.
5673
5734
  if (isLoopActive()) {
5674
5735
  clearLoopTimer();
5736
+ // v0.29.19: provider-error / user-abort turns are NOT iterations —
5737
+ // the model never got a say, so a dead turn carries no stall/stuck/
5738
+ // plateau signal (field 2026-07-31, MiniMax token-plan 429 storm:
5739
+ // hegemon false-plateau'd with 13 open findings, polis with 3+,
5740
+ // hellhunter stuck-stopped at iter 93 — every counted turn was a
5741
+ // dead 429 turn; the v0.28.13/v0.29.4 exemptions only covered the
5742
+ // goal nudge counter). Skip the measure and refire — bounded, so a
5743
+ // real outage stops the loop honestly instead of burning turns.
5744
+ const sr = lastA?.stopReason;
5745
+ if (sr === "error" || sr === "aborted") {
5746
+ const loop = state.loop!;
5747
+ loop.consecutiveErrors = (loop.consecutiveErrors ?? 0) + 1;
5748
+ persistState(ctx);
5749
+ appendLedger(ctx.cwd, "loop_turn_exempt_error", { stopReason: sr, consecutive: loop.consecutiveErrors, iteration: loop.iteration });
5750
+ const cap = sr === "aborted" ? LOOP_MAX_CONSECUTIVE_ABORTS : LOOP_MAX_CONSECUTIVE_ERRORS;
5751
+ if (loop.consecutiveErrors >= cap) {
5752
+ loop.active = false;
5753
+ loop.stopReason = sr === "aborted"
5754
+ ? `stopped by user — ${loop.consecutiveErrors} consecutive aborts (iteration ${loop.iteration} preserved; /loop resume to continue)`
5755
+ : `provider errors — ${loop.consecutiveErrors} consecutive error turns (iteration ${loop.iteration} preserved; /loop resume when the provider recovers)`;
5756
+ persistState(ctx);
5757
+ ctx.ui.notify(`Loop stopped: ${loop.stopReason}`, "warning");
5758
+ appendLedger(ctx.cwd, "loop_stopped", { reason: loop.stopReason, iterations: loop.iteration, best: loop.bestValue });
5759
+ notifyExternal(ctx, `Loop stopped: ${sr === "aborted" ? "user aborts" : "provider errors"} (${loop.consecutiveErrors}×)`);
5760
+ return;
5761
+ }
5762
+ scheduleLoopTick(ctx);
5763
+ return;
5764
+ }
5765
+ if ((state.loop!.consecutiveErrors ?? 0) > 0) state.loop!.consecutiveErrors = 0; // a real turn clears the streak (runLoopTick persists)
5675
5766
  await runLoopTick(ctx, event);
5676
5767
  return;
5677
5768
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.29.18",
3
+ "version": "0.29.20",
4
4
  "description": "Goal. Loop. Audit. Done. \u2014 a pi-coding-agent extension that supervises long-running work, with isolated auditor on each completion. Beat bamboozling by design: the auditor runs in a fresh session with no extensions, no skills, no editor \u2014 only the read tools needed to verify your goal.",
5
5
  "license": "MIT",
6
6
  "author": "dracon",