pi-goal-list-loop-audit 0.29.18 → 0.29.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
* self-reports progress.
|
|
11
11
|
*/
|
|
12
12
|
|
|
13
|
-
import { existsSync, statSync } from "node:fs";
|
|
13
|
+
import { existsSync, readFileSync, statSync } from "node:fs";
|
|
14
14
|
import { join } from "node:path";
|
|
15
15
|
|
|
16
16
|
export type LoopDirection = "min" | "max";
|
|
@@ -55,6 +55,22 @@ export interface LoopState {
|
|
|
55
55
|
* (a real number that didn't improve); a broken measure says nothing
|
|
56
56
|
* about movement and must stop the loop with its own loud reason. */
|
|
57
57
|
consecutiveNullMeasures?: number;
|
|
58
|
+
/** v0.29.19: consecutive provider-error/user-abort turns. Exempt from
|
|
59
|
+
* stall/stuck/plateau accounting (the model never got a say — a dead
|
|
60
|
+
* turn is not evidence about the work); capped so an outage stops the
|
|
61
|
+
* loop with an honest reason instead of burning turns forever. Field
|
|
62
|
+
* 2026-07-31 (MiniMax token-plan 429 storm): hegemon false-plateau'd
|
|
63
|
+
* with 13 open findings, polis with 3+, hellhunter stuck-stopped at
|
|
64
|
+
* iter 93 — every counted turn was a dead 429 turn. */
|
|
65
|
+
consecutiveErrors?: number;
|
|
66
|
+
/** v0.29.19: audit plateau reprieves used so far (open findings remain
|
|
67
|
+
* = the well isn't dry, so the plateau stop stands down). Bounded by
|
|
68
|
+
* AUDIT_PLATEAU_MAX_REPRIEVES — the next plateau stop stands, honestly
|
|
69
|
+
* named. */
|
|
70
|
+
auditPlateauReprieves?: number;
|
|
71
|
+
/** v0.29.19: one-shot shove injected into the next iteration's prompt
|
|
72
|
+
* after a plateau reprieve; cleared on use. */
|
|
73
|
+
auditReprieveNote?: string;
|
|
58
74
|
bestValue: number | null;
|
|
59
75
|
lastValue: number | null;
|
|
60
76
|
/** v0.29.10: audit loops (measure counts open findings) get a deferred
|
|
@@ -430,6 +446,24 @@ export function auditTarget(): string {
|
|
|
430
446
|
return `Audit the project for real problems and fix them, iteration by iteration — FIX-FIRST: the open backlog comes down before new hunting (user design 2026-07-30: "audit to fix then audit then fix again" — not find-and-present). Every iteration: (1) FIX the highest-severity OPEN finding(s) in ${AUDIT_FINDINGS_REL} — real fixes, committed — then check the box: "- [x] … — fixed in <commit>". An iteration that closes nothing while OPEN findings remain is a wasted iteration: if the top findings are genuinely blocked, say what blocks them in one line and work the first unblocked one — "no new action this turn" is never an acceptable iteration while open boxes exist. (2) RE-AUDIT on cadence, not every iteration — run a fresh audit pass (spawn Explore subagents for breadth; hunting real issues: bugs, broken flows, regressions, drift between docs and code, dead code, security holes; not style nits, not speculative refactors) ONLY when no OPEN findings remain, when roughly ten iterations have passed since the last pass, or when your own fixes plausibly broke something. (3) Append every NEW finding as one checkbox line "- [ ] SEVERITY: short description (file:line)" to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed). (4) Honesty law: never fabricate findings to look busy; never mark a finding fixed without the fix commit existing. The orchestrator counts CLOSED findings every iteration (direction=max): discovery alone does not move the metric — landing fixes does. When a full audit pass surfaces nothing new AND no open findings remain, say so plainly — the plateau stop ends the loop when the well is dry.`;
|
|
431
447
|
}
|
|
432
448
|
|
|
449
|
+
/** v0.29.19: how many times an audit loop's plateau stop stands down
|
|
450
|
+
* while open findings remain. The plateau after the last reprieve stops
|
|
451
|
+
* the loop with the honest "no closure despite K open findings" reason. */
|
|
452
|
+
export const AUDIT_PLATEAU_MAX_REPRIEVES = 2;
|
|
453
|
+
|
|
454
|
+
/** v0.29.19: orchestrator-side count of OPEN audit findings — the honest
|
|
455
|
+
* "is the well dry" signal for audit-loop plateau decisions. The plateau
|
|
456
|
+
* stop means "the well is dry"; with K open boxes it is objectively not. */
|
|
457
|
+
export function countOpenAuditFindings(cwd: string): number {
|
|
458
|
+
try {
|
|
459
|
+
const p = join(cwd, AUDIT_FINDINGS_REL);
|
|
460
|
+
if (!existsSync(p)) return 0;
|
|
461
|
+
return readFileSync(p, "utf-8").split("\n").filter((l) => /^- \[ \]/.test(l)).length;
|
|
462
|
+
} catch {
|
|
463
|
+
return 0;
|
|
464
|
+
}
|
|
465
|
+
}
|
|
466
|
+
|
|
433
467
|
// ---- /goal audit-project (v0.29.8) ----
|
|
434
468
|
|
|
435
469
|
/**
|
package/extensions/loops/goal.ts
CHANGED
|
@@ -171,7 +171,11 @@ import {
|
|
|
171
171
|
respecTarget,
|
|
172
172
|
auditMeasureCmd,
|
|
173
173
|
auditTarget,
|
|
174
|
+
AUDIT_PLATEAU_MAX_REPRIEVES,
|
|
175
|
+
countOpenAuditFindings,
|
|
176
|
+
AUDIT_FINDINGS_REL,
|
|
174
177
|
projectAuditTarget,
|
|
178
|
+
type LoopTickOutcome,
|
|
175
179
|
HELD_ON_RESTORE,
|
|
176
180
|
type LoopState,
|
|
177
181
|
} from "../goal-loop-forever.js";
|
|
@@ -267,6 +271,13 @@ export function __testOnlyResetStaleFlag(): void {
|
|
|
267
271
|
extensionApiStale = false;
|
|
268
272
|
}
|
|
269
273
|
|
|
274
|
+
/** Test-only: release the claimed session owner so a later test file can
|
|
275
|
+
* drive agent_end with its own sessionManager identity (ownerSession is
|
|
276
|
+
* process-wide module state; behavioral-orchestrator claims it first). */
|
|
277
|
+
export function __testOnlyResetOwnerSession(): void {
|
|
278
|
+
ownerSession = null;
|
|
279
|
+
}
|
|
280
|
+
|
|
270
281
|
/** v0.28.1 (S3): side-effect-free staleness probe — getSessionName()
|
|
271
282
|
* routes through pi's assertActive() and throws the stale signature iff
|
|
272
283
|
* pi invalidated this factory handle (session replacement). A positive
|
|
@@ -461,6 +472,11 @@ let heartbeatTimer: NodeJS.Timeout | null = null;
|
|
|
461
472
|
|
|
462
473
|
const ZOMBIE_RUN_SILENT_MS = 20 * 60_000;
|
|
463
474
|
const ZOMBIE_RUN_ALERT_THROTTLE_MS = 10 * 60_000;
|
|
475
|
+
// v0.29.19: dead-turn caps (agent_end exemption path). 6 consecutive
|
|
476
|
+
// provider-error turns = a real outage, not bad luck — stop honestly.
|
|
477
|
+
// 3 consecutive user aborts = the user means it (user aborts mean STOP).
|
|
478
|
+
const LOOP_MAX_CONSECUTIVE_ERRORS = 6;
|
|
479
|
+
const LOOP_MAX_CONSECUTIVE_ABORTS = 3;
|
|
464
480
|
|
|
465
481
|
function noteActivity(real = false): void {
|
|
466
482
|
lastActivityAt = Date.now();
|
|
@@ -2338,9 +2354,13 @@ function sendLoopTurn(): void {
|
|
|
2338
2354
|
}
|
|
2339
2355
|
// v0.24.0: a stuck intervention REPLACES the pep talk — the rotating
|
|
2340
2356
|
// directive names why the loop is stuck and what rung of the ladder it's on.
|
|
2341
|
-
|
|
2357
|
+
// v0.29.19: a plateau reprieve's one-shot shove takes priority over the
|
|
2358
|
+
// stuck directive (they can't both be meaningful in the same iteration).
|
|
2359
|
+
const reprieveNote = loop.auditReprieveNote ?? "";
|
|
2360
|
+
if (reprieveNote) loop.auditReprieveNote = undefined;
|
|
2361
|
+
const interventionNote = reprieveNote || ((loop.consecutiveStuck ?? 0) > 0 && loop.lastStuckReason
|
|
2342
2362
|
? loopInterventionDirective(loop.consecutiveStuck!, loop.lastStuckReason, loop.recentTexts ?? [])
|
|
2343
|
-
: "";
|
|
2363
|
+
: "");
|
|
2344
2364
|
// v0.24.0: identical prompts invite identical answers — rotate the base
|
|
2345
2365
|
// instruction (metricless loops; metric loops already vary via values).
|
|
2346
2366
|
const variantNote = metricless ? continueVariant(loop.iteration) : "";
|
|
@@ -2455,7 +2475,7 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
|
|
|
2455
2475
|
loop.consecutiveStuck = 0;
|
|
2456
2476
|
loop.lastStuckReason = undefined;
|
|
2457
2477
|
}
|
|
2458
|
-
|
|
2478
|
+
let outcome: LoopTickOutcome = metricless ? applyMetriclessTick(loop, nowIso()) : applyMeasurement(loop, value, nowIso());
|
|
2459
2479
|
persistState(ctx);
|
|
2460
2480
|
appendLedger(ctx.cwd, "loop_measured", {
|
|
2461
2481
|
iteration: loop.iteration,
|
|
@@ -2494,6 +2514,34 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
|
|
|
2494
2514
|
return;
|
|
2495
2515
|
}
|
|
2496
2516
|
if (outcome.kind === "stop") {
|
|
2517
|
+
// v0.29.19: an audit loop's plateau is only honest when the well is
|
|
2518
|
+
// ACTUALLY dry. Plateauing with open findings means the agent fumbled
|
|
2519
|
+
// (or the provider ate) N turns — not "nothing left" (field: hegemon
|
|
2520
|
+
// stopped at best 74 with 13 OPEN boxes; polis at best 46 with 3+).
|
|
2521
|
+
// Stand the stop down with a strategy shove — bounded: the plateau
|
|
2522
|
+
// after the last reprieve stops with an honest blocked-named reason.
|
|
2523
|
+
if (loop.kind === "audit" && outcome.reason.startsWith("plateau —")) {
|
|
2524
|
+
const open = countOpenAuditFindings(ctx.cwd);
|
|
2525
|
+
if (open > 0) {
|
|
2526
|
+
const reprieves = (loop.auditPlateauReprieves ?? 0) + 1;
|
|
2527
|
+
if (reprieves <= AUDIT_PLATEAU_MAX_REPRIEVES) {
|
|
2528
|
+
loop.active = true;
|
|
2529
|
+
loop.stopReason = undefined;
|
|
2530
|
+
loop.stallCount = 0;
|
|
2531
|
+
loop.auditPlateauReprieves = reprieves;
|
|
2532
|
+
loop.auditReprieveNote = `PLATEAU REPRIEVE (${reprieves}/${AUDIT_PLATEAU_MAX_REPRIEVES}): ${open} finding(s) still OPEN in ${AUDIT_FINDINGS_REL} — the plateau stop does not fire while the well isn't dry. Stop hunting and stop narrating: pick the smallest OPEN finding and CLOSE it this iteration (fix commit + checked box). ${AUDIT_PLATEAU_MAX_REPRIEVES - reprieves} reprieve(s) remain.`;
|
|
2533
|
+
persistState(ctx);
|
|
2534
|
+
appendLedger(ctx.cwd, "audit_plateau_reprieve", { open, reprieves, best: loop.bestValue });
|
|
2535
|
+
ctx.ui.notify(`Audit loop plateau reprieve (${reprieves}/${AUDIT_PLATEAU_MAX_REPRIEVES}): ${open} open findings — the well isn't dry, continuing.`, "info");
|
|
2536
|
+
scheduleLoopTick(ctx);
|
|
2537
|
+
return;
|
|
2538
|
+
}
|
|
2539
|
+
const honest = `plateau — no closure in ${loop.plateauWindow}×${reprieves} iterations despite ${open} open findings (treat as blocked; /loop resume to push again)`;
|
|
2540
|
+
loop.stopReason = honest;
|
|
2541
|
+
persistState(ctx);
|
|
2542
|
+
outcome = { kind: "stop", reason: honest };
|
|
2543
|
+
}
|
|
2544
|
+
}
|
|
2497
2545
|
await finishLoopGit(ctx, loop);
|
|
2498
2546
|
ctx.ui.notify(`Loop stopped: ${outcome.reason}. ${loop.history.length} iterations recorded.`, "info");
|
|
2499
2547
|
appendLedger(ctx.cwd, "loop_stopped", { reason: outcome.reason, iterations: loop.iteration, best: loop.bestValue });
|
|
@@ -2637,14 +2685,23 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
2637
2685
|
return;
|
|
2638
2686
|
}
|
|
2639
2687
|
const stored = state.loop;
|
|
2640
|
-
|
|
2688
|
+
const RESUMABLE_STOP = (r?: string): boolean =>
|
|
2689
|
+
r === HELD_ON_RESTORE ||
|
|
2690
|
+
!!r?.startsWith("provider errors —") ||
|
|
2691
|
+
!!r?.startsWith("stopped by user —") ||
|
|
2692
|
+
!!r?.startsWith("plateau — no closure in") ||
|
|
2693
|
+
!!r?.startsWith("stuck —");
|
|
2694
|
+
if (stored && !stored.active && RESUMABLE_STOP(stored.stopReason)) {
|
|
2641
2695
|
// v0.28.14: one-active-thing — a held loop must not resume over an
|
|
2642
2696
|
// active goal/list-item (this was the last unguarded stacking path).
|
|
2643
2697
|
if (state.goal && state.goal.status === "active") {
|
|
2644
2698
|
ctx.ui.notify("A goal is active — the held loop stays held. /goal pause or /goal cancel it first, then /loop resume.", "warning");
|
|
2645
2699
|
return;
|
|
2646
2700
|
}
|
|
2647
|
-
|
|
2701
|
+
// An explicit resume re-arms the counters: fresh stall window,
|
|
2702
|
+
// cleared dead-turn/stuck streaks, reprieves restored — the user
|
|
2703
|
+
// saying "push again" wins over the ladder's memory (v0.29.19).
|
|
2704
|
+
state.loop = { ...stored, active: true, stopReason: undefined, consecutiveErrors: 0, consecutiveStuck: 0, lastStuckReason: undefined, stallCount: 0, auditPlateauReprieves: 0 };
|
|
2648
2705
|
persistState(ctx);
|
|
2649
2706
|
scheduleLoopTick(ctx);
|
|
2650
2707
|
ctx.ui.notify(
|
|
@@ -5672,6 +5729,36 @@ export default function (pi: ExtensionAPI): void {
|
|
|
5672
5729
|
// Loop 3 runs on the same heartbeat: measure after every agent turn.
|
|
5673
5730
|
if (isLoopActive()) {
|
|
5674
5731
|
clearLoopTimer();
|
|
5732
|
+
// v0.29.19: provider-error / user-abort turns are NOT iterations —
|
|
5733
|
+
// the model never got a say, so a dead turn carries no stall/stuck/
|
|
5734
|
+
// plateau signal (field 2026-07-31, MiniMax token-plan 429 storm:
|
|
5735
|
+
// hegemon false-plateau'd with 13 open findings, polis with 3+,
|
|
5736
|
+
// hellhunter stuck-stopped at iter 93 — every counted turn was a
|
|
5737
|
+
// dead 429 turn; the v0.28.13/v0.29.4 exemptions only covered the
|
|
5738
|
+
// goal nudge counter). Skip the measure and refire — bounded, so a
|
|
5739
|
+
// real outage stops the loop honestly instead of burning turns.
|
|
5740
|
+
const sr = lastA?.stopReason;
|
|
5741
|
+
if (sr === "error" || sr === "aborted") {
|
|
5742
|
+
const loop = state.loop!;
|
|
5743
|
+
loop.consecutiveErrors = (loop.consecutiveErrors ?? 0) + 1;
|
|
5744
|
+
persistState(ctx);
|
|
5745
|
+
appendLedger(ctx.cwd, "loop_turn_exempt_error", { stopReason: sr, consecutive: loop.consecutiveErrors, iteration: loop.iteration });
|
|
5746
|
+
const cap = sr === "aborted" ? LOOP_MAX_CONSECUTIVE_ABORTS : LOOP_MAX_CONSECUTIVE_ERRORS;
|
|
5747
|
+
if (loop.consecutiveErrors >= cap) {
|
|
5748
|
+
loop.active = false;
|
|
5749
|
+
loop.stopReason = sr === "aborted"
|
|
5750
|
+
? `stopped by user — ${loop.consecutiveErrors} consecutive aborts (iteration ${loop.iteration} preserved; /loop resume to continue)`
|
|
5751
|
+
: `provider errors — ${loop.consecutiveErrors} consecutive error turns (iteration ${loop.iteration} preserved; /loop resume when the provider recovers)`;
|
|
5752
|
+
persistState(ctx);
|
|
5753
|
+
ctx.ui.notify(`Loop stopped: ${loop.stopReason}`, "warning");
|
|
5754
|
+
appendLedger(ctx.cwd, "loop_stopped", { reason: loop.stopReason, iterations: loop.iteration, best: loop.bestValue });
|
|
5755
|
+
notifyExternal(ctx, `Loop stopped: ${sr === "aborted" ? "user aborts" : "provider errors"} (${loop.consecutiveErrors}×)`);
|
|
5756
|
+
return;
|
|
5757
|
+
}
|
|
5758
|
+
scheduleLoopTick(ctx);
|
|
5759
|
+
return;
|
|
5760
|
+
}
|
|
5761
|
+
if ((state.loop!.consecutiveErrors ?? 0) > 0) state.loop!.consecutiveErrors = 0; // a real turn clears the streak (runLoopTick persists)
|
|
5675
5762
|
await runLoopTick(ctx, event);
|
|
5676
5763
|
return;
|
|
5677
5764
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-goal-list-loop-audit",
|
|
3
|
-
"version": "0.29.
|
|
3
|
+
"version": "0.29.19",
|
|
4
4
|
"description": "Goal. Loop. Audit. Done. \u2014 a pi-coding-agent extension that supervises long-running work, with isolated auditor on each completion. Beat bamboozling by design: the auditor runs in a fresh session with no extensions, no skills, no editor \u2014 only the read tools needed to verify your goal.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "dracon",
|