pi-goal-list-loop-audit 0.29.9 → 0.29.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -173,10 +173,11 @@ export function buildStatusText(state: State, audit?: AuditDisplayProgress | nul
|
|
|
173
173
|
return `glla: ${paint(theme, pauseIsError(g) ? "error" : "warning", label)}${heldSuffix}`;
|
|
174
174
|
}
|
|
175
175
|
if (g.status === "active") {
|
|
176
|
-
// v0.28.1 (S1/S2): a stale-handle interrupt keeps the goal ACTIVE
|
|
177
|
-
//
|
|
176
|
+
// v0.28.1 (S1/S2): a stale-handle interrupt keeps the goal ACTIVE.
|
|
177
|
+
// v0.29.11: the fresh session HOLDS it (hold-everything restore gate;
|
|
178
|
+
// autoresume=on resumes for you) — name the verb, don't promise auto.
|
|
178
179
|
if (g.interruptedAt) {
|
|
179
|
-
return `glla: ${g.policy} ${paint(theme, "error", "⚠ interrupted — stale handle ·
|
|
180
|
+
return `glla: ${g.policy} ${paint(theme, "error", "⚠ interrupted — stale handle · restart pi → /glla resume")}${heldSuffix}`;
|
|
180
181
|
}
|
|
181
182
|
// v0.24.7: list policy gets its own wording — a queue item is not a goal.
|
|
182
183
|
// v0.28.11 (U10): goal policy joins it — "list 29" read as a command
|
|
@@ -57,6 +57,10 @@ export interface LoopState {
|
|
|
57
57
|
consecutiveNullMeasures?: number;
|
|
58
58
|
bestValue: number | null;
|
|
59
59
|
lastValue: number | null;
|
|
60
|
+
/** v0.29.10: audit loops (measure counts open findings) get a deferred
|
|
61
|
+
* baseline (first REAL measurement seeds best — the pre-discovery 0 is
|
|
62
|
+
* degenerate) and audit-flavoured regression wording. */
|
|
63
|
+
kind?: "audit";
|
|
60
64
|
active: boolean;
|
|
61
65
|
stopReason?: string;
|
|
62
66
|
history: LoopMeasure[];
|
package/extensions/loops/goal.ts
CHANGED
|
@@ -219,7 +219,7 @@ function goStaleTerminal(ctx: ExtensionContext, where: string): void {
|
|
|
219
219
|
if (extensionApiStale) return; // already terminal — don't re-spam
|
|
220
220
|
extensionApiStale = true;
|
|
221
221
|
appendLedger(ctx.cwd, "extension_api_stale", { where, kind: isLoopActive() ? "loop" : "goal" });
|
|
222
|
-
const guidance = "pi invalidated this session's extension handle (session replacement — compaction triggers it in pi 0.82.x). Sends can never land in this process. Restart pi (or reload extensions) —
|
|
222
|
+
const guidance = "pi invalidated this session's extension handle (session replacement — compaction triggers it in pi 0.82.x). Sends can never land in this process. Restart pi (or reload extensions) — the goal/loop holds on the fresh session: /glla resume continues it (autoresume=on resumes for you).";
|
|
223
223
|
if (isLoopActive()) {
|
|
224
224
|
clearLoopTimer();
|
|
225
225
|
state.loop = { ...state.loop!, active: false, stopReason: `extension api stale: ${guidance}` };
|
|
@@ -260,7 +260,7 @@ function warnIfStaleAtEntry(ctx: ExtensionContext, what: string): boolean {
|
|
|
260
260
|
if (!probeExtensionApiStale()) return false;
|
|
261
261
|
appendLedger(ctx.cwd, "extension_api_stale", { where: `entry probe (${what})` });
|
|
262
262
|
ctx.ui.notify(
|
|
263
|
-
`glla: this session's extension handle is stale (pi session replacement) — ${what} can't send continuations in this process. State is safe in .pi-glla/ — restart pi
|
|
263
|
+
`glla: this session's extension handle is stale (pi session replacement) — ${what} can't send continuations in this process. State is safe in .pi-glla/ — restart pi, then /glla resume (autoresume=on resumes for you).`,
|
|
264
264
|
"warning",
|
|
265
265
|
);
|
|
266
266
|
return true;
|
|
@@ -544,9 +544,9 @@ function escalateSendRearmStorm(ctx: ExtensionContext, kind: "continuation" | "l
|
|
|
544
544
|
appendLedger(ctx.cwd, "send_rearm_escalated", { kind, afterMinutes: mins, silentMinutes: silent });
|
|
545
545
|
if (kind === "loop" && isLoopActive()) {
|
|
546
546
|
clearLoopTimer();
|
|
547
|
-
state.loop = { ...state.loop!, active: false, stopReason: `send-retry storm: ${mins}m of re-arms with no session activity for ${silent}m — the session is wedged. Restart pi, then /loop
|
|
547
|
+
state.loop = { ...state.loop!, active: false, stopReason: `send-retry storm: ${mins}m of re-arms with no session activity for ${silent}m — the session is wedged. Restart pi, then /loop resume (the loop holds on restore).` };
|
|
548
548
|
persistState(ctx);
|
|
549
|
-
ctx.ui.notify(`Loop stopped: send-retry storm (${mins}m, session silent ${silent}m). Restart pi
|
|
549
|
+
ctx.ui.notify(`Loop stopped: send-retry storm (${mins}m, session silent ${silent}m). Restart pi, then /loop resume (the loop holds on restore).`, "warning");
|
|
550
550
|
notifyExternal(ctx, "Loop stopped: send-retry storm.");
|
|
551
551
|
return;
|
|
552
552
|
}
|
|
@@ -583,9 +583,9 @@ function escalateStallNow(ctx: ExtensionContext, threshold: number): boolean {
|
|
|
583
583
|
appendLedger(ctx.cwd, "stall_escalated", { threshold, kind: isLoopActive() ? "loop" : "goal" });
|
|
584
584
|
if (isLoopActive()) {
|
|
585
585
|
clearLoopTimer();
|
|
586
|
-
state.loop = { ...state.loop!, active: false, stopReason: `stalled: ${threshold} continuation refires landed no turn — the session is not continuing (wedged message queue or stale API). Restart pi, then /loop
|
|
586
|
+
state.loop = { ...state.loop!, active: false, stopReason: `stalled: ${threshold} continuation refires landed no turn — the session is not continuing (wedged message queue or stale API). Restart pi, then /loop resume (the loop holds on restore).` };
|
|
587
587
|
persistState(ctx);
|
|
588
|
-
ctx.ui.notify(`Loop stopped: ${threshold} refires produced no turn — the continuation is not landing. Restart pi
|
|
588
|
+
ctx.ui.notify(`Loop stopped: ${threshold} refires produced no turn — the continuation is not landing. Restart pi, then /loop resume (the loop holds on restore).`, "warning");
|
|
589
589
|
notifyExternal(ctx, "Loop stopped: stalled (continuation not landing).");
|
|
590
590
|
return true;
|
|
591
591
|
}
|
|
@@ -619,12 +619,16 @@ function heartbeatTick(): void {
|
|
|
619
619
|
// machinery below stays quiet for 3 minutes while the replaced session
|
|
620
620
|
// settles (latch watchdog, wedge alert, refire counting all resume after).
|
|
621
621
|
if (Date.now() < compactionGraceUntil) return;
|
|
622
|
-
// v0.
|
|
623
|
-
//
|
|
624
|
-
//
|
|
625
|
-
//
|
|
626
|
-
//
|
|
627
|
-
|
|
622
|
+
// v0.29.11: PROBE the handle before the stall machinery runs — a
|
|
623
|
+
// session-replaced handle can never land a refire, so detect it on the
|
|
624
|
+
// first tick after replacement and go terminal immediately, instead of
|
|
625
|
+
// burning refires into the void until a send happens to throw (field:
|
|
626
|
+
// polis stall 3/5, endless-td stall 1/5 before the warning fired).
|
|
627
|
+
// v0.28.27: once terminal, ALL stall machinery stays quiet from here
|
|
628
|
+
// on: refiring into a dead process is misleading, and worse, the stall
|
|
629
|
+
// escalation would PAUSE the goal — silently cancelling the
|
|
630
|
+
// interruptedAt → hold-on-restart promise the footer shows.
|
|
631
|
+
if (probeExtensionApiStale()) { goStaleTerminal(ctx, "heartbeat probe"); return; }
|
|
628
632
|
// v0.29.1: stranded-audit recovery. A goal left in "auditing" with NO
|
|
629
633
|
// in-flight audit means the auditor's result never landed (wedged queue
|
|
630
634
|
// ate the tool result; compaction/restart mid-audit). Field-observed in
|
|
@@ -2240,9 +2244,20 @@ function sendLoopTurn(): void {
|
|
|
2240
2244
|
return;
|
|
2241
2245
|
}
|
|
2242
2246
|
const loop = state.loop!;
|
|
2243
|
-
|
|
2244
|
-
|
|
2245
|
-
|
|
2247
|
+
// v0.29.10: "regressed" = the last two measurements moved the WRONG way
|
|
2248
|
+
// — not merely "didn't beat best". The old trigger (any non-improving
|
|
2249
|
+
// iteration) cried REGRESSED on stalls and on the audit loop's
|
|
2250
|
+
// degenerate baseline-0, telling agents to undo GOOD fixes (junk-runner
|
|
2251
|
+
// 2026-07-30: 17→16 was real progress; the prompt demanded a revert).
|
|
2252
|
+
const hist = loop.history;
|
|
2253
|
+
const prevValue = hist.length >= 2 ? hist[hist.length - 2]!.value : null;
|
|
2254
|
+
const lastHistValue = hist.length >= 1 ? hist[hist.length - 1]!.value : null;
|
|
2255
|
+
const trueRegression = prevValue !== null && lastHistValue !== null && loop.direction !== undefined &&
|
|
2256
|
+
(loop.direction === "min" ? lastHistValue > prevValue : lastHistValue < prevValue);
|
|
2257
|
+
const regressionNote = trueRegression
|
|
2258
|
+
? loop.kind === "audit"
|
|
2259
|
+
? "**The open-findings count went UP last iteration — either the fresh audit pass found new problems, or your fix didn't land. Check findings.md, then keep fixing the highest-severity OPEN items.**"
|
|
2260
|
+
: "**Your last change REGRESSED the metric. Undo it first, then try a different small change.**"
|
|
2246
2261
|
: "";
|
|
2247
2262
|
// Strategy rotation (from pi-loop-mode's one good idea): one stall before
|
|
2248
2263
|
// the plateau window closes, stop polishing and change approach entirely.
|
|
@@ -2462,6 +2477,14 @@ interface LoopConfig {
|
|
|
2462
2477
|
tokenBudget?: number;
|
|
2463
2478
|
/** v0.25.1: /loop start toolsamerepeat=N (0 = disable legacy check). */
|
|
2464
2479
|
toolSameRepeat?: number;
|
|
2480
|
+
/** v0.29.10: don't seed bestValue from the pre-work baseline measure —
|
|
2481
|
+
* the first REAL measurement becomes the baseline. For loops whose
|
|
2482
|
+
* metric is created BY the first iteration (the audit loop: 0 open
|
|
2483
|
+
* findings just means findings.md doesn't exist yet); a seeded 0 pins
|
|
2484
|
+
* best at a value no iteration can beat, stalling every iteration. */
|
|
2485
|
+
deferBaseline?: boolean;
|
|
2486
|
+
/** v0.29.10: audit loops get audit-flavoured regression wording. */
|
|
2487
|
+
kind?: "audit";
|
|
2465
2488
|
}
|
|
2466
2489
|
|
|
2467
2490
|
/** Shared loop-start path: /loop start AND propose_loop_draft (after Confirm). */
|
|
@@ -2497,8 +2520,8 @@ async function startLoopFromConfig(ctx: ExtensionContext, cfg: LoopConfig): Prom
|
|
|
2497
2520
|
// v0.23.0: metricless loops skip the baseline entirely — there is no
|
|
2498
2521
|
// measure to run, and no plateau to protect.
|
|
2499
2522
|
const metricless = !cfg.measureCmd;
|
|
2500
|
-
const baseline = metricless ? null : await runMeasure(ctx, cfg.measureCmd);
|
|
2501
|
-
if (!metricless && baseline === null && !(cfg as { force?: boolean }).force) {
|
|
2523
|
+
const baseline = metricless || cfg.deferBaseline ? null : await runMeasure(ctx, cfg.measureCmd);
|
|
2524
|
+
if (!metricless && !cfg.deferBaseline && baseline === null && !(cfg as { force?: boolean }).force) {
|
|
2502
2525
|
ctx.ui.notify(
|
|
2503
2526
|
`/loop start refused: the measure produced no number.\nCommand: ${cfg.measureCmd}\nFix it so it prints exactly one number, or re-run with force=1 if it only works after the agent builds something first.\n(Non-numeric goal — research, docs, features? Use /goal: the independent auditor verifies semantically. /loop only believes a number.)`,
|
|
2504
2527
|
"warning",
|
|
@@ -2516,8 +2539,9 @@ async function startLoopFromConfig(ctx: ExtensionContext, cfg: LoopConfig): Prom
|
|
|
2516
2539
|
maxIterations: cfg.maxIterations,
|
|
2517
2540
|
plateauWindow: cfg.plateauWindow,
|
|
2518
2541
|
stallCount: 0,
|
|
2519
|
-
bestValue: baseline,
|
|
2520
|
-
lastValue: baseline,
|
|
2542
|
+
bestValue: cfg.deferBaseline ? null : baseline,
|
|
2543
|
+
lastValue: cfg.deferBaseline ? null : baseline,
|
|
2544
|
+
kind: cfg.kind,
|
|
2521
2545
|
active: true,
|
|
2522
2546
|
history: [],
|
|
2523
2547
|
startedAt: nowIso(),
|
|
@@ -2536,7 +2560,7 @@ async function startLoopFromConfig(ctx: ExtensionContext, cfg: LoopConfig): Prom
|
|
|
2536
2560
|
metricless
|
|
2537
2561
|
? `Loop started (metricless spec loop — NO plateau stop): ${cfg.target.slice(0, 60)}\nEnds only at ${cfg.maxIterations > 0 ? `max ${cfg.maxIterations} iterations` : "no iteration cap"}${cfg.timeLimitHours ? ` · ${cfg.timeLimitHours}h` : ""}${cfg.tokenBudget ? ` · ${cfg.tokenBudget.toLocaleString()} tokens` : ""} · /loop stop. Every iteration must make ONE real, inspectable change — cosmetic churn is the doorknob failure.` +
|
|
2538
2562
|
(branchName ? `\nbranch mode: committing each iteration to ${branchName}` : "")
|
|
2539
|
-
: `Loop started: ${cfg.target.slice(0, 60)}\nBaseline: ${baseline ?? "(forced without a number — first turn must produce one)"} · direction ${cfg.direction} · window ${cfg.plateauWindow} · ${cfg.maxIterations > 0 ? `max ${cfg.maxIterations}` : "no iteration cap"}` +
|
|
2563
|
+
: `Loop started: ${cfg.target.slice(0, 60)}\nBaseline: ${cfg.deferBaseline ? "deferred — the first real measurement seeds it" : (baseline ?? "(forced without a number — first turn must produce one)")} · direction ${cfg.direction} · window ${cfg.plateauWindow} · ${cfg.maxIterations > 0 ? `max ${cfg.maxIterations}` : "no iteration cap"}` +
|
|
2540
2564
|
(branchName ? `\nbranch mode: committing improvements to ${branchName}` : ""),
|
|
2541
2565
|
"info",
|
|
2542
2566
|
);
|
|
@@ -2695,6 +2719,11 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
2695
2719
|
maxIterations: 0,
|
|
2696
2720
|
branch: false,
|
|
2697
2721
|
force: false,
|
|
2722
|
+
// v0.29.10: the audit loop's metric is CREATED by iteration 1 —
|
|
2723
|
+
// seeding best from the pre-discovery 0 stalls every iteration and
|
|
2724
|
+
// plateau-stops mid-work at the window. Defer the baseline.
|
|
2725
|
+
deferBaseline: true,
|
|
2726
|
+
kind: "audit",
|
|
2698
2727
|
});
|
|
2699
2728
|
return;
|
|
2700
2729
|
}
|
|
@@ -5303,6 +5332,26 @@ export default function (pi: ExtensionAPI): void {
|
|
|
5303
5332
|
) {
|
|
5304
5333
|
ctx.ui.notify("Auto-resume fired (event: session start). Continue working.", "info");
|
|
5305
5334
|
}
|
|
5335
|
+
// v0.29.11: loops stopped by the stale-handle terminal or the stall
|
|
5336
|
+
// escalation told the user "restart pi, then /loop start" — but a
|
|
5337
|
+
// fresh start discards iteration/best/history. Hold them on load like
|
|
5338
|
+
// any restore-held loop: /loop resume continues from the saved state.
|
|
5339
|
+
if (state.loop && !state.loop.active &&
|
|
5340
|
+
(state.loop.stopReason?.startsWith("extension api stale") || state.loop.stopReason?.startsWith("stalled:") || state.loop.stopReason?.startsWith("send-retry storm:"))) {
|
|
5341
|
+
appendLedger(ctx.cwd, "loop_held_for_resume", { was: (state.loop.stopReason ?? "").slice(0, 40) });
|
|
5342
|
+
state.loop = { ...state.loop, stopReason: HELD_ON_RESTORE };
|
|
5343
|
+
persistState(ctx);
|
|
5344
|
+
}
|
|
5345
|
+
// v0.29.10: reseed live/held audit loops stuck on the degenerate
|
|
5346
|
+
// pre-discovery baseline-0 — best can never go below 0, so every
|
|
5347
|
+
// iteration stalls and the prompt cries REGRESSED on real progress
|
|
5348
|
+
// (junk-runner ran pinned like this). Null best = the next measurement
|
|
5349
|
+
// becomes the honest baseline; stall streak resets with it.
|
|
5350
|
+
if (state.loop && state.loop.measureCmd?.includes("audit-loop/findings.md") && state.loop.bestValue === 0) {
|
|
5351
|
+
state.loop = { ...state.loop, bestValue: null, stallCount: 0, kind: state.loop.kind ?? "audit" };
|
|
5352
|
+
persistState(ctx);
|
|
5353
|
+
appendLedger(ctx.cwd, "audit_loop_baseline_reseeded", { from: 0 });
|
|
5354
|
+
}
|
|
5306
5355
|
if (isLoopActive()) {
|
|
5307
5356
|
const l = state.loop!;
|
|
5308
5357
|
if (autoResume) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-goal-list-loop-audit",
|
|
3
|
-
"version": "0.29.
|
|
3
|
+
"version": "0.29.11",
|
|
4
4
|
"description": "Goal. Loop. Audit. Done. \u2014 a pi-coding-agent extension that supervises long-running work, with isolated auditor on each completion. Beat bamboozling by design: the auditor runs in a fresh session with no extensions, no skills, no editor \u2014 only the read tools needed to verify your goal.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "dracon",
|