pi-goal-list-loop-audit 0.29.9 → 0.29.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -57,6 +57,10 @@ export interface LoopState {
57
57
  consecutiveNullMeasures?: number;
58
58
  bestValue: number | null;
59
59
  lastValue: number | null;
60
+ /** v0.29.10: audit loops (measure counts open findings) get a deferred
61
+ * baseline (first REAL measurement seeds best — the pre-discovery 0 is
62
+ * degenerate) and audit-flavoured regression wording. */
63
+ kind?: "audit";
60
64
  active: boolean;
61
65
  stopReason?: string;
62
66
  history: LoopMeasure[];
@@ -2240,9 +2240,20 @@ function sendLoopTurn(): void {
2240
2240
  return;
2241
2241
  }
2242
2242
  const loop = state.loop!;
2243
- const regressedLast = loop.history.length > 0 && !loop.history[loop.history.length - 1]!.improved && loop.lastValue !== null;
2244
- const regressionNote = regressedLast
2245
- ? "**Your last change REGRESSED the metric. Undo it first, then try a different small change.**"
2243
+ // v0.29.10: "regressed" = the last two measurements moved the WRONG way
2244
+ // — not merely "didn't beat best". The old trigger (any non-improving
2245
+ // iteration) cried REGRESSED on stalls and on the audit loop's
2246
+ // degenerate baseline-0, telling agents to undo GOOD fixes (junk-runner
2247
+ // 2026-07-30: 17→16 was real progress; the prompt demanded a revert).
2248
+ const hist = loop.history;
2249
+ const prevValue = hist.length >= 2 ? hist[hist.length - 2]!.value : null;
2250
+ const lastHistValue = hist.length >= 1 ? hist[hist.length - 1]!.value : null;
2251
+ const trueRegression = prevValue !== null && lastHistValue !== null && loop.direction !== undefined &&
2252
+ (loop.direction === "min" ? lastHistValue > prevValue : lastHistValue < prevValue);
2253
+ const regressionNote = trueRegression
2254
+ ? loop.kind === "audit"
2255
+ ? "**The open-findings count went UP last iteration — either the fresh audit pass found new problems, or your fix didn't land. Check findings.md, then keep fixing the highest-severity OPEN items.**"
2256
+ : "**Your last change REGRESSED the metric. Undo it first, then try a different small change.**"
2246
2257
  : "";
2247
2258
  // Strategy rotation (from pi-loop-mode's one good idea): one stall before
2248
2259
  // the plateau window closes, stop polishing and change approach entirely.
@@ -2462,6 +2473,14 @@ interface LoopConfig {
2462
2473
  tokenBudget?: number;
2463
2474
  /** v0.25.1: /loop start toolsamerepeat=N (0 = disable legacy check). */
2464
2475
  toolSameRepeat?: number;
2476
+ /** v0.29.10: don't seed bestValue from the pre-work baseline measure —
2477
+ * the first REAL measurement becomes the baseline. For loops whose
2478
+ * metric is created BY the first iteration (the audit loop: 0 open
2479
+ * findings just means findings.md doesn't exist yet); a seeded 0 pins
2480
+ * best at a value no iteration can beat, stalling every iteration. */
2481
+ deferBaseline?: boolean;
2482
+ /** v0.29.10: audit loops get audit-flavoured regression wording. */
2483
+ kind?: "audit";
2465
2484
  }
2466
2485
 
2467
2486
  /** Shared loop-start path: /loop start AND propose_loop_draft (after Confirm). */
@@ -2497,8 +2516,8 @@ async function startLoopFromConfig(ctx: ExtensionContext, cfg: LoopConfig): Prom
2497
2516
  // v0.23.0: metricless loops skip the baseline entirely — there is no
2498
2517
  // measure to run, and no plateau to protect.
2499
2518
  const metricless = !cfg.measureCmd;
2500
- const baseline = metricless ? null : await runMeasure(ctx, cfg.measureCmd);
2501
- if (!metricless && baseline === null && !(cfg as { force?: boolean }).force) {
2519
+ const baseline = metricless || cfg.deferBaseline ? null : await runMeasure(ctx, cfg.measureCmd);
2520
+ if (!metricless && !cfg.deferBaseline && baseline === null && !(cfg as { force?: boolean }).force) {
2502
2521
  ctx.ui.notify(
2503
2522
  `/loop start refused: the measure produced no number.\nCommand: ${cfg.measureCmd}\nFix it so it prints exactly one number, or re-run with force=1 if it only works after the agent builds something first.\n(Non-numeric goal — research, docs, features? Use /goal: the independent auditor verifies semantically. /loop only believes a number.)`,
2504
2523
  "warning",
@@ -2516,8 +2535,9 @@ async function startLoopFromConfig(ctx: ExtensionContext, cfg: LoopConfig): Prom
2516
2535
  maxIterations: cfg.maxIterations,
2517
2536
  plateauWindow: cfg.plateauWindow,
2518
2537
  stallCount: 0,
2519
- bestValue: baseline,
2520
- lastValue: baseline,
2538
+ bestValue: cfg.deferBaseline ? null : baseline,
2539
+ lastValue: cfg.deferBaseline ? null : baseline,
2540
+ kind: cfg.kind,
2521
2541
  active: true,
2522
2542
  history: [],
2523
2543
  startedAt: nowIso(),
@@ -2536,7 +2556,7 @@ async function startLoopFromConfig(ctx: ExtensionContext, cfg: LoopConfig): Prom
2536
2556
  metricless
2537
2557
  ? `Loop started (metricless spec loop — NO plateau stop): ${cfg.target.slice(0, 60)}\nEnds only at ${cfg.maxIterations > 0 ? `max ${cfg.maxIterations} iterations` : "no iteration cap"}${cfg.timeLimitHours ? ` · ${cfg.timeLimitHours}h` : ""}${cfg.tokenBudget ? ` · ${cfg.tokenBudget.toLocaleString()} tokens` : ""} · /loop stop. Every iteration must make ONE real, inspectable change — cosmetic churn is the doorknob failure.` +
2538
2558
  (branchName ? `\nbranch mode: committing each iteration to ${branchName}` : "")
2539
- : `Loop started: ${cfg.target.slice(0, 60)}\nBaseline: ${baseline ?? "(forced without a number — first turn must produce one)"} · direction ${cfg.direction} · window ${cfg.plateauWindow} · ${cfg.maxIterations > 0 ? `max ${cfg.maxIterations}` : "no iteration cap"}` +
2559
+ : `Loop started: ${cfg.target.slice(0, 60)}\nBaseline: ${cfg.deferBaseline ? "deferred — the first real measurement seeds it" : (baseline ?? "(forced without a number — first turn must produce one)")} · direction ${cfg.direction} · window ${cfg.plateauWindow} · ${cfg.maxIterations > 0 ? `max ${cfg.maxIterations}` : "no iteration cap"}` +
2540
2560
  (branchName ? `\nbranch mode: committing improvements to ${branchName}` : ""),
2541
2561
  "info",
2542
2562
  );
@@ -2695,6 +2715,11 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
2695
2715
  maxIterations: 0,
2696
2716
  branch: false,
2697
2717
  force: false,
2718
+ // v0.29.10: the audit loop's metric is CREATED by iteration 1 —
2719
+ // seeding best from the pre-discovery 0 stalls every iteration and
2720
+ // plateau-stops mid-work at the window. Defer the baseline.
2721
+ deferBaseline: true,
2722
+ kind: "audit",
2698
2723
  });
2699
2724
  return;
2700
2725
  }
@@ -5303,6 +5328,16 @@ export default function (pi: ExtensionAPI): void {
5303
5328
  ) {
5304
5329
  ctx.ui.notify("Auto-resume fired (event: session start). Continue working.", "info");
5305
5330
  }
5331
+ // v0.29.10: reseed live/held audit loops stuck on the degenerate
5332
+ // pre-discovery baseline-0 — best can never go below 0, so every
5333
+ // iteration stalls and the prompt cries REGRESSED on real progress
5334
+ // (junk-runner ran pinned like this). Null best = the next measurement
5335
+ // becomes the honest baseline; stall streak resets with it.
5336
+ if (state.loop && state.loop.measureCmd?.includes("audit-loop/findings.md") && state.loop.bestValue === 0) {
5337
+ state.loop = { ...state.loop, bestValue: null, stallCount: 0, kind: state.loop.kind ?? "audit" };
5338
+ persistState(ctx);
5339
+ appendLedger(ctx.cwd, "audit_loop_baseline_reseeded", { from: 0 });
5340
+ }
5306
5341
  if (isLoopActive()) {
5307
5342
  const l = state.loop!;
5308
5343
  if (autoResume) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.29.9",
3
+ "version": "0.29.10",
4
4
  "description": "Goal. Loop. Audit. Done. \u2014 a pi-coding-agent extension that supervises long-running work, with isolated auditor on each completion. Beat bamboozling by design: the auditor runs in a fresh session with no extensions, no skills, no editor \u2014 only the read tools needed to verify your goal.",
5
5
  "license": "MIT",
6
6
  "author": "dracon",