pi-goal-list-loop-audit 0.29.8 → 0.29.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -9,6 +9,18 @@
|
|
|
9
9
|
*/
|
|
10
10
|
|
|
11
11
|
export const BACKOFF_HARD_CAP_MS = 5 * 60 * 1000;
|
|
12
|
+
|
|
13
|
+
// v0.29.9: ms until the next clock-hour boundary (+ grace). Coding-plan
|
|
14
|
+
// rate-limit windows typically expire at the top of the hour, so a probe
|
|
15
|
+
// scheduled there catches the reset within seconds instead of mid-window.
|
|
16
|
+
// Robust when the premise is wrong too: a non-clock-aligned window is
|
|
17
|
+
// still caught within the hour.
|
|
18
|
+
export function msUntilNextHourBoundary(nowMs: number, graceMs = 60_000): number {
|
|
19
|
+
const d = new Date(nowMs);
|
|
20
|
+
d.setMinutes(0, 0, 0);
|
|
21
|
+
d.setHours(d.getHours() + 1);
|
|
22
|
+
return d.getTime() + graceMs - nowMs;
|
|
23
|
+
}
|
|
12
24
|
export const BACKOFF_IDLE_RETRY_MS = 50; // when adding another iter to queue
|
|
13
25
|
export const BACKOFF_ERROR_BASE_MS = 5_000; // first error retry
|
|
14
26
|
export const BACKOFF_ERROR_MAX_MS = 60_000; // max error retry (separate from stuck cap)
|
|
@@ -57,6 +57,10 @@ export interface LoopState {
|
|
|
57
57
|
consecutiveNullMeasures?: number;
|
|
58
58
|
bestValue: number | null;
|
|
59
59
|
lastValue: number | null;
|
|
60
|
+
/** v0.29.10: audit loops (measure counts open findings) get a deferred
|
|
61
|
+
* baseline (first REAL measurement seeds best — the pre-discovery 0 is
|
|
62
|
+
* degenerate) and audit-flavoured regression wording. */
|
|
63
|
+
kind?: "audit";
|
|
60
64
|
active: boolean;
|
|
61
65
|
stopReason?: string;
|
|
62
66
|
history: LoopMeasure[];
|
package/extensions/loops/goal.ts
CHANGED
|
@@ -183,6 +183,7 @@ import {
|
|
|
183
183
|
shouldWedgeAlert,
|
|
184
184
|
PENDING_LATCH_STUCK_MS,
|
|
185
185
|
shouldFirePendingLatchWatchdog,
|
|
186
|
+
msUntilNextHourBoundary,
|
|
186
187
|
} from "../goal-loop-backoff.js";
|
|
187
188
|
|
|
188
189
|
// =================================================================
|
|
@@ -2239,9 +2240,20 @@ function sendLoopTurn(): void {
|
|
|
2239
2240
|
return;
|
|
2240
2241
|
}
|
|
2241
2242
|
const loop = state.loop!;
|
|
2242
|
-
|
|
2243
|
-
|
|
2244
|
-
|
|
2243
|
+
// v0.29.10: "regressed" = the last two measurements moved the WRONG way
|
|
2244
|
+
// — not merely "didn't beat best". The old trigger (any non-improving
|
|
2245
|
+
// iteration) cried REGRESSED on stalls and on the audit loop's
|
|
2246
|
+
// degenerate baseline-0, telling agents to undo GOOD fixes (junk-runner
|
|
2247
|
+
// 2026-07-30: 17→16 was real progress; the prompt demanded a revert).
|
|
2248
|
+
const hist = loop.history;
|
|
2249
|
+
const prevValue = hist.length >= 2 ? hist[hist.length - 2]!.value : null;
|
|
2250
|
+
const lastHistValue = hist.length >= 1 ? hist[hist.length - 1]!.value : null;
|
|
2251
|
+
const trueRegression = prevValue !== null && lastHistValue !== null && loop.direction !== undefined &&
|
|
2252
|
+
(loop.direction === "min" ? lastHistValue > prevValue : lastHistValue < prevValue);
|
|
2253
|
+
const regressionNote = trueRegression
|
|
2254
|
+
? loop.kind === "audit"
|
|
2255
|
+
? "**The open-findings count went UP last iteration — either the fresh audit pass found new problems, or your fix didn't land. Check findings.md, then keep fixing the highest-severity OPEN items.**"
|
|
2256
|
+
: "**Your last change REGRESSED the metric. Undo it first, then try a different small change.**"
|
|
2245
2257
|
: "";
|
|
2246
2258
|
// Strategy rotation (from pi-loop-mode's one good idea): one stall before
|
|
2247
2259
|
// the plateau window closes, stop polishing and change approach entirely.
|
|
@@ -2461,6 +2473,14 @@ interface LoopConfig {
|
|
|
2461
2473
|
tokenBudget?: number;
|
|
2462
2474
|
/** v0.25.1: /loop start toolsamerepeat=N (0 = disable legacy check). */
|
|
2463
2475
|
toolSameRepeat?: number;
|
|
2476
|
+
/** v0.29.10: don't seed bestValue from the pre-work baseline measure —
|
|
2477
|
+
* the first REAL measurement becomes the baseline. For loops whose
|
|
2478
|
+
* metric is created BY the first iteration (the audit loop: 0 open
|
|
2479
|
+
* findings just means findings.md doesn't exist yet); a seeded 0 pins
|
|
2480
|
+
* best at a value no iteration can beat, stalling every iteration. */
|
|
2481
|
+
deferBaseline?: boolean;
|
|
2482
|
+
/** v0.29.10: audit loops get audit-flavoured regression wording. */
|
|
2483
|
+
kind?: "audit";
|
|
2464
2484
|
}
|
|
2465
2485
|
|
|
2466
2486
|
/** Shared loop-start path: /loop start AND propose_loop_draft (after Confirm). */
|
|
@@ -2496,8 +2516,8 @@ async function startLoopFromConfig(ctx: ExtensionContext, cfg: LoopConfig): Prom
|
|
|
2496
2516
|
// v0.23.0: metricless loops skip the baseline entirely — there is no
|
|
2497
2517
|
// measure to run, and no plateau to protect.
|
|
2498
2518
|
const metricless = !cfg.measureCmd;
|
|
2499
|
-
const baseline = metricless ? null : await runMeasure(ctx, cfg.measureCmd);
|
|
2500
|
-
if (!metricless && baseline === null && !(cfg as { force?: boolean }).force) {
|
|
2519
|
+
const baseline = metricless || cfg.deferBaseline ? null : await runMeasure(ctx, cfg.measureCmd);
|
|
2520
|
+
if (!metricless && !cfg.deferBaseline && baseline === null && !(cfg as { force?: boolean }).force) {
|
|
2501
2521
|
ctx.ui.notify(
|
|
2502
2522
|
`/loop start refused: the measure produced no number.\nCommand: ${cfg.measureCmd}\nFix it so it prints exactly one number, or re-run with force=1 if it only works after the agent builds something first.\n(Non-numeric goal — research, docs, features? Use /goal: the independent auditor verifies semantically. /loop only believes a number.)`,
|
|
2503
2523
|
"warning",
|
|
@@ -2515,8 +2535,9 @@ async function startLoopFromConfig(ctx: ExtensionContext, cfg: LoopConfig): Prom
|
|
|
2515
2535
|
maxIterations: cfg.maxIterations,
|
|
2516
2536
|
plateauWindow: cfg.plateauWindow,
|
|
2517
2537
|
stallCount: 0,
|
|
2518
|
-
bestValue: baseline,
|
|
2519
|
-
lastValue: baseline,
|
|
2538
|
+
bestValue: cfg.deferBaseline ? null : baseline,
|
|
2539
|
+
lastValue: cfg.deferBaseline ? null : baseline,
|
|
2540
|
+
kind: cfg.kind,
|
|
2520
2541
|
active: true,
|
|
2521
2542
|
history: [],
|
|
2522
2543
|
startedAt: nowIso(),
|
|
@@ -2535,7 +2556,7 @@ async function startLoopFromConfig(ctx: ExtensionContext, cfg: LoopConfig): Prom
|
|
|
2535
2556
|
metricless
|
|
2536
2557
|
? `Loop started (metricless spec loop — NO plateau stop): ${cfg.target.slice(0, 60)}\nEnds only at ${cfg.maxIterations > 0 ? `max ${cfg.maxIterations} iterations` : "no iteration cap"}${cfg.timeLimitHours ? ` · ${cfg.timeLimitHours}h` : ""}${cfg.tokenBudget ? ` · ${cfg.tokenBudget.toLocaleString()} tokens` : ""} · /loop stop. Every iteration must make ONE real, inspectable change — cosmetic churn is the doorknob failure.` +
|
|
2537
2558
|
(branchName ? `\nbranch mode: committing each iteration to ${branchName}` : "")
|
|
2538
|
-
: `Loop started: ${cfg.target.slice(0, 60)}\nBaseline: ${baseline ?? "(forced without a number — first turn must produce one)"} · direction ${cfg.direction} · window ${cfg.plateauWindow} · ${cfg.maxIterations > 0 ? `max ${cfg.maxIterations}` : "no iteration cap"}` +
|
|
2559
|
+
: `Loop started: ${cfg.target.slice(0, 60)}\nBaseline: ${cfg.deferBaseline ? "deferred — the first real measurement seeds it" : (baseline ?? "(forced without a number — first turn must produce one)")} · direction ${cfg.direction} · window ${cfg.plateauWindow} · ${cfg.maxIterations > 0 ? `max ${cfg.maxIterations}` : "no iteration cap"}` +
|
|
2539
2560
|
(branchName ? `\nbranch mode: committing improvements to ${branchName}` : ""),
|
|
2540
2561
|
"info",
|
|
2541
2562
|
);
|
|
@@ -2694,6 +2715,11 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
2694
2715
|
maxIterations: 0,
|
|
2695
2716
|
branch: false,
|
|
2696
2717
|
force: false,
|
|
2718
|
+
// v0.29.10: the audit loop's metric is CREATED by iteration 1 —
|
|
2719
|
+
// seeding best from the pre-discovery 0 stalls every iteration and
|
|
2720
|
+
// plateau-stops mid-work at the window. Defer the baseline.
|
|
2721
|
+
deferBaseline: true,
|
|
2722
|
+
kind: "audit",
|
|
2697
2723
|
});
|
|
2698
2724
|
return;
|
|
2699
2725
|
}
|
|
@@ -5302,6 +5328,16 @@ export default function (pi: ExtensionAPI): void {
|
|
|
5302
5328
|
) {
|
|
5303
5329
|
ctx.ui.notify("Auto-resume fired (event: session start). Continue working.", "info");
|
|
5304
5330
|
}
|
|
5331
|
+
// v0.29.10: reseed live/held audit loops stuck on the degenerate
|
|
5332
|
+
// pre-discovery baseline-0 — best can never go below 0, so every
|
|
5333
|
+
// iteration stalls and the prompt cries REGRESSED on real progress
|
|
5334
|
+
// (junk-runner ran pinned like this). Null best = the next measurement
|
|
5335
|
+
// becomes the honest baseline; stall streak resets with it.
|
|
5336
|
+
if (state.loop && state.loop.measureCmd?.includes("audit-loop/findings.md") && state.loop.bestValue === 0) {
|
|
5337
|
+
state.loop = { ...state.loop, bestValue: null, stallCount: 0, kind: state.loop.kind ?? "audit" };
|
|
5338
|
+
persistState(ctx);
|
|
5339
|
+
appendLedger(ctx.cwd, "audit_loop_baseline_reseeded", { from: 0 });
|
|
5340
|
+
}
|
|
5305
5341
|
if (isLoopActive()) {
|
|
5306
5342
|
const l = state.loop!;
|
|
5307
5343
|
if (autoResume) {
|
|
@@ -5540,17 +5576,40 @@ export default function (pi: ExtensionAPI): void {
|
|
|
5540
5576
|
// v0.29.1: brake-cycle CAP. The v0.28.25 ladder slows the thrash
|
|
5541
5577
|
// (1m→16m) but never STOPS it — junk-runner/hellhunter/pully each
|
|
5542
5578
|
// burned 4+ pause↔retry cycles against provider windows that last
|
|
5543
|
-
// hours. After 6 consecutive brakes: park
|
|
5579
|
+
// hours. After 6 consecutive brakes: park. v0.29.9: the park keeps
|
|
5580
|
+
// probing at the top of each hour (clock-aligned window resets).
|
|
5544
5581
|
if (errorBrakeStreak >= 6) {
|
|
5582
|
+
// v0.29.9: park — but keep probing at the top of each hour
|
|
5583
|
+
// (user: "simply adding an hourly retry … just to pick up work
|
|
5584
|
+
// faster assuming the retry expired"). Coding-plan rate-limit
|
|
5585
|
+
// windows typically expire on clock-hour boundaries, so a probe
|
|
5586
|
+
// scheduled for :01 catches the reset within seconds. One dunk
|
|
5587
|
+
// per hour, free (429s are rejected pre-billing); a successful
|
|
5588
|
+
// probe resets the whole error cycle. If the wall is something
|
|
5589
|
+
// else (auth, outage), the hourly probe is a harmless failed
|
|
5590
|
+
// resume attempt that re-parks via the same brake.
|
|
5545
5591
|
updateGoal({
|
|
5546
5592
|
status: "paused",
|
|
5547
5593
|
pauseKind: "error",
|
|
5548
5594
|
pauseReason: `${reason} — 6 error-brakes in a row; the provider has been erroring for an extended window`,
|
|
5549
|
-
pauseSuggestedAction: "
|
|
5595
|
+
pauseSuggestedAction: "Probing at the top of each hour — rate-limit windows typically expire on clock-hour boundaries. /goal resume retries now.",
|
|
5550
5596
|
}, ctx);
|
|
5551
|
-
ctx.ui.notify(`${goalNoun()} parked: ${reason} — 6 brakes in a row
|
|
5552
|
-
notifyExternal(ctx, `${goalNoun()} parked: provider erroring across 6 error-brake cycles.`);
|
|
5597
|
+
ctx.ui.notify(`${goalNoun()} parked: ${reason} — 6 brakes in a row. Hourly top-of-hour probes will pick work back up when the window opens; /goal resume retries now.`, "warning");
|
|
5598
|
+
notifyExternal(ctx, `${goalNoun()} parked: provider erroring across 6 error-brake cycles — hourly top-of-hour probes scheduled.`);
|
|
5553
5599
|
appendLedger(ctx.cwd, "error_brake_capped", { streak: errorBrakeStreak, reason });
|
|
5600
|
+
const probeMs = msUntilNextHourBoundary(Date.now());
|
|
5601
|
+
scheduleQuotaRetry(ctx, probeMs / 1000, reason, () => {
|
|
5602
|
+
// Re-check: only probe if STILL parked by the error-brake cap —
|
|
5603
|
+
// a user pause/resume/cancel meanwhile is never stomped.
|
|
5604
|
+
if (state.goal && state.goal.status === "paused" && state.goal.pauseKind === "error"
|
|
5605
|
+
&& (state.goal.pauseReason ?? "").includes("error-brakes in a row")) {
|
|
5606
|
+
appendLedger(ctx.cwd, "hourly_rate_probe", { goalId: state.goal.id, streak: errorBrakeStreak });
|
|
5607
|
+
updateGoal({ status: "active" }, ctx);
|
|
5608
|
+
appendLedger(ctx.cwd, "goal_resumed", { via: "hourly-rate-probe" });
|
|
5609
|
+
ctx.ui.notify("Hourly probe: resuming (rate-limit windows typically expire at the top of the hour).", "info");
|
|
5610
|
+
scheduleContinuation(ctx, true);
|
|
5611
|
+
}
|
|
5612
|
+
}, "Hourly rate-limit probe");
|
|
5554
5613
|
return;
|
|
5555
5614
|
}
|
|
5556
5615
|
// v0.28.25: the cooldown escalates per CONSECUTIVE brake — a fleet-wide
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-goal-list-loop-audit",
|
|
3
|
-
"version": "0.29.
|
|
3
|
+
"version": "0.29.10",
|
|
4
4
|
"description": "Goal. Loop. Audit. Done. \u2014 a pi-coding-agent extension that supervises long-running work, with isolated auditor on each completion. Beat bamboozling by design: the auditor runs in a fresh session with no extensions, no skills, no editor \u2014 only the read tools needed to verify your goal.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "dracon",
|