pi-goal-list-loop-audit 0.29.12 → 0.29.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -404,14 +404,16 @@ export function respecTarget(specName: string): string {
404
404
  export const AUDIT_FINDINGS_REL = ".pi-glla/audit-loop/findings.md";
405
405
 
406
406
  /**
407
- * The audit-loop measure command: count open findings. Prints exactly one
408
- * number in every file state (missing file / zero matches → 0). This is
409
- * what respec (metricless) and the reviewer cascade (no termination) both
410
- * lacked: an honest metric the plateau stop can believe — audits that stop
411
- * surfacing new findings = the well is dry = the loop ends.
412
- */
407
+ * The audit-loop measure command: count CLOSED findings. Prints exactly one
408
+ * number in every file state (missing file / zero matches → 0). v0.29.14:
409
+ * flipped from open-count/min — the open count PUNISHES DISCOVERY (a fresh
410
+ * audit finding 11 real issues read as a regression: endless-td-style iter
411
+ * 6 went 27→38→37 and nearly plateau-stopped mid-work). The closed count
412
+ * is monotonic under the honesty law (a checked box requires a fix commit):
413
+ * discovery alone doesn't move it, landing fixes does — so the plateau
414
+ * stop fires only when NO FIXES LAND for the window: the honest dry well. */
413
415
  export function auditMeasureCmd(): string {
414
- return `c=$(grep -cE '^- \\[ \\]' ${AUDIT_FINDINGS_REL} 2>/dev/null); echo \${c:-0}`;
416
+ return `c=$(grep -cE '^- \\[[xX]\\]' ${AUDIT_FINDINGS_REL} 2>/dev/null); echo \${c:-0}`;
415
417
  }
416
418
 
417
419
  /**
@@ -425,7 +427,7 @@ export function auditMeasureCmd(): string {
425
427
  * identity, on the current branch — no invented identities or branches).
426
428
  */
427
429
  export function auditTarget(): string {
428
- return `Audit the project for real problems and fix them, iteration by iteration. Every iteration: (1) run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real issues: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding as one checkbox line "- [ ] SEVERITY: short description (file:line)" to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed). (3) Fix the highest-severity OPEN finding(s) — real fixes, committed — then check the box: "- [x] … — fixed in <commit>". (4) Honesty law: never fabricate findings to look busy; never mark a finding fixed without the fix commit existing. When a full audit pass surfaces nothing new AND no open findings remain, say so plainly — the orchestrator counts open findings every iteration and the plateau stop ends the loop when the well is dry.`;
430
+ return `Audit the project for real problems and fix them, iteration by iteration. Every iteration: (1) run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real issues: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding as one checkbox line "- [ ] SEVERITY: short description (file:line)" to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed). (3) Fix the highest-severity OPEN finding(s) — real fixes, committed — then check the box: "- [x] … — fixed in <commit>". (4) Honesty law: never fabricate findings to look busy; never mark a finding fixed without the fix commit existing. The orchestrator counts CLOSED findings every iteration (direction=max): discovery alone does not move the metric — landing fixes does. When a full audit pass surfaces nothing new AND no open findings remain, say so plainly — the plateau stop ends the loop when the well is dry.`;
429
431
  }
430
432
 
431
433
  // ---- /goal audit-project (v0.29.8) ----
@@ -58,6 +58,11 @@ export interface Settings {
58
58
  /** Consecutive stuck interventions before a loop stops (default 5,
59
59
  * 10 under aggressiveMode). */
60
60
  stuckMaxInterventions?: number;
61
+ /** v0.29.13: on a stale-handle terminal, inject /reload into our own
62
+ * tmux pane (keystroke self-heal; pi walls ctx.reload() behind
63
+ * assertActive). Default true; only acts when TMUX and TMUX_PANE are
64
+ * set — pi outside tmux just gets the manual warning. */
65
+ autoReloadOnStale?: boolean;
61
66
  /** v0.26.1: consecutive heartbeat refires without a real turn before
62
67
  * the goal pauses / loop stops (default 5; 0 = never escalate). */
63
68
  stallEscalationRefires?: number;
@@ -186,6 +191,7 @@ export const SETTINGS_KEYS: Array<keyof Settings> = [
186
191
  "quotaRetryMinutes",
187
192
  "stuckMaxInterventions",
188
193
  "stallEscalationRefires",
194
+ "autoReloadOnStale",
189
195
  "stallShortWords",
190
196
  "stallSimilarityThreshold",
191
197
  "postaudit",
@@ -15,6 +15,7 @@
15
15
  */
16
16
 
17
17
  import * as fs from "node:fs";
18
+ import { exec } from "node:child_process";
18
19
  import * as os from "node:os";
19
20
  import * as path from "node:path";
20
21
 
@@ -229,6 +230,29 @@ function goStaleTerminal(ctx: ExtensionContext, where: string): void {
229
230
  }
230
231
  ctx.ui.notify(`glla: ${guidance}`, "warning");
231
232
  notifyExternal(ctx, `glla: extension api stale — restart pi. (${where})`);
233
+ attemptTmuxAutoReload(ctx, where);
234
+ }
235
+
236
+ /** v0.29.13: automatic recovery — the zombie handle can't call ctx.reload()
237
+ * (pi walls EVERY runtime method behind assertActive), but fs/child_process
238
+ * are extension-side and still work. When pi runs inside tmux (this rig's
239
+ * shape), inject /reload as keystrokes into our own pane: pi rebuilds the
240
+ * extension runtime in place, the fresh instance loads .pi-glla state and
241
+ * holds (autoresume=on resumes for you). Opt out: autoReloadOnStale=false.
242
+ * Best-effort: the manual warning already fired, so failures cost nothing. */
243
+ function attemptTmuxAutoReload(ctx: ExtensionContext, where: string): void {
244
+ try {
245
+ if (loadSettings(ctx.cwd).autoReloadOnStale === false) return;
246
+ const pane = process.env.TMUX_PANE;
247
+ if (!process.env.TMUX || !pane || !/^%\d+$/.test(pane)) return;
248
+ appendLedger(ctx.cwd, "auto_reload_injected", { where, pane });
249
+ ctx.ui.notify("glla: injecting /reload into this pane via tmux — extensions rebuild in place and the fresh instance holds state. /glla resume after (automatic with autoresume=on).", "info");
250
+ // Escape dismisses any overlay; the dead session sits at an empty prompt.
251
+ const cmd = `tmux send-keys -t ${pane} Escape && sleep 0.3 && tmux send-keys -t ${pane} -l '/reload' && tmux send-keys -t ${pane} Enter`;
252
+ exec(cmd, () => { /* best-effort */ });
253
+ } catch {
254
+ /* never let recovery take the warning path down */
255
+ }
232
256
  }
233
257
 
234
258
  /** TEST-ONLY hook (tests/harness): the stale flag is process-terminal in
@@ -2256,7 +2280,7 @@ function sendLoopTurn(): void {
2256
2280
  (loop.direction === "min" ? lastHistValue > prevValue : lastHistValue < prevValue);
2257
2281
  const regressionNote = trueRegression
2258
2282
  ? loop.kind === "audit"
2259
- ? "**The open-findings count went UP last iteration — either the fresh audit pass found new problems, or your fix didn't land. Check findings.md, then keep fixing the highest-severity OPEN items.**"
2283
+ ? "**The closed-findings count went DOWN last iteration — a checked finding was reopened or findings.md was rewritten (both forbidden). Restore the closed entries, then keep fixing the highest-severity OPEN items.**"
2260
2284
  : "**Your last change REGRESSED the metric. Undo it first, then try a different small change.**"
2261
2285
  : "";
2262
2286
  // Strategy rotation (from pi-loop-mode's one good idea): one stall before
@@ -2699,10 +2723,11 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
2699
2723
  // v0.29.0: the project-audit loop (user design: "the looper running
2700
2724
  // audits to see where to progress and what to fix — the thing that
2701
2725
  // fires at the end of goals and lists"). Unlike respec this is a
2702
- // METRIC loop: the orchestrator counts open findings every iteration,
2703
- // direction=min, and the plateau stop is the termination — audits that
2704
- // stop surfacing new findings = the well is dry. User typed the
2705
- // command = the act (same auto-start rule as respec).
2726
+ // METRIC loop: the orchestrator counts CLOSED findings every iteration,
2727
+ // direction=max (v0.29.14 — open-count/min punished discovery), and the
2728
+ // plateau stop is the termination — no fixes landing for the window =
2729
+ // the well is dry. User typed the command = the act (same auto-start
2730
+ // rule as respec).
2706
2731
  if (state.goal && state.goal.status === "active") {
2707
2732
  ctx.ui.notify("A goal is active — /goal cancel or /goal pause it before starting a loop.", "warning");
2708
2733
  return;
@@ -2714,7 +2739,7 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
2714
2739
  await startLoopFromConfig(ctx, {
2715
2740
  target: auditTarget(),
2716
2741
  measureCmd: auditMeasureCmd(),
2717
- direction: "min",
2742
+ direction: "max",
2718
2743
  plateauWindow: LOOP_DEFAULTS.plateauWindow,
2719
2744
  maxIterations: 0,
2720
2745
  branch: false,
@@ -5346,15 +5371,18 @@ export default function (pi: ExtensionAPI): void {
5346
5371
  state.loop = { ...state.loop, stopReason: HELD_ON_RESTORE };
5347
5372
  persistState(ctx);
5348
5373
  }
5349
- // v0.29.10: reseed live/held audit loops stuck on the degenerate
5350
- // pre-discovery baseline-0 — best can never go below 0, so every
5351
- // iteration stalls and the prompt cries REGRESSED on real progress
5352
- // (junk-runner ran pinned like this). Null best = the next measurement
5353
- // becomes the honest baseline; stall streak resets with it.
5354
- if (state.loop && state.loop.measureCmd?.includes("audit-loop/findings.md") && state.loop.bestValue === 0) {
5355
- state.loop = { ...state.loop, bestValue: null, stallCount: 0, kind: state.loop.kind ?? "audit" };
5374
+ // v0.29.14: migrate live/held audit loops off the open-count/min
5375
+ // metric — it punished DISCOVERY (11 new real findings read as a
5376
+ // regression; iter 27→38→37 nearly plateau-stopped mid-work). Flip to
5377
+ // closed-count/max, null the pinned best (the next closed-count
5378
+ // measure becomes the honest baseline), reset the stall streak. This
5379
+ // supersedes the v0.29.10 baseline-0 reseed (every old-measure loop
5380
+ // gets nulled here).
5381
+ if (state.loop && state.loop.measureCmd?.includes("audit-loop/findings.md") && state.loop.measureCmd?.includes("\\[ \\]")
5382
+ && state.loop.direction !== "max") {
5383
+ state.loop = { ...state.loop, measureCmd: auditMeasureCmd(), direction: "max", bestValue: null, stallCount: 0, kind: state.loop.kind ?? "audit" };
5356
5384
  persistState(ctx);
5357
- appendLedger(ctx.cwd, "audit_loop_baseline_reseeded", { from: 0 });
5385
+ appendLedger(ctx.cwd, "audit_loop_metric_migrated", { from: "open-count/min", to: "closed-count/max" });
5358
5386
  }
5359
5387
  if (isLoopActive()) {
5360
5388
  const l = state.loop!;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.29.12",
3
+ "version": "0.29.14",
4
4
  "description": "Goal. Loop. Audit. Done. \u2014 a pi-coding-agent extension that supervises long-running work, with isolated auditor on each completion. Beat bamboozling by design: the auditor runs in a fresh session with no extensions, no skills, no editor \u2014 only the read tools needed to verify your goal.",
5
5
  "license": "MIT",
6
6
  "author": "dracon",