pi-goal-list-loop-audit 0.33.1 → 0.33.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -449,7 +449,7 @@ function loopLines(l: LoopState, now: number, theme?: DisplayTheme, width?: numb
449
449
  lines.push(`├─ ${paint(theme, act.ok ? "success" : "error", act.ok ? "✓" : "✗")} ${act.name}${act.arg ? ` ${paint(theme, "dim", truncate(act.arg, 24))}` : ""}${act.ms > 0 ? ` ${paint(theme, "dim", `(${fmtElapsed(act.ms)})`)}` : ""}`);
450
450
  }
451
451
  const footer = !l.measureCmd
452
- ? "metricless (no plateau) · /loop stop · /loop polish"
452
+ ? "metricless (no plateau) · /loop stop · /loop refine" // v0.33.2: the verb exists now
453
453
  : `${l.kind === "audit" ? "metric: closed findings" : truncate(l.measureCmd, budgetFor(width, 3, 30))} · /loop stop`;
454
454
  lines.push(`└─ ${paint(theme, "dim", footer)}`);
455
455
  if (l.branchName) lines.push(`⎇ ${paint(theme, "muted", truncate(l.branchName, budgetFor(width, 3, 50)))}`);
@@ -10,6 +10,7 @@
10
10
  * self-reports progress.
11
11
  */
12
12
 
13
+ import { createHash } from "node:crypto";
13
14
  import { existsSync, readFileSync, statSync } from "node:fs";
14
15
  import { join } from "node:path";
15
16
 
@@ -110,6 +111,20 @@ export interface LoopState {
110
111
  /** v0.25.1: /loop start toolsamerepeat=N — legacy same-tool-same-result
111
112
  * check window. 0 disables it (multi-signal detector only). */
112
113
  toolSameRepeat?: number;
114
+ /** v0.33.2: respec loops carry their spec file — drift detection
115
+ * (specHash compared per tick), checkbox progress (specChecked →
116
+ * spec_item_progress events), and the refine tool's specText write path. */
117
+ specFile?: string;
118
+ specHash?: string;
119
+ specChecked?: number;
120
+ /** v0.33.2: hypothesis feedback loop — the last turn's HYPOTHESIS line
121
+ * plus the verdict computed against the metric movement, injected into
122
+ * the next iteration's prompt. */
123
+ lastHypothesis?: string;
124
+ hypothesisFeedback?: string;
125
+ /** v0.33.2: /loop refine <text> — the operator's respec suggestion rides
126
+ * the next iteration's prompt; the agent proposes via propose_loop_refine. */
127
+ refineHint?: string;
113
128
  /** v0.25.1: per-iteration progress-signal accumulators for the
114
129
  * multi-signal stuck gate. fileWrites bumps on write/edit tool results;
115
130
  * iterationStartHead/At snapshot when the iteration BEGAN so the tick can
@@ -464,6 +479,37 @@ export function countOpenAuditFindings(cwd: string): number {
464
479
  }
465
480
  }
466
481
 
482
+ /** v0.33.2: the first OPEN finding's text — the reprieve note names what
483
+ * to close, not just how many remain. */
484
+ export function topOpenAuditFinding(cwd: string): string | null {
485
+ try {
486
+ const p = join(cwd, AUDIT_FINDINGS_REL);
487
+ if (!existsSync(p)) return null;
488
+ const line = readFileSync(p, "utf-8").split("\n").find((l) => /^- \[ \]/.test(l));
489
+ return line ? line.replace(/^- \[ \]\s*/, "").trim().slice(0, 120) : null;
490
+ } catch {
491
+ return null;
492
+ }
493
+ }
494
+
495
+ /** v0.33.2: spec drift detection — short sha256 of the spec file. */
496
+ export function specFileHash(p: string): string | null {
497
+ try {
498
+ return createHash("sha256").update(readFileSync(p, "utf-8")).digest("hex").slice(0, 16);
499
+ } catch {
500
+ return null;
501
+ }
502
+ }
503
+
504
+ /** v0.33.2: checked checkbox count in a spec file (spec_item_progress). */
505
+ export function countCheckedSpecItems(p: string): number | null {
506
+ try {
507
+ return readFileSync(p, "utf-8").split("\n").filter((l) => /^- \[x\]/i.test(l)).length;
508
+ } catch {
509
+ return null;
510
+ }
511
+ }
512
+
467
513
  // ---- /goal audit-project (v0.29.8) ----
468
514
 
469
515
  /**
@@ -497,7 +543,7 @@ export const LIST_AUDIT_COLLECT_MARKER = "[LIST-AUDIT-COLLECT]";
497
543
 
498
544
  export function listAuditCollectTarget(focus?: string): string {
499
545
  const scope = focus && focus.trim() ? focus.trim() : "the whole project";
500
- return `${LIST_AUDIT_COLLECT_MARKER} Run ONE project audit pass that COLLECTS work — the follow-up fixes are queued as separate list items, so this pass changes no code. Scope: ${scope}. (1) Run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real problems: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed), classified: "- [ ] FIX: SEVERITY: short description (file:line)" for bugs and polish — and "- [?] DECIDE: short description (what the choice is, what each side costs)" for direction, trade-offs, and scope questions where two reasonable answers exist. (3) Change NOTHING — no fixes, no refactors, no drive-by edits: the orchestrator queues each open FIX finding as its own list item after this pass completes, and each fix lands with its own commit and its own audit. (4) DECIDE findings are listed in the completion report — they are presented to the user, never queued and never silently fixed. (5) Honesty law: never fabricate findings to look busy; if the pass is genuinely clean, say so plainly — an empty findings set is a success, not a failure. Done when: the audit pass is complete and every finding it surfaced is appended to ${AUDIT_FINDINGS_REL} with the right classification (or the report states plainly that nothing was found).`;
546
+ return `${LIST_AUDIT_COLLECT_MARKER} Run ONE project audit pass that COLLECTS work — the follow-up fixes are queued as separate list items, so this pass changes no code. Scope: ${scope}. (1) Run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real problems: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed), classified: "- [ ] FIX: SEVERITY: short description (file:line)" for bugs and polish — and "- [?] DECIDE: short description (what the choice is, what each side costs)" for direction, trade-offs, and scope questions where two reasonable answers exist. (3) Change NOTHING — no fixes, no refactors, no drive-by edits: the orchestrator queues each open FIX finding as its own list item after this pass completes, and each fix lands with its own commit and its own audit. (4) DECIDE findings are appended as "- [?]" lines and NOTHING more — the orchestrator raises them to the user as questions after the pass completes; they are never queued and never silently fixed. (5) Honesty law: never fabricate findings to look busy; if the pass is genuinely clean, say so plainly — an empty findings set is a success, not a failure. Done when: the audit pass is complete and every finding it surfaced is appended to ${AUDIT_FINDINGS_REL} with the right classification (or the report states plainly that nothing was found).`;
501
547
  }
502
548
 
503
549
  /** One parsed open finding from the audit findings file. */
@@ -558,5 +604,5 @@ export const LOOP_AUDIT_MARKER = "iteration by iteration — FIX-FIRST";
558
604
 
559
605
  export function projectAuditTarget(focus?: string): string {
560
606
  const scope = focus && focus.trim() ? focus.trim() : "the whole project";
561
- return `${GOAL_AUDIT_ONESHOT_MARKER}. Scope: ${scope}. (1) Run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real problems: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed), classified: "- [ ] FIX: SEVERITY: short description (file:line)" for bugs and polish — whether to fix these is NOT a decision — and "- [?] DECIDE: short description (what the choice is, what each side costs)" for direction, trade-offs, and scope questions where two reasonable answers exist. (3) Fix every NEW FIX finding from this pass — real fixes, committed with the repo's configured identity on the current branch (no invented identities or branches) — then check the box: "- [x] … — fixed in <commit>". (4) Change NOTHING for DECIDE findings — present them in the completion report instead. (5) Honesty law: never fabricate findings to look busy; never check a box without the fix commit existing; never silently turn a DECIDE into a fix. Done when: the audit pass is complete, every new FIX finding has a fix commit and a checked box in ${AUDIT_FINDINGS_REL}, and every DECIDE finding is listed in the file and presented in the completion report.`;
607
+ return `${GOAL_AUDIT_ONESHOT_MARKER}. Scope: ${scope}. (1) Run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real problems: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed), classified: "- [ ] FIX: SEVERITY: short description (file:line)" for bugs and polish — whether to fix these is NOT a decision — and "- [?] DECIDE: short description (what the choice is, what each side costs)" for direction, trade-offs, and scope questions where two reasonable answers exist. (3) Fix every NEW FIX finding from this pass — real fixes, committed with the repo's configured identity on the current branch (no invented identities or branches) — then check the box: "- [x] … — fixed in <commit>". (4) Change NOTHING for DECIDE findings — RAISE them instead: if any "- [?]" findings exist, present each one to the user with ask_user_question BEFORE calling complete_goal (one question per finding, options from the finding's own two sides plus "Defer"; prose numbered list if ask_user_question is unavailable; Esc = Defer), then record every answer in ${AUDIT_FINDINGS_REL} — replace the "- [?]" line with "- [x] DECIDED: <what was chosen> (<date>)" (or "- [x] DEFERRED") so it stops re-surfacing — and queue any chosen work with list_add. (5) Honesty law: never fabricate findings to look busy; never check a box without the fix commit existing; never silently turn a DECIDE into a fix. Done when: the audit pass is complete, every new FIX finding has a fix commit and a checked box in ${AUDIT_FINDINGS_REL}, and every DECIDE finding has been raised to the user and recorded as DECIDED/DEFERRED (or the report states plainly that none were found).`;
562
608
  }
@@ -190,7 +190,19 @@ export function forwardTransitionPaired(input: LoopStuckInput): boolean {
190
190
  * (new detector only), undefined = REPETITION.toolResultRepeat.
191
191
  */
192
192
  export function isActuallyStuck(input: LoopStuckInput, toolSameRepeat?: number): string | undefined {
193
- if ((input.fileWriteCount ?? 0) > 0) return undefined;
193
+ if ((input.fileWriteCount ?? 0) > 0) {
194
+ // v0.33.2: write-exemption abuse — endless cosmetic edits with a
195
+ // near-identical reply are churn, not progress (the metricless
196
+ // doorknob leak). The stuck ladder's first rung is a prompt note;
197
+ // any genuinely different iteration resets it.
198
+ if (input.previousText && input.assistantText && normalizeForPrint(input.assistantText).length > REPETITION.minSimilarLength) {
199
+ const sim = trigramSimilarity(input.assistantText, input.previousText);
200
+ if (sim >= REPETITION.similarityThreshold) {
201
+ return `cosmetic churn: wrote files but the reply is ~${Math.round(sim * 100)}% identical to the previous iteration`;
202
+ }
203
+ }
204
+ return undefined;
205
+ }
194
206
  if ((input.gitCommitCount ?? 0) > 0) return undefined;
195
207
  if ((input.specItemProgressCount ?? 0) > 0) return undefined;
196
208
  if (forwardTransitionPaired(input)) return undefined;
@@ -169,6 +169,9 @@ import {
169
169
  LOOP_DEFAULTS,
170
170
  resolveSpecFiles,
171
171
  respecTarget,
172
+ topOpenAuditFinding,
173
+ specFileHash,
174
+ countCheckedSpecItems,
172
175
  auditMeasureCmd,
173
176
  auditTarget,
174
177
  AUDIT_PLATEAU_MAX_REPRIEVES,
@@ -1382,10 +1385,24 @@ async function fanOutListAuditFindings(ctx: ExtensionContext): Promise<void> {
1382
1385
  // hundreds of items on a single Confirm.
1383
1386
  const fresh = open.filter((f) => !queuedText.includes(f.text.slice(0, 60))).slice(0, 50);
1384
1387
  const alreadyQueued = open.length - fresh.length;
1388
+ // v0.33.3: DECIDE findings are RAISED TO THE USER as real questions
1389
+ // (hegemon 2026-07-31: a truncated notify left the user typing "decide
1390
+ // what" into the void). The orchestrator can't call ask_user_question —
1391
+ // the agent can — so the full untruncated findings go to the agent as a
1392
+ // steer with the raise + record protocol. Fires BEFORE the queueing
1393
+ // early-returns below: decisions need answers even when nothing new
1394
+ // queued or the fan-out was declined.
1395
+ if (decisions.length > 0) {
1396
+ const decList = decisions.slice(0, 8).map((d, i) => `${i + 1}. ${d.slice(0, 500)}`).join("\n");
1397
+ extensionApi?.sendUserMessage(
1398
+ `[DECIDE FINDINGS — user decisions needed] The audit surfaced ${decisions.length} DECIDE finding(s) — direction calls only the user can make (a decision is not a task, so they were NOT queued):\n${decList}\nRaise them to the user NOW with ask_user_question — one question per finding, options from the finding's own two sides plus "Defer" (prose numbered list if ask_user_question is unavailable; Esc = Defer). Then record every answer in ${AUDIT_FINDINGS_REL}: replace the "- [?]" line with "- [x] DECIDED: <what was chosen> (<date>)" (or "- [x] DEFERRED") so it stops re-surfacing, and queue any chosen work with list_add — do NOT start the work inline.`,
1399
+ { deliverAs: ctx.isIdle() ? "followUp" : "steer" },
1400
+ );
1401
+ appendLedger(ctx.cwd, "list_audit_decisions_raised", { decisions: decisions.length });
1402
+ }
1385
1403
  const decideNote =
1386
1404
  decisions.length > 0
1387
- ? `\n${decisions.length} DECIDE finding(s) need YOU (not queued — a decision is not a task):\n` +
1388
- decisions.slice(0, 10).map((d) => ` ? ${d.slice(0, 110)}`).join("\n")
1405
+ ? ` ${decisions.length} DECIDE finding(s) need YOU — raising them as questions now (not queued — a decision is not a task).`
1389
1406
  : "";
1390
1407
  if (fresh.length === 0) {
1391
1408
  ctx.ui.notify(
@@ -2614,7 +2631,7 @@ async function runGit(ctx: ExtensionContext, args: string[]): Promise<{ ok: bool
2614
2631
  }
2615
2632
  }
2616
2633
 
2617
- function loopPrompt(loop: LoopState, regressionNote: string, strategyNote: string, boundsNote: string, interventionNote = "", variantNote = ""): string {
2634
+ function loopPrompt(loop: LoopState, regressionNote: string, strategyNote: string, boundsNote: string, interventionNote = "", variantNote = "", hypothesisNote = "", refineHintNote = ""): string {
2618
2635
  // v0.23.0: metricless loops get their own prompt — no metric section,
2619
2636
  // anti-doorknob rules instead of anti-gaming rules.
2620
2637
  const metricless = !loop.measureCmd;
@@ -2641,7 +2658,9 @@ function loopPrompt(loop: LoopState, regressionNote: string, strategyNote: strin
2641
2658
  .replace(/\$\{STRATEGY_NOTE\}/g, strategyNote)
2642
2659
  .replace(/\$\{BOUNDS_NOTE\}/g, boundsNote)
2643
2660
  .replace(/\$\{INTERVENTION_NOTE\}/g, interventionNote)
2644
- .replace(/\$\{VARIANT_NOTE\}/g, variantNote);
2661
+ .replace(/\$\{VARIANT_NOTE\}/g, variantNote)
2662
+ .replace(/\$\{HYPOTHESIS_NOTE\}/g, hypothesisNote)
2663
+ .replace(/\$\{REFINE_HINT\}/g, refineHintNote);
2645
2664
  }
2646
2665
 
2647
2666
  function scheduleLoopTick(ctx: ExtensionContext): void {
@@ -2693,7 +2712,13 @@ function sendLoopTurn(): void {
2693
2712
  // Strategy rotation (from pi-loop-mode's one good idea): one stall before
2694
2713
  // the plateau window closes, stop polishing and change approach entirely.
2695
2714
  const strategyNote = loop.stallCount >= loop.plateauWindow - 1 && loop.stallCount > 0
2696
- ? "**You are one stall from a plateau stop. Small tweaks are not working — try a FUNDAMENTALLY different approach: different file, different technique, or revert and rethink the angle of attack.**"
2715
+ ? "**You are one stall from a plateau stop. Small tweaks are not working — try a FUNDAMENTALLY different approach: different file, different technique, or revert and rethink the angle of attack.**" +
2716
+ // v0.33.2: a metric flat AT BEST may mean the spec stopped capturing
2717
+ // "better" — the loop holds the evidence, so it says so (was: the
2718
+ // prompt said "call propose_loop_refine" but the loop never suggested it).
2719
+ (loop.lastValue !== null && loop.lastValue === loop.bestValue
2720
+ ? " **The metric has been flat at best — if the spec no longer captures 'better' (saturated metric, drifted target), call propose_loop_refine.**"
2721
+ : "")
2697
2722
  : "";
2698
2723
  // v0.15.0: arbitrary bounds (never "completion") — surface what's armed.
2699
2724
  // v0.23.0: for metricless loops the bounds are the ONLY stop (no
@@ -2724,12 +2749,19 @@ function sendLoopTurn(): void {
2724
2749
  // v0.24.0: identical prompts invite identical answers — rotate the base
2725
2750
  // instruction (metricless loops; metric loops already vary via values).
2726
2751
  const variantNote = metricless ? continueVariant(loop.iteration) : "";
2752
+ // v0.33.2: one-shot prompt payloads, consumed on use.
2753
+ const hypothesisNote = loop.hypothesisFeedback ?? "";
2754
+ if (hypothesisNote) loop.hypothesisFeedback = undefined;
2755
+ const refineHintNote = loop.refineHint
2756
+ ? `**The operator suggests refining the spec:** ${loop.refineHint} — if the current spec no longer captures "better", call propose_loop_refine (target and/or measureCmd${loop.specFile ? " and/or specText/specAppend" : ""}); if it still stands, say why in one line and keep working.`
2757
+ : "";
2758
+ if (refineHintNote) loop.refineHint = undefined;
2727
2759
  try {
2728
2760
  let loopResync = "";
2729
2761
  if (postCompactResyncPending) { try { loopResync = buildPostCompactResync(); } catch { loopResync = ""; } } // v0.33.1
2730
2762
  extensionApi.sendMessage({
2731
2763
  customType: GOAL_EVENT_ENTRY,
2732
- content: loopResync + loopPrompt(loop, regressionNote, strategyNote, boundsNote, interventionNote, variantNote),
2764
+ content: loopResync + loopPrompt(loop, regressionNote, strategyNote, boundsNote, interventionNote, variantNote, hypothesisNote, refineHintNote),
2733
2765
  display: false,
2734
2766
  }, { triggerTurn: true, deliverAs: "followUp" });
2735
2767
  if (loopResync) postCompactResyncPending = false; // consumed only by a landed send
@@ -2800,6 +2832,23 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
2800
2832
  }
2801
2833
  } catch { /* no ledger yet */ }
2802
2834
  }
2835
+ // v0.33.2: respec spec drift + checkbox progress — hash compared per
2836
+ // tick (external edits ledger spec_updated); newly checked boxes emit
2837
+ // the spec_item_progress signal the stuck gate already consumes (it was
2838
+ // consumed-but-never-emitted until now).
2839
+ if (loop.specFile) {
2840
+ const hash = specFileHash(loop.specFile);
2841
+ if (hash && loop.specHash && loop.specHash !== hash) {
2842
+ appendLedger(ctx.cwd, "spec_updated", { via: "external", iteration: loop.iteration });
2843
+ ctx.ui.notify("Spec file changed mid-loop — drift ledgered (spec_updated).", "info");
2844
+ }
2845
+ if (hash) loop.specHash = hash;
2846
+ const checked = countCheckedSpecItems(loop.specFile);
2847
+ if (checked !== null && loop.specChecked !== undefined && checked > loop.specChecked) {
2848
+ appendLedger(ctx.cwd, "spec_item_progress", { iteration: loop.iteration, newlyChecked: checked - loop.specChecked, totalChecked: checked });
2849
+ }
2850
+ if (checked !== null) loop.specChecked = checked;
2851
+ }
2803
2852
  const iterSignals = {
2804
2853
  fileWrites: loop.iterMetrics?.fileWrites ?? 0,
2805
2854
  gitCommits,
@@ -2839,6 +2888,26 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
2839
2888
  loop.lastStuckReason = undefined;
2840
2889
  }
2841
2890
  let outcome: LoopTickOutcome = metricless ? applyMetriclessTick(loop, nowIso()) : applyMeasurement(loop, value, nowIso());
2891
+ // v0.33.2: close the hypothesis feedback loop — the prediction went into
2892
+ // the ledger; now the VERDICT rides the next iteration's prompt.
2893
+ if (loop.lastHypothesis) {
2894
+ const h = loop.history;
2895
+ const cur = h.length >= 1 ? h[h.length - 1]!.value : null;
2896
+ const prev = h.length >= 2 ? h[h.length - 2]!.value : null;
2897
+ if (metricless || cur === null) {
2898
+ loop.hypothesisFeedback = `Last iteration you predicted: "${loop.lastHypothesis}". ${metricless ? "Metricless loop — no number to verify it against; say honestly whether the prediction landed." : "The measure printed no number — the prediction is unverifiable."}`;
2899
+ } else {
2900
+ const moved = prev === null
2901
+ ? `first measurement ${cur}`
2902
+ : cur === prev
2903
+ ? `flat at ${cur}`
2904
+ : loop.direction === "min"
2905
+ ? (cur < prev ? `improved ${prev} → ${cur}` : `regressed ${prev} → ${cur}`)
2906
+ : (cur > prev ? `improved ${prev} → ${cur}` : `regressed ${prev} → ${cur}`);
2907
+ loop.hypothesisFeedback = `Last iteration you predicted: "${loop.lastHypothesis}". Result: metric ${moved} (best ${loop.bestValue}).`;
2908
+ }
2909
+ }
2910
+ loop.lastHypothesis = hypothesis;
2842
2911
  persistState(ctx);
2843
2912
  appendLedger(ctx.cwd, "loop_measured", {
2844
2913
  iteration: loop.iteration,
@@ -2892,7 +2961,8 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
2892
2961
  loop.stopReason = undefined;
2893
2962
  loop.stallCount = 0;
2894
2963
  loop.auditPlateauReprieves = reprieves;
2895
- loop.auditReprieveNote = `PLATEAU REPRIEVE (${reprieves}/${AUDIT_PLATEAU_MAX_REPRIEVES}): ${open} finding(s) still OPEN in ${AUDIT_FINDINGS_REL} — the plateau stop does not fire while the well isn't dry. Stop hunting and stop narrating: pick the smallest OPEN finding and CLOSE it this iteration (fix commit + checked box). ${AUDIT_PLATEAU_MAX_REPRIEVES - reprieves} reprieve(s) remain.`;
2964
+ const topFinding = topOpenAuditFinding(ctx.cwd); // v0.33.2: name what to close, not just the count
2965
+ loop.auditReprieveNote = `PLATEAU REPRIEVE (${reprieves}/${AUDIT_PLATEAU_MAX_REPRIEVES}): ${open} finding(s) still OPEN in ${AUDIT_FINDINGS_REL} — the plateau stop does not fire while the well isn't dry. Stop hunting and stop narrating: pick the smallest OPEN finding and CLOSE it this iteration (fix commit + checked box).${topFinding ? ` Top open: ${topFinding}` : ""} ${AUDIT_PLATEAU_MAX_REPRIEVES - reprieves} reprieve(s) remain.`;
2896
2966
  persistState(ctx);
2897
2967
  appendLedger(ctx.cwd, "audit_plateau_reprieve", { open, reprieves, best: loop.bestValue });
2898
2968
  ctx.ui.notify(`Audit loop plateau reprieve (${reprieves}/${AUDIT_PLATEAU_MAX_REPRIEVES}): ${open} open findings — the well isn't dry, continuing.`, "info");
@@ -2951,6 +3021,9 @@ interface LoopConfig {
2951
3021
  deferBaseline?: boolean;
2952
3022
  /** v0.29.10: audit loops get audit-flavoured regression wording. */
2953
3023
  kind?: "audit";
3024
+ /** v0.33.2: respec loops carry their spec file (drift detection,
3025
+ * checkbox progress, refine specText writes). */
3026
+ specFile?: string;
2954
3027
  }
2955
3028
 
2956
3029
  /** Shared loop-start path: /loop start AND propose_loop_draft (after Confirm). */
@@ -3017,6 +3090,9 @@ async function startLoopFromConfig(ctx: ExtensionContext, cfg: LoopConfig): Prom
3017
3090
  branchName,
3018
3091
  originalBranch,
3019
3092
  toolSameRepeat: cfg.toolSameRepeat,
3093
+ specFile: cfg.specFile,
3094
+ specHash: cfg.specFile ? specFileHash(cfg.specFile) ?? undefined : undefined,
3095
+ specChecked: cfg.specFile ? countCheckedSpecItems(cfg.specFile) ?? undefined : undefined,
3020
3096
  iterMetrics: { fileWrites: 0, iterationStartAt: nowIso() },
3021
3097
  },
3022
3098
  };
@@ -3135,6 +3211,28 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
3135
3211
 
3136
3212
  // v0.28.14: /loop cancel is a first-class alias — users reached for
3137
3213
  // /goal cancel to kill loops because "cancel" is the verb they know.
3214
+ if (sub === "refine" || sub === "polish") {
3215
+ // v0.33.2: the operator's respec verb. The refine flow stays
3216
+ // agent-proposed + user-confirmed (propose_loop_refine) — this command
3217
+ // queues the operator's suggestion into the next iteration's prompt.
3218
+ // ("polish" accepted as an alias: the widget footer advertised it
3219
+ // before the command existed — now it does.)
3220
+ if (!isLoopActive()) {
3221
+ ctx.ui.notify("No active loop to refine — /loop start first.", "warning");
3222
+ return;
3223
+ }
3224
+ const hint = rest.trim();
3225
+ if (!hint) {
3226
+ ctx.ui.notify("Usage: /loop refine <what the spec should capture better> — the suggestion rides the next iteration's prompt; the agent proposes via propose_loop_refine and you confirm.", "info");
3227
+ return;
3228
+ }
3229
+ state.loop!.refineHint = hint.slice(0, 300);
3230
+ persistState(ctx);
3231
+ appendLedger(ctx.cwd, "loop_refine_hint", { iteration: state.loop!.iteration, hint: state.loop!.refineHint });
3232
+ ctx.ui.notify("Refine hint queued — it rides the next iteration's prompt.", "info");
3233
+ return;
3234
+ }
3235
+
3138
3236
  if (sub === "stop" || sub === "cancel") {
3139
3237
  if (!state.loop) {
3140
3238
  ctx.ui.notify("No loop to stop.", "info");
@@ -3273,6 +3371,7 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
3273
3371
  maxIterations: 0,
3274
3372
  branch: false,
3275
3373
  force: false,
3374
+ specFile: specPath, // v0.33.2
3276
3375
  });
3277
3376
  return;
3278
3377
  }
@@ -4124,12 +4223,14 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4124
4223
  parameters: Type.Object({
4125
4224
  target: Type.Optional(Type.String({ description: "The sharpened target text (omit to keep the current target)" })),
4126
4225
  measureCmd: Type.Optional(Type.String({ description: "The new measure command printing ONE number (omit to keep the current metric)" })),
4226
+ specText: Type.Optional(Type.String({ description: "v0.33.2: full replacement text for the loop's spec file (respec loops only) — the orchestrator owns the write on user confirm" })),
4227
+ specAppend: Type.Optional(Type.String({ description: "v0.33.2: lines to append to the loop's spec file (respec loops only)" })),
4127
4228
  rationale: Type.String({ description: "Why the current spec no longer captures 'better' — shown to the user in the Confirm dialog" }),
4128
4229
  }),
4129
4230
  async execute(_id, params, _signal, _onUpdate, execCtx) {
4130
4231
  const foreign4 = foreignToolGuard(execCtx);
4131
4232
  if (foreign4) return { content: [{ type: "text", text: foreign4 }], details: {} };
4132
- const p = params as { target?: string; measureCmd?: string; rationale: string };
4233
+ const p = params as { target?: string; measureCmd?: string; specText?: string; specAppend?: string; rationale: string };
4133
4234
  const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
4134
4235
  const loop = state.loop;
4135
4236
  if (!loop?.active) {
@@ -4142,8 +4243,12 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4142
4243
  if (!loop.measureCmd && p.measureCmd?.trim()) {
4143
4244
  return { content: [{ type: "text", text: "This loop is metricless — refining it into a measured loop isn't supported. /loop stop, then /loop start with a metric." }], details: {} };
4144
4245
  }
4145
- if (newTarget === loop.target && newMeasure === loop.measureCmd) {
4146
- return { content: [{ type: "text", text: "Refinement proposed no changes — provide a new target, a new measureCmd, or both." }], details: {} };
4246
+ const specChange = (p.specText?.trim() || p.specAppend?.trim()) ? true : false;
4247
+ if (specChange && !loop.specFile) {
4248
+ return { content: [{ type: "text", text: "This loop has no spec file (specText/specAppend apply to /loop respec loops). Refine the target instead." }], details: {} };
4249
+ }
4250
+ if (newTarget === loop.target && newMeasure === loop.measureCmd && !specChange) {
4251
+ return { content: [{ type: "text", text: "Refinement proposed no changes — provide a new target, a new measureCmd, a spec change, or any combination." }], details: {} };
4147
4252
  }
4148
4253
  // Measure change → orchestrator test-runs the new command first.
4149
4254
  let newBaseline: number | null = null;
@@ -4174,7 +4279,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4174
4279
  confirmed = (await confirmDraft(
4175
4280
  liveCtx,
4176
4281
  "Confirm loop spec refinement",
4177
- `Rationale: ${p.rationale}\n\nTarget:\n old: ${loop.target.slice(0, 120)}\n new: ${newTarget.slice(0, 120)}\n\nMeasure:\n old: ${loop.measureCmd}\n new: ${newMeasure}${newMeasure !== loop.measureCmd ? `\n test-run: ${testOutput.slice(0, 120)} → ${newBaseline}` : ""}\n\nThe loop keeps running against the refined spec (iteration ${loop.iteration} so far). Apply?`,
4282
+ `Rationale: ${p.rationale}\n\nTarget:\n old: ${loop.target.slice(0, 120)}\n new: ${newTarget.slice(0, 120)}\n\nMeasure:\n old: ${loop.measureCmd}\n new: ${newMeasure}${newMeasure !== loop.measureCmd ? `\n test-run: ${testOutput.slice(0, 120)} → ${newBaseline}` : ""}${specChange ? `\n\nSpec file (${loop.specFile}):\n ${p.specText?.trim() ? `REPLACE with ${p.specText!.trim().length} chars` : ""}${p.specText?.trim() && p.specAppend?.trim() ? " + " : ""}${p.specAppend?.trim() ? `APPEND: ${p.specAppend!.trim().slice(0, 120)}` : ""}` : ""}\n\nThe loop keeps running against the refined spec (iteration ${loop.iteration} so far). Apply?`,
4178
4283
  )) === "yes";
4179
4284
  } catch {
4180
4285
  confirmed = false;
@@ -4191,9 +4296,22 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4191
4296
  oldMeasureCmd: loop.measureCmd ?? "",
4192
4297
  newMeasureCmd: newMeasure,
4193
4298
  }, newBaseline);
4299
+ // v0.33.2: the orchestrator owns the spec write (honesty stays
4300
+ // inspectable — the agent never edits the spec it's judged against
4301
+ // outside a confirmed refine).
4302
+ if (specChange && loop.specFile) {
4303
+ try {
4304
+ if (p.specText?.trim()) fs.writeFileSync(loop.specFile, p.specText.trim() + "\n");
4305
+ if (p.specAppend?.trim()) fs.appendFileSync(loop.specFile, (p.specText?.trim() ? "" : "\n") + p.specAppend.trim() + "\n");
4306
+ loop.specHash = specFileHash(loop.specFile) ?? undefined;
4307
+ appendLedger(liveCtx.cwd, "spec_updated", { via: "refine", iteration: loop.iteration, replaced: Boolean(p.specText?.trim()), appended: Boolean(p.specAppend?.trim()) });
4308
+ } catch (e) {
4309
+ return { content: [{ type: "text", text: `Spec file write failed: ${String(e).slice(0, 200)}. The target/measure refinement was applied; re-propose the spec change.` }], details: {} };
4310
+ }
4311
+ }
4194
4312
  persistState(liveCtx);
4195
- appendLedger(liveCtx.cwd, "loop_refined", { iteration: loop.iteration, newTarget, newMeasureCmd: newMeasure, newBaseline });
4196
- liveCtx.ui.notify(`Loop spec refined at iteration ${loop.iteration}.${newBaseline !== null ? ` New baseline: ${newBaseline}.` : ""}`, "info");
4313
+ appendLedger(liveCtx.cwd, "loop_refined", { iteration: loop.iteration, newTarget, newMeasureCmd: newMeasure, newBaseline, specChanged: specChange || undefined });
4314
+ liveCtx.ui.notify(`Loop spec refined at iteration ${loop.iteration}.${newBaseline !== null ? ` New baseline: ${newBaseline}.` : ""}${specChange ? " Spec file updated." : ""}`, "info");
4197
4315
  return { content: [{ type: "text", text: "Refinement confirmed and applied. Continue improving against the NEW spec — one small change per turn." }], details: {} };
4198
4316
  },
4199
4317
  }));
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.33.1",
3
+ "version": "0.33.3",
4
4
  "description": "Goal. Loop. Audit. Done. \u2014 a pi-coding-agent extension that supervises long-running work, with isolated auditor on each completion. Beat bamboozling by design: the auditor runs in a fresh session with no extensions, no skills, no editor \u2014 only the read tools needed to verify your goal.",
5
5
  "license": "MIT",
6
6
  "author": "dracon",
@@ -31,6 +31,8 @@ fails, retry with a different approach — just continue, don't stall the loop
31
31
  asking permission. You remain the single writer: apply the edit yourself.
32
32
 
33
33
  ${INTERVENTION_NOTE}
34
+ ${HYPOTHESIS_NOTE}
35
+ ${REFINE_HINT}
34
36
  ${REGRESSION_NOTE}
35
37
  ${STRATEGY_NOTE}
36
38
 
@@ -35,6 +35,8 @@ fails, retry with a different approach — just continue, don't stall the loop
35
35
  asking permission. You remain the single writer: apply the edit yourself.
36
36
 
37
37
  ${INTERVENTION_NOTE}
38
+ ${HYPOTHESIS_NOTE}
39
+ ${REFINE_HINT}
38
40
  ${REGRESSION_NOTE}
39
41
  ${STRATEGY_NOTE}
40
42