pi-goal-list-loop-audit 0.33.1 → 0.33.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/extensions/goal-loop-display.ts +1 -1
- package/extensions/goal-loop-forever.ts +48 -2
- package/extensions/goal-loop-repetition.ts +13 -1
- package/extensions/loops/goal.ts +131 -13
- package/package.json +1 -1
- package/prompts/goal-loop-forever-metricless.md +2 -0
- package/prompts/goal-loop-forever.md +2 -0
|
@@ -449,7 +449,7 @@ function loopLines(l: LoopState, now: number, theme?: DisplayTheme, width?: numb
|
|
|
449
449
|
lines.push(`├─ ${paint(theme, act.ok ? "success" : "error", act.ok ? "✓" : "✗")} ${act.name}${act.arg ? ` ${paint(theme, "dim", truncate(act.arg, 24))}` : ""}${act.ms > 0 ? ` ${paint(theme, "dim", `(${fmtElapsed(act.ms)})`)}` : ""}`);
|
|
450
450
|
}
|
|
451
451
|
const footer = !l.measureCmd
|
|
452
|
-
? "metricless (no plateau) · /loop stop · /loop
|
|
452
|
+
? "metricless (no plateau) · /loop stop · /loop refine" // v0.33.2: the verb exists now
|
|
453
453
|
: `${l.kind === "audit" ? "metric: closed findings" : truncate(l.measureCmd, budgetFor(width, 3, 30))} · /loop stop`;
|
|
454
454
|
lines.push(`└─ ${paint(theme, "dim", footer)}`);
|
|
455
455
|
if (l.branchName) lines.push(`⎇ ${paint(theme, "muted", truncate(l.branchName, budgetFor(width, 3, 50)))}`);
|
|
@@ -10,6 +10,7 @@
|
|
|
10
10
|
* self-reports progress.
|
|
11
11
|
*/
|
|
12
12
|
|
|
13
|
+
import { createHash } from "node:crypto";
|
|
13
14
|
import { existsSync, readFileSync, statSync } from "node:fs";
|
|
14
15
|
import { join } from "node:path";
|
|
15
16
|
|
|
@@ -110,6 +111,20 @@ export interface LoopState {
|
|
|
110
111
|
/** v0.25.1: /loop start toolsamerepeat=N — legacy same-tool-same-result
|
|
111
112
|
* check window. 0 disables it (multi-signal detector only). */
|
|
112
113
|
toolSameRepeat?: number;
|
|
114
|
+
/** v0.33.2: respec loops carry their spec file — drift detection
|
|
115
|
+
* (specHash compared per tick), checkbox progress (specChecked →
|
|
116
|
+
* spec_item_progress events), and the refine tool's specText write path. */
|
|
117
|
+
specFile?: string;
|
|
118
|
+
specHash?: string;
|
|
119
|
+
specChecked?: number;
|
|
120
|
+
/** v0.33.2: hypothesis feedback loop — the last turn's HYPOTHESIS line
|
|
121
|
+
* plus the verdict computed against the metric movement, injected into
|
|
122
|
+
* the next iteration's prompt. */
|
|
123
|
+
lastHypothesis?: string;
|
|
124
|
+
hypothesisFeedback?: string;
|
|
125
|
+
/** v0.33.2: /loop refine <text> — the operator's respec suggestion rides
|
|
126
|
+
* the next iteration's prompt; the agent proposes via propose_loop_refine. */
|
|
127
|
+
refineHint?: string;
|
|
113
128
|
/** v0.25.1: per-iteration progress-signal accumulators for the
|
|
114
129
|
* multi-signal stuck gate. fileWrites bumps on write/edit tool results;
|
|
115
130
|
* iterationStartHead/At snapshot when the iteration BEGAN so the tick can
|
|
@@ -464,6 +479,37 @@ export function countOpenAuditFindings(cwd: string): number {
|
|
|
464
479
|
}
|
|
465
480
|
}
|
|
466
481
|
|
|
482
|
+
/** v0.33.2: the first OPEN finding's text — the reprieve note names what
|
|
483
|
+
* to close, not just how many remain. */
|
|
484
|
+
export function topOpenAuditFinding(cwd: string): string | null {
|
|
485
|
+
try {
|
|
486
|
+
const p = join(cwd, AUDIT_FINDINGS_REL);
|
|
487
|
+
if (!existsSync(p)) return null;
|
|
488
|
+
const line = readFileSync(p, "utf-8").split("\n").find((l) => /^- \[ \]/.test(l));
|
|
489
|
+
return line ? line.replace(/^- \[ \]\s*/, "").trim().slice(0, 120) : null;
|
|
490
|
+
} catch {
|
|
491
|
+
return null;
|
|
492
|
+
}
|
|
493
|
+
}
|
|
494
|
+
|
|
495
|
+
/** v0.33.2: spec drift detection — short sha256 of the spec file. */
|
|
496
|
+
export function specFileHash(p: string): string | null {
|
|
497
|
+
try {
|
|
498
|
+
return createHash("sha256").update(readFileSync(p, "utf-8")).digest("hex").slice(0, 16);
|
|
499
|
+
} catch {
|
|
500
|
+
return null;
|
|
501
|
+
}
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
/** v0.33.2: checked checkbox count in a spec file (spec_item_progress). */
|
|
505
|
+
export function countCheckedSpecItems(p: string): number | null {
|
|
506
|
+
try {
|
|
507
|
+
return readFileSync(p, "utf-8").split("\n").filter((l) => /^- \[x\]/i.test(l)).length;
|
|
508
|
+
} catch {
|
|
509
|
+
return null;
|
|
510
|
+
}
|
|
511
|
+
}
|
|
512
|
+
|
|
467
513
|
// ---- /goal audit-project (v0.29.8) ----
|
|
468
514
|
|
|
469
515
|
/**
|
|
@@ -497,7 +543,7 @@ export const LIST_AUDIT_COLLECT_MARKER = "[LIST-AUDIT-COLLECT]";
|
|
|
497
543
|
|
|
498
544
|
export function listAuditCollectTarget(focus?: string): string {
|
|
499
545
|
const scope = focus && focus.trim() ? focus.trim() : "the whole project";
|
|
500
|
-
return `${LIST_AUDIT_COLLECT_MARKER} Run ONE project audit pass that COLLECTS work — the follow-up fixes are queued as separate list items, so this pass changes no code. Scope: ${scope}. (1) Run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real problems: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed), classified: "- [ ] FIX: SEVERITY: short description (file:line)" for bugs and polish — and "- [?] DECIDE: short description (what the choice is, what each side costs)" for direction, trade-offs, and scope questions where two reasonable answers exist. (3) Change NOTHING — no fixes, no refactors, no drive-by edits: the orchestrator queues each open FIX finding as its own list item after this pass completes, and each fix lands with its own commit and its own audit. (4) DECIDE findings are
|
|
546
|
+
return `${LIST_AUDIT_COLLECT_MARKER} Run ONE project audit pass that COLLECTS work — the follow-up fixes are queued as separate list items, so this pass changes no code. Scope: ${scope}. (1) Run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real problems: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed), classified: "- [ ] FIX: SEVERITY: short description (file:line)" for bugs and polish — and "- [?] DECIDE: short description (what the choice is, what each side costs)" for direction, trade-offs, and scope questions where two reasonable answers exist. (3) Change NOTHING — no fixes, no refactors, no drive-by edits: the orchestrator queues each open FIX finding as its own list item after this pass completes, and each fix lands with its own commit and its own audit. (4) DECIDE findings are appended as "- [?]" lines and NOTHING more — the orchestrator raises them to the user as questions after the pass completes; they are never queued and never silently fixed. (5) Honesty law: never fabricate findings to look busy; if the pass is genuinely clean, say so plainly — an empty findings set is a success, not a failure. Done when: the audit pass is complete and every finding it surfaced is appended to ${AUDIT_FINDINGS_REL} with the right classification (or the report states plainly that nothing was found).`;
|
|
501
547
|
}
|
|
502
548
|
|
|
503
549
|
/** One parsed open finding from the audit findings file. */
|
|
@@ -558,5 +604,5 @@ export const LOOP_AUDIT_MARKER = "iteration by iteration — FIX-FIRST";
|
|
|
558
604
|
|
|
559
605
|
export function projectAuditTarget(focus?: string): string {
|
|
560
606
|
const scope = focus && focus.trim() ? focus.trim() : "the whole project";
|
|
561
|
-
return `${GOAL_AUDIT_ONESHOT_MARKER}. Scope: ${scope}. (1) Run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real problems: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed), classified: "- [ ] FIX: SEVERITY: short description (file:line)" for bugs and polish — whether to fix these is NOT a decision — and "- [?] DECIDE: short description (what the choice is, what each side costs)" for direction, trade-offs, and scope questions where two reasonable answers exist. (3) Fix every NEW FIX finding from this pass — real fixes, committed with the repo's configured identity on the current branch (no invented identities or branches) — then check the box: "- [x] … — fixed in <commit>". (4) Change NOTHING for DECIDE findings —
|
|
607
|
+
return `${GOAL_AUDIT_ONESHOT_MARKER}. Scope: ${scope}. (1) Run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real problems: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed), classified: "- [ ] FIX: SEVERITY: short description (file:line)" for bugs and polish — whether to fix these is NOT a decision — and "- [?] DECIDE: short description (what the choice is, what each side costs)" for direction, trade-offs, and scope questions where two reasonable answers exist. (3) Fix every NEW FIX finding from this pass — real fixes, committed with the repo's configured identity on the current branch (no invented identities or branches) — then check the box: "- [x] … — fixed in <commit>". (4) Change NOTHING for DECIDE findings — RAISE them instead: if any "- [?]" findings exist, present each one to the user with ask_user_question BEFORE calling complete_goal (one question per finding, options from the finding's own two sides plus "Defer"; prose numbered list if ask_user_question is unavailable; Esc = Defer), then record every answer in ${AUDIT_FINDINGS_REL} — replace the "- [?]" line with "- [x] DECIDED: <what was chosen> (<date>)" (or "- [x] DEFERRED") so it stops re-surfacing — and queue any chosen work with list_add. (5) Honesty law: never fabricate findings to look busy; never check a box without the fix commit existing; never silently turn a DECIDE into a fix. Done when: the audit pass is complete, every new FIX finding has a fix commit and a checked box in ${AUDIT_FINDINGS_REL}, and every DECIDE finding has been raised to the user and recorded as DECIDED/DEFERRED (or the report states plainly that none were found).`;
|
|
562
608
|
}
|
|
@@ -190,7 +190,19 @@ export function forwardTransitionPaired(input: LoopStuckInput): boolean {
|
|
|
190
190
|
* (new detector only), undefined = REPETITION.toolResultRepeat.
|
|
191
191
|
*/
|
|
192
192
|
export function isActuallyStuck(input: LoopStuckInput, toolSameRepeat?: number): string | undefined {
|
|
193
|
-
if ((input.fileWriteCount ?? 0) > 0)
|
|
193
|
+
if ((input.fileWriteCount ?? 0) > 0) {
|
|
194
|
+
// v0.33.2: write-exemption abuse — endless cosmetic edits with a
|
|
195
|
+
// near-identical reply are churn, not progress (the metricless
|
|
196
|
+
// doorknob leak). The stuck ladder's first rung is a prompt note;
|
|
197
|
+
// any genuinely different iteration resets it.
|
|
198
|
+
if (input.previousText && input.assistantText && normalizeForPrint(input.assistantText).length > REPETITION.minSimilarLength) {
|
|
199
|
+
const sim = trigramSimilarity(input.assistantText, input.previousText);
|
|
200
|
+
if (sim >= REPETITION.similarityThreshold) {
|
|
201
|
+
return `cosmetic churn: wrote files but the reply is ~${Math.round(sim * 100)}% identical to the previous iteration`;
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
return undefined;
|
|
205
|
+
}
|
|
194
206
|
if ((input.gitCommitCount ?? 0) > 0) return undefined;
|
|
195
207
|
if ((input.specItemProgressCount ?? 0) > 0) return undefined;
|
|
196
208
|
if (forwardTransitionPaired(input)) return undefined;
|
package/extensions/loops/goal.ts
CHANGED
|
@@ -169,6 +169,9 @@ import {
|
|
|
169
169
|
LOOP_DEFAULTS,
|
|
170
170
|
resolveSpecFiles,
|
|
171
171
|
respecTarget,
|
|
172
|
+
topOpenAuditFinding,
|
|
173
|
+
specFileHash,
|
|
174
|
+
countCheckedSpecItems,
|
|
172
175
|
auditMeasureCmd,
|
|
173
176
|
auditTarget,
|
|
174
177
|
AUDIT_PLATEAU_MAX_REPRIEVES,
|
|
@@ -1382,10 +1385,24 @@ async function fanOutListAuditFindings(ctx: ExtensionContext): Promise<void> {
|
|
|
1382
1385
|
// hundreds of items on a single Confirm.
|
|
1383
1386
|
const fresh = open.filter((f) => !queuedText.includes(f.text.slice(0, 60))).slice(0, 50);
|
|
1384
1387
|
const alreadyQueued = open.length - fresh.length;
|
|
1388
|
+
// v0.33.3: DECIDE findings are RAISED TO THE USER as real questions
|
|
1389
|
+
// (hegemon 2026-07-31: a truncated notify left the user typing "decide
|
|
1390
|
+
// what" into the void). The orchestrator can't call ask_user_question —
|
|
1391
|
+
// the agent can — so the full untruncated findings go to the agent as a
|
|
1392
|
+
// steer with the raise + record protocol. Fires BEFORE the queueing
|
|
1393
|
+
// early-returns below: decisions need answers even when nothing new
|
|
1394
|
+
// queued or the fan-out was declined.
|
|
1395
|
+
if (decisions.length > 0) {
|
|
1396
|
+
const decList = decisions.slice(0, 8).map((d, i) => `${i + 1}. ${d.slice(0, 500)}`).join("\n");
|
|
1397
|
+
extensionApi?.sendUserMessage(
|
|
1398
|
+
`[DECIDE FINDINGS — user decisions needed] The audit surfaced ${decisions.length} DECIDE finding(s) — direction calls only the user can make (a decision is not a task, so they were NOT queued):\n${decList}\nRaise them to the user NOW with ask_user_question — one question per finding, options from the finding's own two sides plus "Defer" (prose numbered list if ask_user_question is unavailable; Esc = Defer). Then record every answer in ${AUDIT_FINDINGS_REL}: replace the "- [?]" line with "- [x] DECIDED: <what was chosen> (<date>)" (or "- [x] DEFERRED") so it stops re-surfacing, and queue any chosen work with list_add — do NOT start the work inline.`,
|
|
1399
|
+
{ deliverAs: ctx.isIdle() ? "followUp" : "steer" },
|
|
1400
|
+
);
|
|
1401
|
+
appendLedger(ctx.cwd, "list_audit_decisions_raised", { decisions: decisions.length });
|
|
1402
|
+
}
|
|
1385
1403
|
const decideNote =
|
|
1386
1404
|
decisions.length > 0
|
|
1387
|
-
?
|
|
1388
|
-
decisions.slice(0, 10).map((d) => ` ? ${d.slice(0, 110)}`).join("\n")
|
|
1405
|
+
? ` ${decisions.length} DECIDE finding(s) need YOU — raising them as questions now (not queued — a decision is not a task).`
|
|
1389
1406
|
: "";
|
|
1390
1407
|
if (fresh.length === 0) {
|
|
1391
1408
|
ctx.ui.notify(
|
|
@@ -2614,7 +2631,7 @@ async function runGit(ctx: ExtensionContext, args: string[]): Promise<{ ok: bool
|
|
|
2614
2631
|
}
|
|
2615
2632
|
}
|
|
2616
2633
|
|
|
2617
|
-
function loopPrompt(loop: LoopState, regressionNote: string, strategyNote: string, boundsNote: string, interventionNote = "", variantNote = ""): string {
|
|
2634
|
+
function loopPrompt(loop: LoopState, regressionNote: string, strategyNote: string, boundsNote: string, interventionNote = "", variantNote = "", hypothesisNote = "", refineHintNote = ""): string {
|
|
2618
2635
|
// v0.23.0: metricless loops get their own prompt — no metric section,
|
|
2619
2636
|
// anti-doorknob rules instead of anti-gaming rules.
|
|
2620
2637
|
const metricless = !loop.measureCmd;
|
|
@@ -2641,7 +2658,9 @@ function loopPrompt(loop: LoopState, regressionNote: string, strategyNote: strin
|
|
|
2641
2658
|
.replace(/\$\{STRATEGY_NOTE\}/g, strategyNote)
|
|
2642
2659
|
.replace(/\$\{BOUNDS_NOTE\}/g, boundsNote)
|
|
2643
2660
|
.replace(/\$\{INTERVENTION_NOTE\}/g, interventionNote)
|
|
2644
|
-
.replace(/\$\{VARIANT_NOTE\}/g, variantNote)
|
|
2661
|
+
.replace(/\$\{VARIANT_NOTE\}/g, variantNote)
|
|
2662
|
+
.replace(/\$\{HYPOTHESIS_NOTE\}/g, hypothesisNote)
|
|
2663
|
+
.replace(/\$\{REFINE_HINT\}/g, refineHintNote);
|
|
2645
2664
|
}
|
|
2646
2665
|
|
|
2647
2666
|
function scheduleLoopTick(ctx: ExtensionContext): void {
|
|
@@ -2693,7 +2712,13 @@ function sendLoopTurn(): void {
|
|
|
2693
2712
|
// Strategy rotation (from pi-loop-mode's one good idea): one stall before
|
|
2694
2713
|
// the plateau window closes, stop polishing and change approach entirely.
|
|
2695
2714
|
const strategyNote = loop.stallCount >= loop.plateauWindow - 1 && loop.stallCount > 0
|
|
2696
|
-
? "**You are one stall from a plateau stop. Small tweaks are not working — try a FUNDAMENTALLY different approach: different file, different technique, or revert and rethink the angle of attack.**"
|
|
2715
|
+
? "**You are one stall from a plateau stop. Small tweaks are not working — try a FUNDAMENTALLY different approach: different file, different technique, or revert and rethink the angle of attack.**" +
|
|
2716
|
+
// v0.33.2: a metric flat AT BEST may mean the spec stopped capturing
|
|
2717
|
+
// "better" — the loop holds the evidence, so it says so (was: the
|
|
2718
|
+
// prompt said "call propose_loop_refine" but the loop never suggested it).
|
|
2719
|
+
(loop.lastValue !== null && loop.lastValue === loop.bestValue
|
|
2720
|
+
? " **The metric has been flat at best — if the spec no longer captures 'better' (saturated metric, drifted target), call propose_loop_refine.**"
|
|
2721
|
+
: "")
|
|
2697
2722
|
: "";
|
|
2698
2723
|
// v0.15.0: arbitrary bounds (never "completion") — surface what's armed.
|
|
2699
2724
|
// v0.23.0: for metricless loops the bounds are the ONLY stop (no
|
|
@@ -2724,12 +2749,19 @@ function sendLoopTurn(): void {
|
|
|
2724
2749
|
// v0.24.0: identical prompts invite identical answers — rotate the base
|
|
2725
2750
|
// instruction (metricless loops; metric loops already vary via values).
|
|
2726
2751
|
const variantNote = metricless ? continueVariant(loop.iteration) : "";
|
|
2752
|
+
// v0.33.2: one-shot prompt payloads, consumed on use.
|
|
2753
|
+
const hypothesisNote = loop.hypothesisFeedback ?? "";
|
|
2754
|
+
if (hypothesisNote) loop.hypothesisFeedback = undefined;
|
|
2755
|
+
const refineHintNote = loop.refineHint
|
|
2756
|
+
? `**The operator suggests refining the spec:** ${loop.refineHint} — if the current spec no longer captures "better", call propose_loop_refine (target and/or measureCmd${loop.specFile ? " and/or specText/specAppend" : ""}); if it still stands, say why in one line and keep working.`
|
|
2757
|
+
: "";
|
|
2758
|
+
if (refineHintNote) loop.refineHint = undefined;
|
|
2727
2759
|
try {
|
|
2728
2760
|
let loopResync = "";
|
|
2729
2761
|
if (postCompactResyncPending) { try { loopResync = buildPostCompactResync(); } catch { loopResync = ""; } } // v0.33.1
|
|
2730
2762
|
extensionApi.sendMessage({
|
|
2731
2763
|
customType: GOAL_EVENT_ENTRY,
|
|
2732
|
-
content: loopResync + loopPrompt(loop, regressionNote, strategyNote, boundsNote, interventionNote, variantNote),
|
|
2764
|
+
content: loopResync + loopPrompt(loop, regressionNote, strategyNote, boundsNote, interventionNote, variantNote, hypothesisNote, refineHintNote),
|
|
2733
2765
|
display: false,
|
|
2734
2766
|
}, { triggerTurn: true, deliverAs: "followUp" });
|
|
2735
2767
|
if (loopResync) postCompactResyncPending = false; // consumed only by a landed send
|
|
@@ -2800,6 +2832,23 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
|
|
|
2800
2832
|
}
|
|
2801
2833
|
} catch { /* no ledger yet */ }
|
|
2802
2834
|
}
|
|
2835
|
+
// v0.33.2: respec spec drift + checkbox progress — hash compared per
|
|
2836
|
+
// tick (external edits ledger spec_updated); newly checked boxes emit
|
|
2837
|
+
// the spec_item_progress signal the stuck gate already consumes (it was
|
|
2838
|
+
// consumed-but-never-emitted until now).
|
|
2839
|
+
if (loop.specFile) {
|
|
2840
|
+
const hash = specFileHash(loop.specFile);
|
|
2841
|
+
if (hash && loop.specHash && loop.specHash !== hash) {
|
|
2842
|
+
appendLedger(ctx.cwd, "spec_updated", { via: "external", iteration: loop.iteration });
|
|
2843
|
+
ctx.ui.notify("Spec file changed mid-loop — drift ledgered (spec_updated).", "info");
|
|
2844
|
+
}
|
|
2845
|
+
if (hash) loop.specHash = hash;
|
|
2846
|
+
const checked = countCheckedSpecItems(loop.specFile);
|
|
2847
|
+
if (checked !== null && loop.specChecked !== undefined && checked > loop.specChecked) {
|
|
2848
|
+
appendLedger(ctx.cwd, "spec_item_progress", { iteration: loop.iteration, newlyChecked: checked - loop.specChecked, totalChecked: checked });
|
|
2849
|
+
}
|
|
2850
|
+
if (checked !== null) loop.specChecked = checked;
|
|
2851
|
+
}
|
|
2803
2852
|
const iterSignals = {
|
|
2804
2853
|
fileWrites: loop.iterMetrics?.fileWrites ?? 0,
|
|
2805
2854
|
gitCommits,
|
|
@@ -2839,6 +2888,26 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
|
|
|
2839
2888
|
loop.lastStuckReason = undefined;
|
|
2840
2889
|
}
|
|
2841
2890
|
let outcome: LoopTickOutcome = metricless ? applyMetriclessTick(loop, nowIso()) : applyMeasurement(loop, value, nowIso());
|
|
2891
|
+
// v0.33.2: close the hypothesis feedback loop — the prediction went into
|
|
2892
|
+
// the ledger; now the VERDICT rides the next iteration's prompt.
|
|
2893
|
+
if (loop.lastHypothesis) {
|
|
2894
|
+
const h = loop.history;
|
|
2895
|
+
const cur = h.length >= 1 ? h[h.length - 1]!.value : null;
|
|
2896
|
+
const prev = h.length >= 2 ? h[h.length - 2]!.value : null;
|
|
2897
|
+
if (metricless || cur === null) {
|
|
2898
|
+
loop.hypothesisFeedback = `Last iteration you predicted: "${loop.lastHypothesis}". ${metricless ? "Metricless loop — no number to verify it against; say honestly whether the prediction landed." : "The measure printed no number — the prediction is unverifiable."}`;
|
|
2899
|
+
} else {
|
|
2900
|
+
const moved = prev === null
|
|
2901
|
+
? `first measurement ${cur}`
|
|
2902
|
+
: cur === prev
|
|
2903
|
+
? `flat at ${cur}`
|
|
2904
|
+
: loop.direction === "min"
|
|
2905
|
+
? (cur < prev ? `improved ${prev} → ${cur}` : `regressed ${prev} → ${cur}`)
|
|
2906
|
+
: (cur > prev ? `improved ${prev} → ${cur}` : `regressed ${prev} → ${cur}`);
|
|
2907
|
+
loop.hypothesisFeedback = `Last iteration you predicted: "${loop.lastHypothesis}". Result: metric ${moved} (best ${loop.bestValue}).`;
|
|
2908
|
+
}
|
|
2909
|
+
}
|
|
2910
|
+
loop.lastHypothesis = hypothesis;
|
|
2842
2911
|
persistState(ctx);
|
|
2843
2912
|
appendLedger(ctx.cwd, "loop_measured", {
|
|
2844
2913
|
iteration: loop.iteration,
|
|
@@ -2892,7 +2961,8 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
|
|
|
2892
2961
|
loop.stopReason = undefined;
|
|
2893
2962
|
loop.stallCount = 0;
|
|
2894
2963
|
loop.auditPlateauReprieves = reprieves;
|
|
2895
|
-
|
|
2964
|
+
const topFinding = topOpenAuditFinding(ctx.cwd); // v0.33.2: name what to close, not just the count
|
|
2965
|
+
loop.auditReprieveNote = `PLATEAU REPRIEVE (${reprieves}/${AUDIT_PLATEAU_MAX_REPRIEVES}): ${open} finding(s) still OPEN in ${AUDIT_FINDINGS_REL} — the plateau stop does not fire while the well isn't dry. Stop hunting and stop narrating: pick the smallest OPEN finding and CLOSE it this iteration (fix commit + checked box).${topFinding ? ` Top open: ${topFinding}` : ""} ${AUDIT_PLATEAU_MAX_REPRIEVES - reprieves} reprieve(s) remain.`;
|
|
2896
2966
|
persistState(ctx);
|
|
2897
2967
|
appendLedger(ctx.cwd, "audit_plateau_reprieve", { open, reprieves, best: loop.bestValue });
|
|
2898
2968
|
ctx.ui.notify(`Audit loop plateau reprieve (${reprieves}/${AUDIT_PLATEAU_MAX_REPRIEVES}): ${open} open findings — the well isn't dry, continuing.`, "info");
|
|
@@ -2951,6 +3021,9 @@ interface LoopConfig {
|
|
|
2951
3021
|
deferBaseline?: boolean;
|
|
2952
3022
|
/** v0.29.10: audit loops get audit-flavoured regression wording. */
|
|
2953
3023
|
kind?: "audit";
|
|
3024
|
+
/** v0.33.2: respec loops carry their spec file (drift detection,
|
|
3025
|
+
* checkbox progress, refine specText writes). */
|
|
3026
|
+
specFile?: string;
|
|
2954
3027
|
}
|
|
2955
3028
|
|
|
2956
3029
|
/** Shared loop-start path: /loop start AND propose_loop_draft (after Confirm). */
|
|
@@ -3017,6 +3090,9 @@ async function startLoopFromConfig(ctx: ExtensionContext, cfg: LoopConfig): Prom
|
|
|
3017
3090
|
branchName,
|
|
3018
3091
|
originalBranch,
|
|
3019
3092
|
toolSameRepeat: cfg.toolSameRepeat,
|
|
3093
|
+
specFile: cfg.specFile,
|
|
3094
|
+
specHash: cfg.specFile ? specFileHash(cfg.specFile) ?? undefined : undefined,
|
|
3095
|
+
specChecked: cfg.specFile ? countCheckedSpecItems(cfg.specFile) ?? undefined : undefined,
|
|
3020
3096
|
iterMetrics: { fileWrites: 0, iterationStartAt: nowIso() },
|
|
3021
3097
|
},
|
|
3022
3098
|
};
|
|
@@ -3135,6 +3211,28 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
3135
3211
|
|
|
3136
3212
|
// v0.28.14: /loop cancel is a first-class alias — users reached for
|
|
3137
3213
|
// /goal cancel to kill loops because "cancel" is the verb they know.
|
|
3214
|
+
if (sub === "refine" || sub === "polish") {
|
|
3215
|
+
// v0.33.2: the operator's respec verb. The refine flow stays
|
|
3216
|
+
// agent-proposed + user-confirmed (propose_loop_refine) — this command
|
|
3217
|
+
// queues the operator's suggestion into the next iteration's prompt.
|
|
3218
|
+
// ("polish" accepted as an alias: the widget footer advertised it
|
|
3219
|
+
// before the command existed — now it does.)
|
|
3220
|
+
if (!isLoopActive()) {
|
|
3221
|
+
ctx.ui.notify("No active loop to refine — /loop start first.", "warning");
|
|
3222
|
+
return;
|
|
3223
|
+
}
|
|
3224
|
+
const hint = rest.trim();
|
|
3225
|
+
if (!hint) {
|
|
3226
|
+
ctx.ui.notify("Usage: /loop refine <what the spec should capture better> — the suggestion rides the next iteration's prompt; the agent proposes via propose_loop_refine and you confirm.", "info");
|
|
3227
|
+
return;
|
|
3228
|
+
}
|
|
3229
|
+
state.loop!.refineHint = hint.slice(0, 300);
|
|
3230
|
+
persistState(ctx);
|
|
3231
|
+
appendLedger(ctx.cwd, "loop_refine_hint", { iteration: state.loop!.iteration, hint: state.loop!.refineHint });
|
|
3232
|
+
ctx.ui.notify("Refine hint queued — it rides the next iteration's prompt.", "info");
|
|
3233
|
+
return;
|
|
3234
|
+
}
|
|
3235
|
+
|
|
3138
3236
|
if (sub === "stop" || sub === "cancel") {
|
|
3139
3237
|
if (!state.loop) {
|
|
3140
3238
|
ctx.ui.notify("No loop to stop.", "info");
|
|
@@ -3273,6 +3371,7 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
3273
3371
|
maxIterations: 0,
|
|
3274
3372
|
branch: false,
|
|
3275
3373
|
force: false,
|
|
3374
|
+
specFile: specPath, // v0.33.2
|
|
3276
3375
|
});
|
|
3277
3376
|
return;
|
|
3278
3377
|
}
|
|
@@ -4124,12 +4223,14 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4124
4223
|
parameters: Type.Object({
|
|
4125
4224
|
target: Type.Optional(Type.String({ description: "The sharpened target text (omit to keep the current target)" })),
|
|
4126
4225
|
measureCmd: Type.Optional(Type.String({ description: "The new measure command printing ONE number (omit to keep the current metric)" })),
|
|
4226
|
+
specText: Type.Optional(Type.String({ description: "v0.33.2: full replacement text for the loop's spec file (respec loops only) — the orchestrator owns the write on user confirm" })),
|
|
4227
|
+
specAppend: Type.Optional(Type.String({ description: "v0.33.2: lines to append to the loop's spec file (respec loops only)" })),
|
|
4127
4228
|
rationale: Type.String({ description: "Why the current spec no longer captures 'better' — shown to the user in the Confirm dialog" }),
|
|
4128
4229
|
}),
|
|
4129
4230
|
async execute(_id, params, _signal, _onUpdate, execCtx) {
|
|
4130
4231
|
const foreign4 = foreignToolGuard(execCtx);
|
|
4131
4232
|
if (foreign4) return { content: [{ type: "text", text: foreign4 }], details: {} };
|
|
4132
|
-
const p = params as { target?: string; measureCmd?: string; rationale: string };
|
|
4233
|
+
const p = params as { target?: string; measureCmd?: string; specText?: string; specAppend?: string; rationale: string };
|
|
4133
4234
|
const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
|
|
4134
4235
|
const loop = state.loop;
|
|
4135
4236
|
if (!loop?.active) {
|
|
@@ -4142,8 +4243,12 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4142
4243
|
if (!loop.measureCmd && p.measureCmd?.trim()) {
|
|
4143
4244
|
return { content: [{ type: "text", text: "This loop is metricless — refining it into a measured loop isn't supported. /loop stop, then /loop start with a metric." }], details: {} };
|
|
4144
4245
|
}
|
|
4145
|
-
|
|
4146
|
-
|
|
4246
|
+
const specChange = (p.specText?.trim() || p.specAppend?.trim()) ? true : false;
|
|
4247
|
+
if (specChange && !loop.specFile) {
|
|
4248
|
+
return { content: [{ type: "text", text: "This loop has no spec file (specText/specAppend apply to /loop respec loops). Refine the target instead." }], details: {} };
|
|
4249
|
+
}
|
|
4250
|
+
if (newTarget === loop.target && newMeasure === loop.measureCmd && !specChange) {
|
|
4251
|
+
return { content: [{ type: "text", text: "Refinement proposed no changes — provide a new target, a new measureCmd, a spec change, or any combination." }], details: {} };
|
|
4147
4252
|
}
|
|
4148
4253
|
// Measure change → orchestrator test-runs the new command first.
|
|
4149
4254
|
let newBaseline: number | null = null;
|
|
@@ -4174,7 +4279,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4174
4279
|
confirmed = (await confirmDraft(
|
|
4175
4280
|
liveCtx,
|
|
4176
4281
|
"Confirm loop spec refinement",
|
|
4177
|
-
`Rationale: ${p.rationale}\n\nTarget:\n old: ${loop.target.slice(0, 120)}\n new: ${newTarget.slice(0, 120)}\n\nMeasure:\n old: ${loop.measureCmd}\n new: ${newMeasure}${newMeasure !== loop.measureCmd ? `\n test-run: ${testOutput.slice(0, 120)} → ${newBaseline}` : ""}\n\nThe loop keeps running against the refined spec (iteration ${loop.iteration} so far). Apply?`,
|
|
4282
|
+
`Rationale: ${p.rationale}\n\nTarget:\n old: ${loop.target.slice(0, 120)}\n new: ${newTarget.slice(0, 120)}\n\nMeasure:\n old: ${loop.measureCmd}\n new: ${newMeasure}${newMeasure !== loop.measureCmd ? `\n test-run: ${testOutput.slice(0, 120)} → ${newBaseline}` : ""}${specChange ? `\n\nSpec file (${loop.specFile}):\n ${p.specText?.trim() ? `REPLACE with ${p.specText!.trim().length} chars` : ""}${p.specText?.trim() && p.specAppend?.trim() ? " + " : ""}${p.specAppend?.trim() ? `APPEND: ${p.specAppend!.trim().slice(0, 120)}` : ""}` : ""}\n\nThe loop keeps running against the refined spec (iteration ${loop.iteration} so far). Apply?`,
|
|
4178
4283
|
)) === "yes";
|
|
4179
4284
|
} catch {
|
|
4180
4285
|
confirmed = false;
|
|
@@ -4191,9 +4296,22 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4191
4296
|
oldMeasureCmd: loop.measureCmd ?? "",
|
|
4192
4297
|
newMeasureCmd: newMeasure,
|
|
4193
4298
|
}, newBaseline);
|
|
4299
|
+
// v0.33.2: the orchestrator owns the spec write (honesty stays
|
|
4300
|
+
// inspectable — the agent never edits the spec it's judged against
|
|
4301
|
+
// outside a confirmed refine).
|
|
4302
|
+
if (specChange && loop.specFile) {
|
|
4303
|
+
try {
|
|
4304
|
+
if (p.specText?.trim()) fs.writeFileSync(loop.specFile, p.specText.trim() + "\n");
|
|
4305
|
+
if (p.specAppend?.trim()) fs.appendFileSync(loop.specFile, (p.specText?.trim() ? "" : "\n") + p.specAppend.trim() + "\n");
|
|
4306
|
+
loop.specHash = specFileHash(loop.specFile) ?? undefined;
|
|
4307
|
+
appendLedger(liveCtx.cwd, "spec_updated", { via: "refine", iteration: loop.iteration, replaced: Boolean(p.specText?.trim()), appended: Boolean(p.specAppend?.trim()) });
|
|
4308
|
+
} catch (e) {
|
|
4309
|
+
return { content: [{ type: "text", text: `Spec file write failed: ${String(e).slice(0, 200)}. The target/measure refinement was applied; re-propose the spec change.` }], details: {} };
|
|
4310
|
+
}
|
|
4311
|
+
}
|
|
4194
4312
|
persistState(liveCtx);
|
|
4195
|
-
appendLedger(liveCtx.cwd, "loop_refined", { iteration: loop.iteration, newTarget, newMeasureCmd: newMeasure, newBaseline });
|
|
4196
|
-
liveCtx.ui.notify(`Loop spec refined at iteration ${loop.iteration}.${newBaseline !== null ? ` New baseline: ${newBaseline}.` : ""}`, "info");
|
|
4313
|
+
appendLedger(liveCtx.cwd, "loop_refined", { iteration: loop.iteration, newTarget, newMeasureCmd: newMeasure, newBaseline, specChanged: specChange || undefined });
|
|
4314
|
+
liveCtx.ui.notify(`Loop spec refined at iteration ${loop.iteration}.${newBaseline !== null ? ` New baseline: ${newBaseline}.` : ""}${specChange ? " Spec file updated." : ""}`, "info");
|
|
4197
4315
|
return { content: [{ type: "text", text: "Refinement confirmed and applied. Continue improving against the NEW spec — one small change per turn." }], details: {} };
|
|
4198
4316
|
},
|
|
4199
4317
|
}));
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-goal-list-loop-audit",
|
|
3
|
-
"version": "0.33.
|
|
3
|
+
"version": "0.33.3",
|
|
4
4
|
"description": "Goal. Loop. Audit. Done. \u2014 a pi-coding-agent extension that supervises long-running work, with isolated auditor on each completion. Beat bamboozling by design: the auditor runs in a fresh session with no extensions, no skills, no editor \u2014 only the read tools needed to verify your goal.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "dracon",
|
|
@@ -31,6 +31,8 @@ fails, retry with a different approach — just continue, don't stall the loop
|
|
|
31
31
|
asking permission. You remain the single writer: apply the edit yourself.
|
|
32
32
|
|
|
33
33
|
${INTERVENTION_NOTE}
|
|
34
|
+
${HYPOTHESIS_NOTE}
|
|
35
|
+
${REFINE_HINT}
|
|
34
36
|
${REGRESSION_NOTE}
|
|
35
37
|
${STRATEGY_NOTE}
|
|
36
38
|
|
|
@@ -35,6 +35,8 @@ fails, retry with a different approach — just continue, don't stall the loop
|
|
|
35
35
|
asking permission. You remain the single writer: apply the edit yourself.
|
|
36
36
|
|
|
37
37
|
${INTERVENTION_NOTE}
|
|
38
|
+
${HYPOTHESIS_NOTE}
|
|
39
|
+
${REFINE_HINT}
|
|
38
40
|
${REGRESSION_NOTE}
|
|
39
41
|
${STRATEGY_NOTE}
|
|
40
42
|
|