pi-goal-list-loop-audit 0.28.33 → 0.29.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +29 -16
- package/extensions/goal-loop-forever.ts +33 -0
- package/extensions/loops/goal.ts +72 -7
- package/extensions/reviewer.ts +1 -1
- package/extensions/settings-menu.ts +2 -2
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -66,6 +66,7 @@ matches `/list show`.
|
|
|
66
66
|
/loop # draft the loop (agent grills; measure is test-run before you confirm)
|
|
67
67
|
/loop start "keep polishing the UI" # infinite metricless loop (v0.23.6): no plateau, no cap — ends at time=/tokens= or /loop stop
|
|
68
68
|
/loop respec # infinite metricless loop reconciling the codebase against the root SPEC.md / spec.md (v0.24.3) — 2 specs = you pick, 0 specs = drafting, 1 spec = auto-start (v0.24.4)
|
|
69
|
+
/loop audit # project-audit loop (v0.29.0): each iteration audits fresh, appends findings to .pi-glla/audit-loop/findings.md, fixes the top ones — the orchestrator counts open findings and the plateau stop ends it when the well is dry
|
|
69
70
|
/loop start "reduce TODOs" measure="grep -c TODO src.txt | head -1" direction=min
|
|
70
71
|
/loop start "shrink the bundle" measure="..." direction=min time=4 tokens=500000 # arbitrary bounds
|
|
71
72
|
/loop start "reduce TODOs" measure="..." direction=min branch=1 # scratch-branch mode
|
|
@@ -201,7 +202,7 @@ No external watchdog plugin needed.
|
|
|
201
202
|
/glla # open the settings UI
|
|
202
203
|
/glla model=provider/id # auditor model override → GLOBAL
|
|
203
204
|
/glla thinking=high # auditor thinking → GLOBAL
|
|
204
|
-
/glla notify='cmd "$1"' # push
|
|
205
|
+
/glla notify='cmd "$1"' # custom push cmd · unset = auto-detect (notify-send/osascript) · off = silent → GLOBAL
|
|
205
206
|
/glla tokenlimit=10000000 # per-goal token budget (default: off) → GLOBAL
|
|
206
207
|
/glla tokenlimit=0 # explicitly no cap (the default)
|
|
207
208
|
/glla wedgealert=30 # hung-command alert minutes (default: 30, 0 = off)
|
|
@@ -233,15 +234,14 @@ activates the moment the agent proposes it, with a notification and a
|
|
|
233
234
|
`draft_autoaccepted` ledger entry (auto-accept is never silent). The seed
|
|
234
235
|
carries the intent. Pair with `autoresume=on` for fully unattended rigs.
|
|
235
236
|
|
|
236
|
-
## Subagents
|
|
237
|
+
## Subagents
|
|
237
238
|
|
|
238
|
-
|
|
239
|
-
|
|
239
|
+
glla's guarantees here come from glla itself (session-handle
|
|
240
|
+
discrimination), not from any specific subagent plugin — any Agent-tool
|
|
241
|
+
provider gets them. Subagent sessions bind extensions too, so glla loads
|
|
242
|
+
there — by design the **main session owns the goal/loop/list; subagents
|
|
243
|
+
are workers** (v0.23.8):
|
|
240
244
|
|
|
241
|
-
- Read-only agents (Explore, Plan) get no glla tools (pi-subagents gates
|
|
242
|
-
them); general-purpose agents see them but state-mutating calls
|
|
243
|
-
(`complete_goal`, `propose_*`, `list_add`, `pause_goal`, …) are refused
|
|
244
|
-
with "report back to the main agent".
|
|
245
245
|
- A subagent session never clobbers the loop's session handle, never runs
|
|
246
246
|
the restore gate, and never drives continuation — so the heartbeat,
|
|
247
247
|
wedge alert, and auto-resume machinery always act on the main session.
|
|
@@ -249,6 +249,11 @@ Subagent sessions bind extensions too, so glla loads there — by design the
|
|
|
249
249
|
is the discriminator.)
|
|
250
250
|
- Subagent tool activity counts as activity for the wedge clock — a long
|
|
251
251
|
subagent run is work, not a hang.
|
|
252
|
+
- With `@tintinweb/pi-subagents` specifically (the one we test against):
|
|
253
|
+
read-only agents (Explore, Plan) get no glla tools; general-purpose
|
|
254
|
+
agents see them but state-mutating calls (`complete_goal`, `propose_*`,
|
|
255
|
+
`list_add`, `pause_goal`, …) are refused with "report back to the main
|
|
256
|
+
agent".
|
|
252
257
|
|
|
253
258
|
## Token guard
|
|
254
259
|
|
|
@@ -287,17 +292,25 @@ contradictory turns. One driver at a time:
|
|
|
287
292
|
fine to keep, never run a ralph loop while a goal/list/loop is active.
|
|
288
293
|
|
|
289
294
|
**Goes well with it**: `@juicesharp/rpiv-ask-user-question` (drafting uses its
|
|
290
|
-
structured forms), `@tintinweb/pi-subagents`
|
|
291
|
-
inside goal work),
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
295
|
+
structured forms), any subagent provider — e.g. `@tintinweb/pi-subagents` —
|
|
296
|
+
(spawn research/review subagents inside goal work), `pi-chrome` (the
|
|
297
|
+
research/search path for goals — logged-in browsing with no extra services;
|
|
298
|
+
standalone search skills like `mmx-cli`/`pi-search-skill` are optional
|
|
299
|
+
conveniences for bulk queries, not requirements).
|
|
300
|
+
|
|
301
|
+
**Overlaps — pick one**: `@tintinweb/pi-tasks` is a second task list next to
|
|
302
|
+
`/list`, and in practice the glla list *is* the task list (queue, statuses,
|
|
303
|
+
per-item audit trail) while the todos end up the weaker copy. We ran both
|
|
304
|
+
and removed pi-tasks. If you truly need session-wide dependency DAGs beyond
|
|
305
|
+
one ordered queue, it exists — but installing both is not the ideal combo.
|
|
296
306
|
|
|
297
307
|
**Two footnotes**: (1) extension-registered providers work in the main session
|
|
298
308
|
but not the auditor's extension-less session — if audits fail auth, set the
|
|
299
|
-
override once with `/glla model=`. (2) `pi-notify-agent` notifies on every
|
|
300
|
-
|
|
309
|
+
override once with `/glla model=`. (2) `pi-notify-agent` notifies on every
|
|
310
|
+
turn; glla pushes fire only where there is something to DO (pauses, verdicts,
|
|
311
|
+
storms, wedge) and work out of the box — with no `notify=` configured glla
|
|
312
|
+
auto-detects `notify-send`/`osascript`; `notify=off` silences, `notify='<cmd>'`
|
|
313
|
+
customizes.
|
|
301
314
|
|
|
302
315
|
## Files
|
|
303
316
|
|
|
@@ -389,3 +389,36 @@ export function resolveSpecFile(cwd: string): string | null {
|
|
|
389
389
|
export function respecTarget(specName: string): string {
|
|
390
390
|
return `Reconcile the codebase against ${specName} (the project spec in the root). Read the spec critically first: if a requirement is stale, contradictory, or wrong for the current codebase, report the discrepancy and move on — never force the code to match a bad spec. Otherwise pick the next gap between spec and code and close it. Rotate: one iteration implements a missing or outdated spec item, the next audits something already "implemented" against the spec and fixes what drifted.`;
|
|
391
391
|
}
|
|
392
|
+
|
|
393
|
+
// ---- /loop audit (v0.29.0) ----
|
|
394
|
+
|
|
395
|
+
/**
|
|
396
|
+
* The audit loop's findings file — checkbox lines, append-only. The agent
|
|
397
|
+
* appends new findings and checks off fixed ones; the ORCHESTRATOR counts
|
|
398
|
+
* open boxes every iteration. The agent never self-reports progress.
|
|
399
|
+
*/
|
|
400
|
+
export const AUDIT_FINDINGS_REL = ".pi-glla/audit-loop/findings.md";
|
|
401
|
+
|
|
402
|
+
/**
|
|
403
|
+
* The audit-loop measure command: count open findings. Prints exactly one
|
|
404
|
+
* number in every file state (missing file / zero matches → 0). This is
|
|
405
|
+
* what respec (metricless) and the reviewer cascade (no termination) both
|
|
406
|
+
* lacked: an honest metric the plateau stop can believe — audits that stop
|
|
407
|
+
* surfacing new findings = the well is dry = the loop ends.
|
|
408
|
+
*/
|
|
409
|
+
export function auditMeasureCmd(): string {
|
|
410
|
+
return `c=$(grep -cE '^- \\[ \\]' ${AUDIT_FINDINGS_REL} 2>/dev/null); echo \${c:-0}`;
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
/**
|
|
414
|
+
* The audit target. User's design (2026-07-29): "the looper running audits
|
|
415
|
+
* to see where to progress and what to fix" — the thing that fires at the
|
|
416
|
+
* end of goals and lists, finds the next batch of work, and works it.
|
|
417
|
+
* Each iteration: fresh audit pass → append NEW findings → fix the top
|
|
418
|
+
* open ones → check them off with the fix commit. Honesty laws: never
|
|
419
|
+
* fabricate findings, never rewrite the file's history, never check a box
|
|
420
|
+
* without the fix commit existing.
|
|
421
|
+
*/
|
|
422
|
+
export function auditTarget(): string {
|
|
423
|
+
return `Audit the project for real problems and fix them, iteration by iteration. Every iteration: (1) run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real issues: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding as one checkbox line "- [ ] SEVERITY: short description (file:line)" to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed). (3) Fix the highest-severity OPEN finding(s) — real fixes, committed — then check the box: "- [x] … — fixed in <commit>". (4) Honesty law: never fabricate findings to look busy; never mark a finding fixed without the fix commit existing. When a full audit pass surfaces nothing new AND no open findings remain, say so plainly — the orchestrator counts open findings every iteration and the plateau stop ends the loop when the well is dry.`;
|
|
424
|
+
}
|
package/extensions/loops/goal.ts
CHANGED
|
@@ -161,6 +161,8 @@ import {
|
|
|
161
161
|
LOOP_DEFAULTS,
|
|
162
162
|
resolveSpecFiles,
|
|
163
163
|
respecTarget,
|
|
164
|
+
auditMeasureCmd,
|
|
165
|
+
auditTarget,
|
|
164
166
|
HELD_ON_RESTORE,
|
|
165
167
|
type LoopState,
|
|
166
168
|
} from "../goal-loop-forever.js";
|
|
@@ -978,7 +980,12 @@ function archiveCurrentGoal(ctx: ExtensionContext, status: Status, stopReason?:
|
|
|
978
980
|
if (goal.policy === "list" && status === "complete") {
|
|
979
981
|
const advanced = activateNextListItem(ctx);
|
|
980
982
|
// v0.26.0: the queue just EMPTIED on a completion → list-complete.
|
|
981
|
-
if (!advanced)
|
|
983
|
+
if (!advanced) {
|
|
984
|
+
fireReviewer(ctx, { kind: "list", goalId: goal.id, objective: goal.objective, terminal: "goal-complete" });
|
|
985
|
+
// v0.29.0: the well ran dry — point at the project-audit loop. A
|
|
986
|
+
// suggestion, not an action: consent, never auto-start (v0.28.28).
|
|
987
|
+
ctx.ui.notify("List complete. /loop audit to sweep the project for the next batch of work.", "info");
|
|
988
|
+
}
|
|
982
989
|
return;
|
|
983
990
|
}
|
|
984
991
|
// v0.26.0: a /goal (non-list) reached a terminal state → maybe fire.
|
|
@@ -1932,15 +1939,44 @@ function addSingleItem(ctx: ExtensionContext, raw: string): void {
|
|
|
1932
1939
|
}
|
|
1933
1940
|
|
|
1934
1941
|
/**
|
|
1935
|
-
*
|
|
1936
|
-
*
|
|
1937
|
-
*
|
|
1942
|
+
* Push notification, folded IN by default (v0.28.34 — user: "leaving it to
|
|
1943
|
+
* the user to set up sucks, cause then they won't have it"). Resolution:
|
|
1944
|
+
* notifyCmd === "off" → silent (explicit opt-out)
|
|
1945
|
+
* notifyCmd set → that command, message passed as $1
|
|
1946
|
+
* notifyCmd unset → auto-detect ONCE per session: notify-send
|
|
1947
|
+
* (Linux) or osascript (macOS); none → silent.
|
|
1948
|
+
* Pushes fire only where there is something to DO — pauses, auditor
|
|
1949
|
+
* verdicts, storms, wedge, persistence degradation — never per-turn noise.
|
|
1950
|
+
* Fire-and-forget: a broken notifier never blocks the loop.
|
|
1938
1951
|
*/
|
|
1952
|
+
let autoNotifyCmd: string | null | undefined; // undefined = not probed yet
|
|
1953
|
+
|
|
1954
|
+
function probeAutoNotify(ctx: ExtensionContext): void {
|
|
1955
|
+
if (autoNotifyCmd !== undefined || !extensionApi) return;
|
|
1956
|
+
autoNotifyCmd = null; // probing sentinel — drops at most the first push
|
|
1957
|
+
void extensionApi
|
|
1958
|
+
.exec("bash", ["-c", "command -v notify-send || command -v osascript || true"], { cwd: ctx.cwd })
|
|
1959
|
+
.then((r) => {
|
|
1960
|
+
const found = String((r as { stdout?: string }).stdout ?? "").trim();
|
|
1961
|
+
if (found.endsWith("notify-send")) autoNotifyCmd = `notify-send "pi-goal-list-loop-audit" "$1"`;
|
|
1962
|
+
// env-var handoff: the message never touches AppleScript quoting.
|
|
1963
|
+
else if (found.endsWith("osascript")) autoNotifyCmd = `GLLA_MSG="$1" osascript -e 'display notification (system attribute "GLLA_MSG") with title "pi-goal-list-loop-audit"'`;
|
|
1964
|
+
else autoNotifyCmd = null;
|
|
1965
|
+
})
|
|
1966
|
+
.catch(() => {
|
|
1967
|
+
autoNotifyCmd = null;
|
|
1968
|
+
});
|
|
1969
|
+
}
|
|
1970
|
+
|
|
1939
1971
|
function notifyExternal(ctx: ExtensionContext, message: string): void {
|
|
1940
1972
|
try {
|
|
1941
1973
|
const settings = loadSettings(ctx.cwd);
|
|
1942
|
-
|
|
1943
|
-
|
|
1974
|
+
if (settings.notifyCmd === "off" || !extensionApi) return;
|
|
1975
|
+
const cmd = settings.notifyCmd ?? autoNotifyCmd;
|
|
1976
|
+
if (!cmd) {
|
|
1977
|
+
if (settings.notifyCmd === undefined && autoNotifyCmd === undefined) probeAutoNotify(ctx);
|
|
1978
|
+
return;
|
|
1979
|
+
}
|
|
1944
1980
|
void extensionApi.exec("bash", ["-c", cmd, "pi-goal-list-loop-audit", message], { cwd: ctx.cwd }).catch(() => {});
|
|
1945
1981
|
} catch {
|
|
1946
1982
|
// non-fatal by design
|
|
@@ -2481,6 +2517,34 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
2481
2517
|
return;
|
|
2482
2518
|
}
|
|
2483
2519
|
|
|
2520
|
+
if (sub === "audit") {
|
|
2521
|
+
// v0.29.0: the project-audit loop (user design: "the looper running
|
|
2522
|
+
// audits to see where to progress and what to fix — the thing that
|
|
2523
|
+
// fires at the end of goals and lists"). Unlike respec this is a
|
|
2524
|
+
// METRIC loop: the orchestrator counts open findings every iteration,
|
|
2525
|
+
// direction=min, and the plateau stop is the termination — audits that
|
|
2526
|
+
// stop surfacing new findings = the well is dry. User typed the
|
|
2527
|
+
// command = the act (same auto-start rule as respec).
|
|
2528
|
+
if (state.goal && state.goal.status === "active") {
|
|
2529
|
+
ctx.ui.notify("A goal is active — /goal cancel or /goal pause it before starting a loop.", "warning");
|
|
2530
|
+
return;
|
|
2531
|
+
}
|
|
2532
|
+
if (isLoopActive()) {
|
|
2533
|
+
ctx.ui.notify("A loop is already active. /loop stop first.", "warning");
|
|
2534
|
+
return;
|
|
2535
|
+
}
|
|
2536
|
+
await startLoopFromConfig(ctx, {
|
|
2537
|
+
target: auditTarget(),
|
|
2538
|
+
measureCmd: auditMeasureCmd(),
|
|
2539
|
+
direction: "min",
|
|
2540
|
+
plateauWindow: LOOP_DEFAULTS.plateauWindow,
|
|
2541
|
+
maxIterations: 0,
|
|
2542
|
+
branch: false,
|
|
2543
|
+
force: false,
|
|
2544
|
+
});
|
|
2545
|
+
return;
|
|
2546
|
+
}
|
|
2547
|
+
|
|
2484
2548
|
if (sub === "respec") {
|
|
2485
2549
|
// v0.24.3: reconcile the codebase against the root spec, forever.
|
|
2486
2550
|
// Same auto-start path as /loop start (the user typed the command —
|
|
@@ -3892,7 +3956,7 @@ export async function handleSettingChoice(id: string, ctx: ExtensionContext): Pr
|
|
|
3892
3956
|
// current effective subagent models. Treat as no-op.
|
|
3893
3957
|
return;
|
|
3894
3958
|
case "notifyCmd": {
|
|
3895
|
-
const v = await ctx.ui.input("Notify command — the event message is passed as $1", "
|
|
3959
|
+
const v = await ctx.ui.input("Notify command — the event message is passed as $1", "custom command · empty = auto-detect (notify-send/osascript) · 'off' = silent");
|
|
3896
3960
|
if (v !== undefined) saveSettings("global", ctx.cwd, { notifyCmd: v.trim() || undefined });
|
|
3897
3961
|
return;
|
|
3898
3962
|
}
|
|
@@ -4821,6 +4885,7 @@ export default function (pi: ExtensionAPI): void {
|
|
|
4821
4885
|
getArgumentCompletions: completions([
|
|
4822
4886
|
["start", "skip drafting: /loop start \"<target>\" measure=\"<cmd>\" direction=min|max [window=5] [max=50]"],
|
|
4823
4887
|
["respec", "infinite metricless loop reconciling the codebase against the root SPEC.md"],
|
|
4888
|
+
["audit", "project-audit loop: each iteration audits fresh, appends findings, fixes the top ones — plateau stops when the well is dry (v0.29.0)"],
|
|
4824
4889
|
["status", "show metric, iteration, best/last values, stall count"],
|
|
4825
4890
|
["stop", "end the loop (keeps the best state)"],
|
|
4826
4891
|
["cancel", "alias of /loop stop — end the loop"],
|
package/extensions/reviewer.ts
CHANGED
|
@@ -40,7 +40,7 @@ export const DEFAULT_REVIEWER_CONFIG: ReviewerConfig = {
|
|
|
40
40
|
mode: "on",
|
|
41
41
|
fireOn: ["goal-complete", "list-complete"],
|
|
42
42
|
doNotFireOn: ["goal-aborted", "goal-paused"],
|
|
43
|
-
cascade: ["convert-findings-to-list", "queue-leftovers", "
|
|
43
|
+
cascade: ["convert-findings-to-list", "queue-leftovers", "notify-and-idle"],
|
|
44
44
|
auditCadence: "every-clean-completion",
|
|
45
45
|
auditScope: "regression-scan",
|
|
46
46
|
leverageMode: "fix-without-confirm",
|
|
@@ -323,9 +323,9 @@ export function buildSettingsRows(
|
|
|
323
323
|
id: "notifyCmd",
|
|
324
324
|
section: "other",
|
|
325
325
|
label: "Notify command",
|
|
326
|
-
valueText: show("notifyCmd", "
|
|
326
|
+
valueText: show("notifyCmd", "auto"),
|
|
327
327
|
sourceText: src("notifyCmd"),
|
|
328
|
-
description: "
|
|
328
|
+
description: "custom command ($1 = message) · unset = auto-detect notify-send/osascript · 'off' = silent",
|
|
329
329
|
},
|
|
330
330
|
{
|
|
331
331
|
id: "tokenLimit",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-goal-list-loop-audit",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.29.0",
|
|
4
4
|
"description": "Goal. Loop. Audit. Done. \u2014 a pi-coding-agent extension that supervises long-running work, with isolated auditor on each completion. Beat bamboozling by design: the auditor runs in a fresh session with no extensions, no skills, no editor \u2014 only the read tools needed to verify your goal.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "dracon",
|