pi-goal-list-loop-audit 0.28.33 → 0.29.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -66,6 +66,7 @@ matches `/list show`.
66
66
  /loop # draft the loop (agent grills; measure is test-run before you confirm)
67
67
  /loop start "keep polishing the UI" # infinite metricless loop (v0.23.6): no plateau, no cap — ends at time=/tokens= or /loop stop
68
68
  /loop respec # infinite metricless loop reconciling the codebase against the root SPEC.md / spec.md (v0.24.3) — 2 specs = you pick, 0 specs = drafting, 1 spec = auto-start (v0.24.4)
69
+ /loop audit # project-audit loop (v0.29.0): each iteration audits fresh, appends findings to .pi-glla/audit-loop/findings.md, fixes the top ones — the orchestrator counts open findings and the plateau stop ends it when the well is dry
69
70
  /loop start "reduce TODOs" measure="grep -c TODO src.txt | head -1" direction=min
70
71
  /loop start "shrink the bundle" measure="..." direction=min time=4 tokens=500000 # arbitrary bounds
71
72
  /loop start "reduce TODOs" measure="..." direction=min branch=1 # scratch-branch mode
@@ -201,7 +202,7 @@ No external watchdog plugin needed.
201
202
  /glla # open the settings UI
202
203
  /glla model=provider/id # auditor model override → GLOBAL
203
204
  /glla thinking=high # auditor thinking → GLOBAL
204
- /glla notify='cmd "$1"' # push on complete/pause/stop → GLOBAL
205
+ /glla notify='cmd "$1"' # custom push cmd · unset = auto-detect (notify-send/osascript) · off = silent → GLOBAL
205
206
  /glla tokenlimit=10000000 # per-goal token budget (default: off) → GLOBAL
206
207
  /glla tokenlimit=0 # explicitly no cap (the default)
207
208
  /glla wedgealert=30 # hung-command alert minutes (default: 30, 0 = off)
@@ -233,15 +234,14 @@ activates the moment the agent proposes it, with a notification and a
233
234
  `draft_autoaccepted` ledger entry (auto-accept is never silent). The seed
234
235
  carries the intent. Pair with `autoresume=on` for fully unattended rigs.
235
236
 
236
- ## Subagents (`@tintinweb/pi-subagents`)
237
+ ## Subagents
237
238
 
238
- Subagent sessions bind extensions too, so glla loads there — by design the
239
- **main session owns the goal/loop/list; subagents are workers** (v0.23.8):
239
+ glla's guarantees here come from glla itself (session-handle
240
+ discrimination), not from any specific subagent plugin — any Agent-tool
241
+ provider gets them. Subagent sessions bind extensions too, so glla loads
242
+ there — by design the **main session owns the goal/loop/list; subagents
243
+ are workers** (v0.23.8):
240
244
 
241
- - Read-only agents (Explore, Plan) get no glla tools (pi-subagents gates
242
- them); general-purpose agents see them but state-mutating calls
243
- (`complete_goal`, `propose_*`, `list_add`, `pause_goal`, …) are refused
244
- with "report back to the main agent".
245
245
  - A subagent session never clobbers the loop's session handle, never runs
246
246
  the restore gate, and never drives continuation — so the heartbeat,
247
247
  wedge alert, and auto-resume machinery always act on the main session.
@@ -249,6 +249,11 @@ Subagent sessions bind extensions too, so glla loads there — by design the
249
249
  is the discriminator.)
250
250
  - Subagent tool activity counts as activity for the wedge clock — a long
251
251
  subagent run is work, not a hang.
252
+ - With `@tintinweb/pi-subagents` specifically (the one we test against):
253
+ read-only agents (Explore, Plan) get no glla tools; general-purpose
254
+ agents see them but state-mutating calls (`complete_goal`, `propose_*`,
255
+ `list_add`, `pause_goal`, …) are refused with "report back to the main
256
+ agent".
252
257
 
253
258
  ## Token guard
254
259
 
@@ -287,17 +292,25 @@ contradictory turns. One driver at a time:
287
292
  fine to keep, never run a ralph loop while a goal/list/loop is active.
288
293
 
289
294
  **Goes well with it**: `@juicesharp/rpiv-ask-user-question` (drafting uses its
290
- structured forms), `@tintinweb/pi-subagents` (spawn research/review subagents
291
- inside goal work), `@tintinweb/pi-tasks` (session-wide DAGs vs our goal-scoped
292
- task lists — different granularity), `pi-chrome` (the research/search path for
293
- goals — logged-in browsing with no extra services; standalone search skills
294
- like `mmx-cli`/`pi-search-skill` are optional conveniences for bulk queries,
295
- not requirements).
295
+ structured forms), any subagent provider — e.g. `@tintinweb/pi-subagents` —
296
+ (spawn research/review subagents inside goal work), `pi-chrome` (the
297
+ research/search path for goals — logged-in browsing with no extra services;
298
+ standalone search skills like `mmx-cli`/`pi-search-skill` are optional
299
+ conveniences for bulk queries, not requirements).
300
+
301
+ **Overlaps — pick one**: `@tintinweb/pi-tasks` is a second task list next to
302
+ `/list`, and in practice the glla list *is* the task list (queue, statuses,
303
+ per-item audit trail) while the todos end up the weaker copy. We ran both
304
+ and removed pi-tasks. If you truly need session-wide dependency DAGs beyond
305
+ one ordered queue, it exists — but installing both is not the ideal combo.
296
306
 
297
307
  **Two footnotes**: (1) extension-registered providers work in the main session
298
308
  but not the auditor's extension-less session — if audits fail auth, set the
299
- override once with `/glla model=`. (2) `pi-notify-agent` notifies on every turn;
300
- `/glla notify=` fires only on goal complete/pause/loop stop.
309
+ override once with `/glla model=`. (2) `pi-notify-agent` notifies on every
310
+ turn; glla pushes fire only where there is something to DO (pauses, verdicts,
311
+ storms, wedge) and work out of the box — with no `notify=` configured glla
312
+ auto-detects `notify-send`/`osascript`; `notify=off` silences, `notify='<cmd>'`
313
+ customizes.
301
314
 
302
315
  ## Files
303
316
 
@@ -389,3 +389,36 @@ export function resolveSpecFile(cwd: string): string | null {
389
389
  export function respecTarget(specName: string): string {
390
390
  return `Reconcile the codebase against ${specName} (the project spec in the root). Read the spec critically first: if a requirement is stale, contradictory, or wrong for the current codebase, report the discrepancy and move on — never force the code to match a bad spec. Otherwise pick the next gap between spec and code and close it. Rotate: one iteration implements a missing or outdated spec item, the next audits something already "implemented" against the spec and fixes what drifted.`;
391
391
  }
392
+
393
+ // ---- /loop audit (v0.29.0) ----
394
+
395
+ /**
396
+ * The audit loop's findings file — checkbox lines, append-only. The agent
397
+ * appends new findings and checks off fixed ones; the ORCHESTRATOR counts
398
+ * open boxes every iteration. The agent never self-reports progress.
399
+ */
400
+ export const AUDIT_FINDINGS_REL = ".pi-glla/audit-loop/findings.md";
401
+
402
+ /**
403
+ * The audit-loop measure command: count open findings. Prints exactly one
404
+ * number in every file state (missing file / zero matches → 0). This is
405
+ * what respec (metricless) and the reviewer cascade (no termination) both
406
+ * lacked: an honest metric the plateau stop can believe — audits that stop
407
+ * surfacing new findings = the well is dry = the loop ends.
408
+ */
409
+ export function auditMeasureCmd(): string {
410
+ return `c=$(grep -cE '^- \\[ \\]' ${AUDIT_FINDINGS_REL} 2>/dev/null); echo \${c:-0}`;
411
+ }
412
+
413
+ /**
414
+ * The audit target. User's design (2026-07-29): "the looper running audits
415
+ * to see where to progress and what to fix" — the thing that fires at the
416
+ * end of goals and lists, finds the next batch of work, and works it.
417
+ * Each iteration: fresh audit pass → append NEW findings → fix the top
418
+ * open ones → check them off with the fix commit. Honesty laws: never
419
+ * fabricate findings, never rewrite the file's history, never check a box
420
+ * without the fix commit existing.
421
+ */
422
+ export function auditTarget(): string {
423
+ return `Audit the project for real problems and fix them, iteration by iteration. Every iteration: (1) run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real issues: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding as one checkbox line "- [ ] SEVERITY: short description (file:line)" to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed). (3) Fix the highest-severity OPEN finding(s) — real fixes, committed — then check the box: "- [x] … — fixed in <commit>". (4) Honesty law: never fabricate findings to look busy; never mark a finding fixed without the fix commit existing. When a full audit pass surfaces nothing new AND no open findings remain, say so plainly — the orchestrator counts open findings every iteration and the plateau stop ends the loop when the well is dry.`;
424
+ }
@@ -161,6 +161,8 @@ import {
161
161
  LOOP_DEFAULTS,
162
162
  resolveSpecFiles,
163
163
  respecTarget,
164
+ auditMeasureCmd,
165
+ auditTarget,
164
166
  HELD_ON_RESTORE,
165
167
  type LoopState,
166
168
  } from "../goal-loop-forever.js";
@@ -978,7 +980,12 @@ function archiveCurrentGoal(ctx: ExtensionContext, status: Status, stopReason?:
978
980
  if (goal.policy === "list" && status === "complete") {
979
981
  const advanced = activateNextListItem(ctx);
980
982
  // v0.26.0: the queue just EMPTIED on a completion → list-complete.
981
- if (!advanced) fireReviewer(ctx, { kind: "list", goalId: goal.id, objective: goal.objective, terminal: "goal-complete" });
983
+ if (!advanced) {
984
+ fireReviewer(ctx, { kind: "list", goalId: goal.id, objective: goal.objective, terminal: "goal-complete" });
985
+ // v0.29.0: the well ran dry — point at the project-audit loop. A
986
+ // suggestion, not an action: consent, never auto-start (v0.28.28).
987
+ ctx.ui.notify("List complete. /loop audit to sweep the project for the next batch of work.", "info");
988
+ }
982
989
  return;
983
990
  }
984
991
  // v0.26.0: a /goal (non-list) reached a terminal state → maybe fire.
@@ -1932,15 +1939,44 @@ function addSingleItem(ctx: ExtensionContext, raw: string): void {
1932
1939
  }
1933
1940
 
1934
1941
  /**
1935
- * Config-gated push notification: if settings.notifyCmd is set, shell out
1936
- * with the message as $1. Fire-and-forget — a broken notify command never
1937
- * blocks the loop. /glla notify='<cmd>' to configure.
1942
+ * Push notification, folded IN by default (v0.28.34 — user: "leaving it to
1943
+ * the user to set up sucks, cause then they won't have it"). Resolution:
1944
+ * notifyCmd === "off" → silent (explicit opt-out)
1945
+ * notifyCmd set → that command, message passed as $1
1946
+ * notifyCmd unset → auto-detect ONCE per session: notify-send
1947
+ * (Linux) or osascript (macOS); none → silent.
1948
+ * Pushes fire only where there is something to DO — pauses, auditor
1949
+ * verdicts, storms, wedge, persistence degradation — never per-turn noise.
1950
+ * Fire-and-forget: a broken notifier never blocks the loop.
1938
1951
  */
1952
+ let autoNotifyCmd: string | null | undefined; // undefined = not probed yet
1953
+
1954
+ function probeAutoNotify(ctx: ExtensionContext): void {
1955
+ if (autoNotifyCmd !== undefined || !extensionApi) return;
1956
+ autoNotifyCmd = null; // probing sentinel — drops at most the first push
1957
+ void extensionApi
1958
+ .exec("bash", ["-c", "command -v notify-send || command -v osascript || true"], { cwd: ctx.cwd })
1959
+ .then((r) => {
1960
+ const found = String((r as { stdout?: string }).stdout ?? "").trim();
1961
+ if (found.endsWith("notify-send")) autoNotifyCmd = `notify-send "pi-goal-list-loop-audit" "$1"`;
1962
+ // env-var handoff: the message never touches AppleScript quoting.
1963
+ else if (found.endsWith("osascript")) autoNotifyCmd = `GLLA_MSG="$1" osascript -e 'display notification (system attribute "GLLA_MSG") with title "pi-goal-list-loop-audit"'`;
1964
+ else autoNotifyCmd = null;
1965
+ })
1966
+ .catch(() => {
1967
+ autoNotifyCmd = null;
1968
+ });
1969
+ }
1970
+
1939
1971
  function notifyExternal(ctx: ExtensionContext, message: string): void {
1940
1972
  try {
1941
1973
  const settings = loadSettings(ctx.cwd);
1942
- const cmd = settings.notifyCmd;
1943
- if (!cmd || !extensionApi) return;
1974
+ if (settings.notifyCmd === "off" || !extensionApi) return;
1975
+ const cmd = settings.notifyCmd ?? autoNotifyCmd;
1976
+ if (!cmd) {
1977
+ if (settings.notifyCmd === undefined && autoNotifyCmd === undefined) probeAutoNotify(ctx);
1978
+ return;
1979
+ }
1944
1980
  void extensionApi.exec("bash", ["-c", cmd, "pi-goal-list-loop-audit", message], { cwd: ctx.cwd }).catch(() => {});
1945
1981
  } catch {
1946
1982
  // non-fatal by design
@@ -2481,6 +2517,34 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
2481
2517
  return;
2482
2518
  }
2483
2519
 
2520
+ if (sub === "audit") {
2521
+ // v0.29.0: the project-audit loop (user design: "the looper running
2522
+ // audits to see where to progress and what to fix — the thing that
2523
+ // fires at the end of goals and lists"). Unlike respec this is a
2524
+ // METRIC loop: the orchestrator counts open findings every iteration,
2525
+ // direction=min, and the plateau stop is the termination — audits that
2526
+ // stop surfacing new findings = the well is dry. User typed the
2527
+ // command = the act (same auto-start rule as respec).
2528
+ if (state.goal && state.goal.status === "active") {
2529
+ ctx.ui.notify("A goal is active — /goal cancel or /goal pause it before starting a loop.", "warning");
2530
+ return;
2531
+ }
2532
+ if (isLoopActive()) {
2533
+ ctx.ui.notify("A loop is already active. /loop stop first.", "warning");
2534
+ return;
2535
+ }
2536
+ await startLoopFromConfig(ctx, {
2537
+ target: auditTarget(),
2538
+ measureCmd: auditMeasureCmd(),
2539
+ direction: "min",
2540
+ plateauWindow: LOOP_DEFAULTS.plateauWindow,
2541
+ maxIterations: 0,
2542
+ branch: false,
2543
+ force: false,
2544
+ });
2545
+ return;
2546
+ }
2547
+
2484
2548
  if (sub === "respec") {
2485
2549
  // v0.24.3: reconcile the codebase against the root spec, forever.
2486
2550
  // Same auto-start path as /loop start (the user typed the command —
@@ -3892,7 +3956,7 @@ export async function handleSettingChoice(id: string, ctx: ExtensionContext): Pr
3892
3956
  // current effective subagent models. Treat as no-op.
3893
3957
  return;
3894
3958
  case "notifyCmd": {
3895
- const v = await ctx.ui.input("Notify command — the event message is passed as $1", "e.g. a desktop-notification or push command; empty = off");
3959
+ const v = await ctx.ui.input("Notify command — the event message is passed as $1", "custom command · empty = auto-detect (notify-send/osascript) · 'off' = silent");
3896
3960
  if (v !== undefined) saveSettings("global", ctx.cwd, { notifyCmd: v.trim() || undefined });
3897
3961
  return;
3898
3962
  }
@@ -4821,6 +4885,7 @@ export default function (pi: ExtensionAPI): void {
4821
4885
  getArgumentCompletions: completions([
4822
4886
  ["start", "skip drafting: /loop start \"<target>\" measure=\"<cmd>\" direction=min|max [window=5] [max=50]"],
4823
4887
  ["respec", "infinite metricless loop reconciling the codebase against the root SPEC.md"],
4888
+ ["audit", "project-audit loop: each iteration audits fresh, appends findings, fixes the top ones — plateau stops when the well is dry (v0.29.0)"],
4824
4889
  ["status", "show metric, iteration, best/last values, stall count"],
4825
4890
  ["stop", "end the loop (keeps the best state)"],
4826
4891
  ["cancel", "alias of /loop stop — end the loop"],
@@ -40,7 +40,7 @@ export const DEFAULT_REVIEWER_CONFIG: ReviewerConfig = {
40
40
  mode: "on",
41
41
  fireOn: ["goal-complete", "list-complete"],
42
42
  doNotFireOn: ["goal-aborted", "goal-paused"],
43
- cascade: ["convert-findings-to-list", "queue-leftovers", "fire-audit-on-clean", "notify-and-idle"],
43
+ cascade: ["convert-findings-to-list", "queue-leftovers", "notify-and-idle"],
44
44
  auditCadence: "every-clean-completion",
45
45
  auditScope: "regression-scan",
46
46
  leverageMode: "fix-without-confirm",
@@ -323,9 +323,9 @@ export function buildSettingsRows(
323
323
  id: "notifyCmd",
324
324
  section: "other",
325
325
  label: "Notify command",
326
- valueText: show("notifyCmd", "off"),
326
+ valueText: show("notifyCmd", "auto"),
327
327
  sourceText: src("notifyCmd"),
328
- description: "desktop push command; the event message is passed as $1",
328
+ description: "custom command ($1 = message) · unset = auto-detect notify-send/osascript · 'off' = silent",
329
329
  },
330
330
  {
331
331
  id: "tokenLimit",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.28.33",
3
+ "version": "0.29.0",
4
4
  "description": "Goal. Loop. Audit. Done. \u2014 a pi-coding-agent extension that supervises long-running work, with isolated auditor on each completion. Beat bamboozling by design: the auditor runs in a fresh session with no extensions, no skills, no editor \u2014 only the read tools needed to verify your goal.",
5
5
  "license": "MIT",
6
6
  "author": "dracon",