pi-goal-list-loop-audit 0.29.0 → 0.29.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -417,7 +417,8 @@ export function auditMeasureCmd(): string {
417
417
  * Each iteration: fresh audit pass → append NEW findings → fix the top
418
418
  * open ones → check them off with the fix commit. Honesty laws: never
419
419
  * fabricate findings, never rewrite the file's history, never check a box
420
- * without the fix commit existing.
420
+ * without the fix commit existing (committed with the repo's configured
421
+ * identity, on the current branch — no invented identities or branches).
421
422
  */
422
423
  export function auditTarget(): string {
423
424
  return `Audit the project for real problems and fix them, iteration by iteration. Every iteration: (1) run a FRESH audit pass over the codebase — spawn Explore subagents for breadth — hunting real issues: bugs, broken flows, regressions, drift between docs and code, dead code, security holes. Not style nits, not speculative refactors. (2) Append every NEW finding as one checkbox line "- [ ] SEVERITY: short description (file:line)" to ${AUDIT_FINDINGS_REL} (create the file on the first finding; append-only — never delete, rewrite, or reorder existing lines; never re-report a finding already listed). (3) Fix the highest-severity OPEN finding(s) — real fixes, committed — then check the box: "- [x] … — fixed in <commit>". (4) Honesty law: never fabricate findings to look busy; never mark a finding fixed without the fix commit existing. When a full audit pass surfaces nothing new AND no open findings remain, say so plainly — the orchestrator counts open findings every iteration and the plateau stop ends the loop when the well is dry.`;
@@ -114,6 +114,7 @@ import {
114
114
  } from "../goal-settings.js";
115
115
  import {
116
116
  DEFAULT_REVIEWER_CONFIG,
117
+ normalizeObjective,
117
118
  resolveReviewerConfig,
118
119
  reviewerMenuOptions,
119
120
  runReviewer,
@@ -546,6 +547,21 @@ function escalateSendRearmStorm(ctx: ExtensionContext, kind: "continuation" | "l
546
547
  notifyExternal(ctx, "Loop stopped: send-retry storm.");
547
548
  return;
548
549
  }
550
+ if (
551
+ state.goal &&
552
+ (state.goal.status === "auditing" || completionAuditInFlight || state.goal.pendingCompletion)
553
+ ) {
554
+ // v0.29.1: NEVER storm-pause the completion lifecycle. An isolated
555
+ // auditor run takes minutes and the main session is EXPECTED to be
556
+ // silent while it works — 15m of wedged re-arms + that silence is the
557
+ // exact trigger shape, so completing a goal under a wedged queue used
558
+ // to guarantee a mid-audit pause (field-observed in pully + hellhunter
559
+ // + junk-runner: "complete ending in a pause retry storm"). The audit
560
+ // lifecycle owns its own pauses (quota etc.).
561
+ appendLedger(ctx.cwd, "send_rearm_escalated_suppressed", { reason: "audit-lifecycle" });
562
+ ctx.ui.notify("Send-retry storm during the completion audit — NOT pausing; the auditor's silence is expected. If pi is wedged, restart; the stored claim survives.", "info");
563
+ return;
564
+ }
549
565
  if (state.goal && state.goal.status === "active") {
550
566
  updateGoal({
551
567
  status: "paused",
@@ -606,6 +622,30 @@ function heartbeatTick(): void {
606
622
  // worse, the stall escalation would PAUSE the goal — silently cancelling
607
623
  // the interruptedAt → auto-resume-on-restart promise the footer shows.
608
624
  if (extensionApiStale) return;
625
+ // v0.29.1: stranded-audit recovery. A goal left in "auditing" with NO
626
+ // in-flight audit means the auditor's result never landed (wedged queue
627
+ // ate the tool result; compaction/restart mid-audit). Field-observed in
628
+ // pully: 12h+ stuck "auditing" while the model had already confabulated
629
+ // the closure narrative. The audit silence is expected ONLY while
630
+ // completionAuditInFlight — its absence here means the run is orphaned.
631
+ // Recover: a stored claim re-runs the auditor directly; otherwise resume
632
+ // active so the agent re-calls complete_goal.
633
+ if (
634
+ state.goal?.status === "auditing" &&
635
+ !completionAuditInFlight &&
636
+ Date.now() - lastActivityAt >= 90_000
637
+ ) {
638
+ appendLedger(ctx.cwd, "stranded_audit_recovered", { goalId: state.goal.id, via: state.goal.pendingCompletion ? "stored-claim" : "resume-active" });
639
+ if (state.goal.pendingCompletion) {
640
+ ctx.ui.notify("Recovering a completion audit whose result never landed — re-running the auditor with the stored claim.", "info");
641
+ void retryStoredCompletionAudit(ctx, "quota-retry");
642
+ } else {
643
+ updateGoal({ status: "active" }, ctx);
644
+ ctx.ui.notify("A completion audit was interrupted (its result never landed). Resuming — re-call complete_goal when the deliverable still stands.", "warning");
645
+ scheduleContinuation(ctx, true);
646
+ }
647
+ return;
648
+ }
609
649
  // v0.26.5: pending-latch watchdog — a queued continuation whose turn
610
650
  // trigger was dropped (field-observed post-compaction: continuation
611
651
  // ACCEPTED at compact+0s, then 22 minutes of silence). The stuck latch
@@ -970,7 +1010,7 @@ function archiveCurrentGoal(ctx: ExtensionContext, status: Status, stopReason?:
970
1010
  try { fs.unlinkSync(goalMdPath(ctx.cwd, goal.id)); } catch {}
971
1011
  }
972
1012
  state = { ...state, goal: { ...goal, status, archivedPath: path.relative(ctx.cwd, target) || target, stopReason } };
973
- appendLedger(ctx.cwd, "goal_archived", { goalId: goal.id, status, stopReason });
1013
+ appendLedger(ctx.cwd, "goal_archived", { goalId: goal.id, status, stopReason, objective: goal.objective.slice(0, 300) });
974
1014
  persistState(ctx);
975
1015
  // Loop 2: a list-sourced goal COMPLETED → auto-activate the next item.
976
1016
  // Aborts are user actions (/list next, /goal cancel, list_activate) which
@@ -1672,8 +1712,57 @@ async function cmdTweak(args: string, ctx: ExtensionContext): Promise<void> {
1672
1712
  * contract extraction) → appended to the queue → persisted → first item
1673
1713
  * activated when nothing is running. Returns the count enqueued.
1674
1714
  */
1715
+ // v0.29.1: zombie-twin guard. A draft/enqueue whose objective matches a
1716
+ // goal COMPLETED in the last 24h re-creates just-finished work — field-
1717
+ // observed in junk-runner: the INFRA-NEW-18 close was re-drafted 3 minutes
1718
+ // after the auditor approved it and autoaccept waved the twin straight in,
1719
+ // where it stormed for 9h against a dead provider. Normalized compare (goal
1720
+ // ids stripped), 24h lookback, loud skip — never silent.
1721
+ const DUPLICATE_LOOKBACK_MS = 24 * 60 * 60 * 1000;
1722
+ const LEDGER_TAIL_BYTES = 256 * 1024;
1723
+ function recentlyCompletedObjectives(cwd: string): Set<string> {
1724
+ const done = new Set<string>();
1725
+ try {
1726
+ const p = ledgerPath(cwd);
1727
+ const size = fs.statSync(p).size;
1728
+ const buf = Buffer.alloc(Math.min(size, LEDGER_TAIL_BYTES));
1729
+ const fd = fs.openSync(p, "r");
1730
+ fs.readSync(fd, buf, 0, buf.length, Math.max(0, size - buf.length));
1731
+ fs.closeSync(fd);
1732
+ const cutoff = Date.now() - DUPLICATE_LOOKBACK_MS;
1733
+ for (const line of buf.toString("utf-8").split("\n")) {
1734
+ if (!line.includes('"goal_archived"') || !line.includes('"complete"')) continue;
1735
+ try {
1736
+ const e = JSON.parse(line);
1737
+ if (e?.type !== "goal_archived" || e.value?.status !== "complete") continue;
1738
+ if (!(Date.parse(e.ts ?? "") >= cutoff)) continue;
1739
+ // v0.29.1+ entries carry the objective inline; older entries fall
1740
+ // back to the archived goal file (## Objective → "> …" line).
1741
+ let objective = typeof e.value?.objective === "string" ? e.value.objective : "";
1742
+ if (!objective && e.value?.goalId) {
1743
+ try {
1744
+ const md = fs.readFileSync(archivedGoalPath(cwd, e.value.goalId), "utf-8");
1745
+ objective = md.split("## Objective")[1]?.split("\n").find((l: string) => l.startsWith("> "))?.slice(2) ?? "";
1746
+ } catch { /* archived file gone — skip */ }
1747
+ }
1748
+ if (objective) done.add(normalizeObjective(objective));
1749
+ } catch { /* malformed line — skip */ }
1750
+ }
1751
+ } catch { /* no ledger yet */ }
1752
+ return done;
1753
+ }
1754
+
1675
1755
  function enqueueItems(ctx: ExtensionContext, texts: string[], source: string, opts?: { autoActivate?: boolean }): number {
1676
- const items = texts.map((text) => {
1756
+ const recentlyDone = recentlyCompletedObjectives(ctx.cwd);
1757
+ const fresh = texts.filter((t) => !recentlyDone.has(normalizeObjective(extractVerificationContract(t).objective)));
1758
+ const skipped = texts.length - fresh.length;
1759
+ if (skipped > 0) {
1760
+ const first = texts.find((t) => recentlyDone.has(normalizeObjective(extractVerificationContract(t).objective))) ?? "";
1761
+ appendLedger(ctx.cwd, "list_duplicate_skipped", { source, count: skipped, objective: first.slice(0, 200) });
1762
+ ctx.ui.notify(`Skipped ${skipped} item(s) duplicating work COMPLETED in the last 24h (zombie-twin guard): ${first.slice(0, 90)}`, "warning");
1763
+ }
1764
+ if (fresh.length === 0) return 0;
1765
+ const items = fresh.map((text) => {
1677
1766
  const extracted = extractVerificationContract(text);
1678
1767
  return { id: newGoalId(), objective: extracted.objective, verificationContract: extracted.verificationContract || undefined, addedAt: nowIso() };
1679
1768
  });
@@ -3258,6 +3347,20 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3258
3347
  details: {},
3259
3348
  };
3260
3349
  }
3350
+ // v0.29.1: zombie-twin guard — a draft (auto-accepted OR confirmed)
3351
+ // whose objective duplicates a goal COMPLETED in the last 24h is
3352
+ // re-creating finished work. The Confirm dialog never said it was a
3353
+ // duplicate, so the gate belongs here. Junk-runner field case: the
3354
+ // just-approved INFRA-NEW-18 close re-drafted itself 3 minutes later.
3355
+ if (recentlyCompletedObjectives(liveCtx.cwd).has(normalizeObjective(p.objective.trim()))) {
3356
+ draftingTarget = null;
3357
+ appendLedger(liveCtx.cwd, "draft_duplicate_skipped", { kind: isListDraft ? "list" : "goal", objective: p.objective.trim().slice(0, 200) });
3358
+ liveCtx.ui.notify(`Draft REJECTED (zombie-twin guard): this objective matches a goal completed in the last 24h. Tell the user the work is already done.`, "warning");
3359
+ return {
3360
+ content: [{ type: "text", text: "This draft duplicates a goal that was COMPLETED within the last 24 hours (normalized objective match). Do NOT re-propose the same work. Report to the user that the objective is already done (see /glla audits or the archive) and ask what genuinely new work to take on instead." }],
3361
+ details: {},
3362
+ };
3363
+ }
3261
3364
  const confirmedTarget = draftingTarget;
3262
3365
  draftingTarget = null;
3263
3366
  const full = p.objective.trim() + (normContract ? `\nDone when:\n${normContract}` : "");
@@ -5344,6 +5447,22 @@ export default function (pi: ExtensionAPI): void {
5344
5447
  // 60-second provider hiccup waiting on a manual /goal resume.
5345
5448
  const detail = text.trim() ? ` (last: ${text.trim().replace(/\s+/g, " ").slice(0, 160)})` : "";
5346
5449
  const reason = `5 consecutive errors${detail}`;
5450
+ // v0.29.1: brake-cycle CAP. The v0.28.25 ladder slows the thrash
5451
+ // (1m→16m) but never STOPS it — junk-runner/hellhunter/pully each
5452
+ // burned 4+ pause↔retry cycles against provider windows that last
5453
+ // hours. After 6 consecutive brakes: park, no more auto-retries.
5454
+ if (errorBrakeStreak >= 6) {
5455
+ updateGoal({
5456
+ status: "paused",
5457
+ pauseKind: "error",
5458
+ pauseReason: `${reason} — 6 error-brakes in a row; the provider has been erroring for an extended window`,
5459
+ pauseSuggestedAction: "Check the provider/account (quota, outage), then /goal resume. No more automatic retries.",
5460
+ }, ctx);
5461
+ ctx.ui.notify(`${goalNoun()} parked: ${reason} — 6 brakes in a row, no more auto-retries. Check the provider, then /goal resume.`, "warning");
5462
+ notifyExternal(ctx, `${goalNoun()} parked: provider erroring across 6 error-brake cycles.`);
5463
+ appendLedger(ctx.cwd, "error_brake_capped", { streak: errorBrakeStreak, reason });
5464
+ return;
5465
+ }
5347
5466
  // v0.28.25: the cooldown escalates per CONSECUTIVE brake — a fleet-wide
5348
5467
  // 403 window is not cleared by re-braking every 60 seconds.
5349
5468
  const cooldownMs = 60_000 * 2 ** Math.min(errorBrakeStreak, 4);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.29.0",
3
+ "version": "0.29.2",
4
4
  "description": "Goal. Loop. Audit. Done. \u2014 a pi-coding-agent extension that supervises long-running work, with isolated auditor on each completion. Beat bamboozling by design: the auditor runs in a fresh session with no extensions, no skills, no editor \u2014 only the read tools needed to verify your goal.",
5
5
  "license": "MIT",
6
6
  "author": "dracon",
@@ -150,6 +150,7 @@ When the user must CHOOSE between paths, use `pause_goal` with `kind="decision"`
150
150
 
151
151
  - **Do not modify the objective silently.** The objective is the user's; if it has drifted from what makes sense, use `complete_goal`'s `newObjective` at completion time, or `pause_goal` and propose a `/goal tweak` mid-flight — never just work on something else and claim the original.
152
152
  - **Do not pretend completion.** If verification evidence is missing, call `pause_goal` instead of `complete_goal`.
153
+ - **Git discipline: never touch identity or branches.** Commit with the repo's configured identity exactly as-is — no `git config user.*`, no per-commit `git -c user.name=…` overrides, no invented identities like `<task>-agent <…@local>` (field-observed: a phase agent branded itself `phase-e-agent <phase-e@local>` and polluted the history). No creating or switching branches either — commit on the branch you found (usually `main`) and push to its upstream. If git refuses a commit for a missing identity, STOP and ask the user — never invent one.
153
154
  - **Do not polish doorknobs.** If you are out of work and the goal is satisfied, call `complete_goal` instead of inventing a side-improvement.
154
155
  - **Do not give up early.** If a task is hard, run it down properly. The auditor will catch doorknobs; the agent's job is to do the real work.
155
156
 
@@ -45,6 +45,8 @@ ${STRATEGY_NOTE}
45
45
  - Never repeat yourself: before acting, check what earlier iterations
46
46
  already did (git log / the artifacts themselves) and pick the next
47
47
  unaddressed piece of the spec.
48
+ - Git discipline: commit with the repo's configured identity as-is, on the
49
+ branch you found — never set `git config user.*` or invent an identity.
48
50
  - When the spec is genuinely, verifiably exhausted — every section
49
51
  addressed, every case handled — say so plainly in your reply instead of
50
52
  inventing cosmetic work. The user watches an honest loop gladly; a
@@ -44,6 +44,10 @@ ${STRATEGY_NOTE}
44
44
  - Do not modify the measure command or anything it reads for configuration;
45
45
  gaming the metric is bamboozling and the plateau detector will simply stop
46
46
  the loop.
47
+ - Git discipline: commit with the repo's configured identity as-is, on the
48
+ branch you found — never set `git config user.*`, never use per-commit
49
+ identity overrides, never invent `<task>-agent <…@local>` identities or
50
+ scratch branches.
47
51
  - The spec is ALIVE: if the target needs sharpening or the metric no longer
48
52
  captures "better", call propose_loop_refine with your rationale — the user
49
53
  confirms or rejects. Never game or silently abandon the metric; refine it.