pi-goal-list-loop-audit 0.34.19 → 0.34.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1313,15 +1313,42 @@ export interface InfraRetryOutcome<T> {
1313
1313
  * (retried once)". The failed pair is never a verdict on the work. */
1314
1314
  export async function runWithInfraRetry<T extends { error?: string; approved: boolean; disapproved: boolean }>(
1315
1315
  run: () => Promise<T>,
1316
- opts: { backoffMs?: number; sleep?: (ms: number) => Promise<void>; onRetry?: (error: string) => void } = {},
1316
+ opts: {
1317
+ backoffMs?: number;
1318
+ sleep?: (ms: number) => Promise<void>;
1319
+ onRetry?: (error: string) => void;
1320
+ /**
1321
+ * v0.34.20: delayed retry callers can fail closed across a session
1322
+ * replacement. The first attempt may finish after its ExtensionContext
1323
+ * was invalidated; never launch the second attempt unless the caller can
1324
+ * prove that its session/generation is still live.
1325
+ */
1326
+ shouldRetry?: () => boolean;
1327
+ } = {},
1317
1328
  ): Promise<InfraRetryOutcome<T>> {
1318
1329
  const sleep = opts.sleep ?? ((ms: number) => new Promise<void>((r) => setTimeout(r, ms)));
1319
1330
  const first = await run();
1320
1331
  if (first.approved || first.disapproved || !isRetriableInfraError(first.error)) {
1321
1332
  return { result: first, retriedOnce: false };
1322
1333
  }
1334
+ if (opts.shouldRetry) {
1335
+ try {
1336
+ if (!opts.shouldRetry()) return { result: first, retriedOnce: false };
1337
+ } catch {
1338
+ // A lifecycle probe that cannot establish liveness is a hard stop, not
1339
+ // permission to retry an old session.
1340
+ return { result: first, retriedOnce: false };
1341
+ }
1342
+ }
1323
1343
  opts.onRetry?.(first.error!);
1324
1344
  await sleep(opts.backoffMs ?? 5000);
1345
+ if (opts.shouldRetry) {
1346
+ try {
1347
+ if (!opts.shouldRetry()) return { result: first, retriedOnce: false };
1348
+ } catch {
1349
+ return { result: first, retriedOnce: false };
1350
+ }
1351
+ }
1325
1352
  const second = await run();
1326
1353
  return { result: second, retriedOnce: true };
1327
1354
  }
@@ -706,6 +706,9 @@ let carryoverResolved = true;
706
706
  // set while complete_goal's isolated audit runs, so the heartbeat never
707
707
  // refires into an in-flight completion.
708
708
  let completionAuditInFlight = false;
709
+ // v0.34.20: an old auditor's finally block must not clear the in-flight
710
+ // marker belonging to a fresh lifecycle generation.
711
+ let completionAuditGeneration: number | null = null;
709
712
  // v0.32.0: consecutive stored-claim quota retries (capped at 5, then hold).
710
713
  let quotaRetryStreak = 0;
711
714
  let heartbeatTimer: NodeJS.Timeout | null = null;
@@ -1079,7 +1082,7 @@ function heartbeatTick(): void {
1079
1082
  appendLedger(ctx.cwd, "stranded_audit_recovered", { goalId: state.goal.id, via: state.goal.pendingCompletion ? "stored-claim" : "resume-active" });
1080
1083
  if (state.goal.pendingCompletion) {
1081
1084
  ctx.ui.notify("Recovering a completion audit whose result never landed — re-running the auditor with the stored claim.", "info");
1082
- void retryStoredCompletionAudit(ctx, "quota-retry");
1085
+ void retryStoredCompletionAudit("quota-retry");
1083
1086
  } else {
1084
1087
  updateGoal({ status: "active" }, ctx);
1085
1088
  ctx.ui.notify("A completion audit was interrupted (its result never landed). Resuming — re-call complete_goal when the deliverable still stands.", "warning");
@@ -1278,6 +1281,53 @@ function freshCtx(): ExtensionContext | null {
1278
1281
  }
1279
1282
  }
1280
1283
 
1284
+ /**
1285
+ * v0.34.20: a timer can already be queued when clearSessionOwnedTimers()
1286
+ * runs, and an async audit can finish after a replacement without a queued
1287
+ * timer at all. Delayed work must prove both facts before touching pi:
1288
+ * generation identity is unchanged and the context probe succeeds. A null
1289
+ * result is a normal fail-closed handoff, not a reason to use the caller's
1290
+ * captured context as a fallback.
1291
+ */
1292
+ function freshCtxForGeneration(generation: number): ExtensionContext | null {
1293
+ if (
1294
+ generation !== sessionGeneration ||
1295
+ sessionHandoffPending ||
1296
+ initialSessionLoadPending ||
1297
+ extensionApiStale ||
1298
+ staleTerminalDone ||
1299
+ zombieStoodDown
1300
+ ) return null;
1301
+ return freshCtx();
1302
+ }
1303
+
1304
+ /**
1305
+ * v0.34.20: the generic quota helper owns only the wall-clock timer and the
1306
+ * immediate notification. This adapter owns the session boundary: callbacks
1307
+ * receive a context proven fresh at fire time and may not close over the
1308
+ * scheduling event's ctx.
1309
+ */
1310
+ function scheduleQuotaRetryForSession(
1311
+ ctx: ExtensionContext,
1312
+ retryAfterSec: number,
1313
+ reason: string,
1314
+ fire: (ctx: ExtensionContext) => void | Promise<void>,
1315
+ label?: string,
1316
+ ): void {
1317
+ const generation = sessionGeneration;
1318
+ scheduleQuotaRetry(ctx, retryAfterSec, reason, () => {
1319
+ const current = freshCtxForGeneration(generation);
1320
+ if (!current) return;
1321
+ try {
1322
+ void Promise.resolve(fire(current)).catch((err) => {
1323
+ if (isStaleApiError(err)) extensionApiStale = true;
1324
+ });
1325
+ } catch (err) {
1326
+ if (isStaleApiError(err)) extensionApiStale = true;
1327
+ }
1328
+ }, label);
1329
+ }
1330
+
1281
1331
  // v0.34.15 (hegemon 2026-08-01): pi ACCEPTED the continuation — footer showed
1282
1332
  // "1 queued" — but the turn trigger was dead, so the message sat queued while
1283
1333
  // pi idled. The 0.34.11 watchdog gates on "pi reported NO pending" and the
@@ -1603,11 +1653,13 @@ function autoArbitrateStackedState(ctx: ExtensionContext): void {
1603
1653
  * against the live queue), present DECIDE findings without queueing them.
1604
1654
  * Confirm-gated like every bulk import (v0.23.7: the user reads what lands
1605
1655
  * in the queue); a decline leaves the findings open for a later re-run.
1656
+ * v0.34.20: this detached operation retains only cwd + generation. Every
1657
+ * context use after the confirmation await must come from the fresh session.
1606
1658
  */
1607
- async function fanOutListAuditFindings(ctx: ExtensionContext): Promise<void> {
1659
+ async function fanOutListAuditFindings(cwd: string, generation: number): Promise<void> {
1608
1660
  let md = "";
1609
1661
  try {
1610
- md = fs.readFileSync(path.join(ctx.cwd, AUDIT_FINDINGS_REL), "utf-8");
1662
+ md = fs.readFileSync(path.join(cwd, AUDIT_FINDINGS_REL), "utf-8");
1611
1663
  } catch {
1612
1664
  /* no findings file — the audit was clean or never wrote */
1613
1665
  }
@@ -1619,6 +1671,8 @@ async function fanOutListAuditFindings(ctx: ExtensionContext): Promise<void> {
1619
1671
  // hundreds of items on a single Confirm.
1620
1672
  const fresh = open.filter((f) => !queuedText.includes(f.text.slice(0, 60))).slice(0, 50);
1621
1673
  const alreadyQueued = open.length - fresh.length;
1674
+ const current = freshCtxForGeneration(generation);
1675
+ if (!current) return;
1622
1676
  // v0.33.3: DECIDE findings are RAISED TO THE USER as real questions
1623
1677
  // (hegemon 2026-07-31: a truncated notify left the user typing "decide
1624
1678
  // what" into the void). The orchestrator can't call ask_user_question —
@@ -1628,41 +1682,48 @@ async function fanOutListAuditFindings(ctx: ExtensionContext): Promise<void> {
1628
1682
  // queued or the fan-out was declined.
1629
1683
  if (decisions.length > 0) {
1630
1684
  const decList = decisions.slice(0, 8).map((d, i) => `${i + 1}. ${d.slice(0, 500)}`).join("\n");
1631
- if (safeSteerUser(ctx,
1685
+ if (safeSteerUser(current,
1632
1686
  `[DECIDE FINDINGS — user decisions needed] The audit surfaced ${decisions.length} DECIDE finding(s) — direction calls only the user can make (a decision is not a task, so they were NOT queued):\n${decList}\nRaise them to the user NOW with ask_user_question — one question per finding, options from the finding's own two sides plus "Defer" (prose numbered list if ask_user_question is unavailable; Esc = Defer). Then record every answer in ${AUDIT_FINDINGS_REL}: replace the "- [?]" line with "- [x] DECIDED: <what was chosen> (<date>)" (or "- [x] DEFERRED") so it stops re-surfacing, and queue any chosen work with list_add — do NOT start the work inline.`))
1633
- appendLedger(ctx.cwd, "list_audit_decisions_raised", { decisions: decisions.length });
1687
+ appendLedger(cwd, "list_audit_decisions_raised", { decisions: decisions.length });
1634
1688
  }
1635
1689
  const decideNote =
1636
1690
  decisions.length > 0
1637
1691
  ? ` ${decisions.length} DECIDE finding(s) need YOU — raising them as questions now (not queued — a decision is not a task).`
1638
1692
  : "";
1639
1693
  if (fresh.length === 0) {
1640
- ctx.ui.notify(
1694
+ const afterDecision = freshCtxForGeneration(generation);
1695
+ if (!afterDecision) return;
1696
+ afterDecision.ui.notify(
1641
1697
  open.length > 0
1642
1698
  ? `Audit collected ${open.length} open finding(s) — all already queued.${decideNote}`
1643
1699
  : `Audit complete — no open findings; the project is clean, nothing to queue.${decideNote}`,
1644
1700
  "info",
1645
1701
  );
1646
- appendLedger(ctx.cwd, "list_audit_fanout_empty", { open: open.length, decisions: decisions.length });
1702
+ appendLedger(cwd, "list_audit_fanout_empty", { open: open.length, decisions: decisions.length });
1647
1703
  return;
1648
1704
  }
1649
1705
  const preview = fresh.map((f, i) => ` ${i + 1}. ${f.text.slice(0, 110)}`).join("\n");
1650
1706
  let confirmed = true;
1651
- if (ctx.hasUI) {
1707
+ const beforeConfirm = freshCtxForGeneration(generation);
1708
+ if (!beforeConfirm) return;
1709
+ if (beforeConfirm.hasUI) {
1652
1710
  try {
1653
- confirmed = await ctx.ui.confirm(`Queue ${fresh.length} audit finding(s) as list items?`, preview);
1711
+ confirmed = await beforeConfirm.ui.confirm(`Queue ${fresh.length} audit finding(s) as list items?`, preview);
1654
1712
  } catch {
1655
1713
  confirmed = false;
1656
1714
  }
1657
1715
  }
1716
+ // A confirm result from an old session is not consent for the replacement.
1717
+ const afterConfirm = freshCtxForGeneration(generation);
1718
+ if (!afterConfirm) return;
1658
1719
  if (!confirmed) {
1659
- appendLedger(ctx.cwd, "list_audit_fanout_declined", { findings: fresh.length });
1660
- ctx.ui.notify(`Fan-out declined — the findings stay open in ${AUDIT_FINDINGS_REL}; /list audit re-queues them any time.`, "info");
1720
+ appendLedger(cwd, "list_audit_fanout_declined", { findings: fresh.length });
1721
+ afterConfirm.ui.notify(`Fan-out declined — the findings stay open in ${AUDIT_FINDINGS_REL}; /list audit re-queues them any time.`, "info");
1661
1722
  return;
1662
1723
  }
1663
- const n = enqueueItems(ctx, fresh.map((f) => listAuditFanoutItemText(f.text)), "list audit fan-out");
1664
- appendLedger(ctx.cwd, "list_audit_fanout", { queued: n, alreadyQueued, decisions: decisions.length });
1665
- ctx.ui.notify(
1724
+ const n = enqueueItems(afterConfirm, fresh.map((f) => listAuditFanoutItemText(f.text)), "list audit fan-out");
1725
+ appendLedger(cwd, "list_audit_fanout", { queued: n, alreadyQueued, decisions: decisions.length });
1726
+ afterConfirm.ui.notify(
1666
1727
  `Queued ${n} finding(s) — the list drains them fix by fix, each with its own audited commit.${alreadyQueued > 0 ? ` (${alreadyQueued} already queued.)` : ""}${decideNote}`,
1667
1728
  "info",
1668
1729
  );
@@ -1702,10 +1763,13 @@ function archiveCurrentGoal(ctx: ExtensionContext, status: Status, stopReason?:
1702
1763
  const isListAuditCollect = goal.objective.includes(LIST_AUDIT_COLLECT_MARKER);
1703
1764
  // v0.34.7: the float gets a catch — ANY rejection here used to become
1704
1765
  // an uncaughtException and kill pi (darklord 2026-08-01).
1705
- if (isListAuditCollect)
1706
- void fanOutListAuditFindings(ctx).catch((err) => {
1707
- appendLedger(ctx.cwd, "list_audit_fanout_error", { error: String(err).slice(0, 200) });
1766
+ if (isListAuditCollect) {
1767
+ const fanoutCwd = ctx.cwd;
1768
+ const fanoutGeneration = sessionGeneration;
1769
+ void fanOutListAuditFindings(fanoutCwd, fanoutGeneration).catch((err) => {
1770
+ appendLedger(fanoutCwd, "list_audit_fanout_error", { error: String(err).slice(0, 200) });
1708
1771
  });
1772
+ }
1709
1773
  const advanced = activateNextListItem(ctx);
1710
1774
  // v0.26.0: the queue just EMPTIED on a completion → list-complete.
1711
1775
  if (!advanced && !isListAuditCollect) {
@@ -1738,11 +1802,17 @@ function archiveCurrentGoal(ctx: ExtensionContext, status: Status, stopReason?:
1738
1802
  * infra) → hand back to the agent: resume active + continuation, verdict
1739
1803
  * durable in auditHistory.
1740
1804
  */
1741
- async function retryStoredCompletionAudit(ctx: ExtensionContext, origin: "quota-retry" | "manual" = "quota-retry"): Promise<void> {
1805
+ async function retryStoredCompletionAudit(origin: "quota-retry" | "manual" = "quota-retry"): Promise<void> {
1742
1806
  const goal = state.goal;
1743
1807
  if (!goal?.pendingCompletion) return;
1808
+ const goalId = goal.id;
1744
1809
  if (completionAuditInFlight) return;
1745
- const liveCtx = freshCtx() ?? ctx;
1810
+ const generation = sessionGeneration;
1811
+ // Delayed audit recovery has no safe fallback: if the current generation
1812
+ // cannot be proven live, the fresh session must rehydrate the durable claim.
1813
+ const initialCtx = freshCtxForGeneration(generation);
1814
+ if (!initialCtx) return;
1815
+ let liveCtx: ExtensionContext = initialCtx;
1746
1816
  const claim = goal.pendingCompletion;
1747
1817
  updateGoal({ status: "auditing" }, liveCtx);
1748
1818
  appendLedger(liveCtx.cwd, "goal_resumed", { via: origin === "manual" ? "manual-audit" : "quota-retry-direct-audit" });
@@ -1754,6 +1824,7 @@ async function retryStoredCompletionAudit(ctx: ExtensionContext, origin: "quota-
1754
1824
  if (modelError) liveCtx.ui.notify(`Auditor model issue: ${modelError}`, "warning");
1755
1825
  latestAuditProgress = { label: "quota-retry", lastEventAt: Date.now() };
1756
1826
  completionAuditInFlight = true;
1827
+ completionAuditGeneration = generation;
1757
1828
  const auditStartMs = Date.now();
1758
1829
  let result: Awaited<ReturnType<typeof runGoalCompletionAuditor>>;
1759
1830
  try {
@@ -1761,23 +1832,36 @@ async function retryStoredCompletionAudit(ctx: ExtensionContext, origin: "quota-
1761
1832
  () =>
1762
1833
  runGoalCompletionAuditor({
1763
1834
  ctx: liveCtx,
1764
- goal: state.goal!,
1835
+ goal,
1765
1836
  completionSummary: claim.completionSummary,
1766
1837
  verificationSummary: claim.verificationSummary,
1767
1838
  model: auditorModel,
1768
1839
  thinkingLevel: (settings.auditorThinkingLevel ?? "high") as any, // may be "max" — pi ≥0.83 understands it; the dev-types predate it
1769
1840
  onProgress: (progress) => {
1841
+ const current = freshCtxForGeneration(generation);
1842
+ if (!current) return;
1770
1843
  latestAuditProgress = { currentTool: progress.currentTool, label: progress.label, elapsedMs: progress.elapsedMs, lastEventAt: Date.now() };
1771
- refreshUI(liveCtx);
1844
+ refreshUI(current);
1772
1845
  },
1773
1846
  }),
1774
- { onRetry: (err) => appendLedger(liveCtx.cwd, "audit_infra_retry", { goalId: state.goal?.id, error: err.slice(0, 200) }) },
1847
+ {
1848
+ shouldRetry: () => freshCtxForGeneration(generation) !== null,
1849
+ onRetry: (err) => {
1850
+ const current = freshCtxForGeneration(generation);
1851
+ if (current) appendLedger(current.cwd, "audit_infra_retry", { goalId, error: err.slice(0, 200) });
1852
+ },
1853
+ },
1775
1854
  ));
1776
1855
  } finally {
1777
- completionAuditInFlight = false;
1778
- latestAuditProgress = null;
1856
+ if (completionAuditGeneration === generation) {
1857
+ completionAuditInFlight = false;
1858
+ completionAuditGeneration = null;
1859
+ latestAuditProgress = null;
1860
+ }
1779
1861
  }
1780
- if (!state.goal) return; // aborted mid-audit
1862
+ const currentAfterAudit = freshCtxForGeneration(generation);
1863
+ if (!currentAfterAudit || !state.goal || state.goal.id !== goalId) return; // replacement/stale/goal boundary — fresh session rebinds durable state
1864
+ liveCtx = currentAfterAudit;
1781
1865
 
1782
1866
  // Record the run in history (same compact shape as the tool path).
1783
1867
  const auditorRan = result.output.trim().length > 0;
@@ -1834,9 +1918,9 @@ async function retryStoredCompletionAudit(ctx: ExtensionContext, origin: "quota-
1834
1918
  return;
1835
1919
  }
1836
1920
  liveCtx.ui.notify(`Auditor still quota-limited — next auto-retry in ${retryMin}m (your completion claim is stored; no action needed).`, "warning");
1837
- scheduleQuotaRetry(liveCtx, quota.retryAfterSec, result.error, () => {
1921
+ scheduleQuotaRetryForSession(liveCtx, quota.retryAfterSec, result.error, (fresh) => {
1838
1922
  if (state.goal && state.goal.status === "paused" && (state.goal.pauseReason ?? "").startsWith("auditor quota:") && state.goal.pendingCompletion) {
1839
- void retryStoredCompletionAudit(liveCtx, origin);
1923
+ void retryStoredCompletionAudit(origin);
1840
1924
  }
1841
1925
  });
1842
1926
  return;
@@ -2136,7 +2220,7 @@ async function cmdGoal(args: string, ctx: ExtensionContext): Promise<void> {
2136
2220
  },
2137
2221
  }, ctx);
2138
2222
  appendLedger(ctx.cwd, "manual_audit_requested", { goalId: state.goal.id });
2139
- void retryStoredCompletionAudit(ctx, "manual");
2223
+ void retryStoredCompletionAudit("manual");
2140
2224
  return;
2141
2225
  }
2142
2226
  if (route.name === "tweak") return cmdTweak(route.rest, ctx);
@@ -2288,8 +2372,17 @@ async function cmdResume(ctx: ExtensionContext): Promise<void> {
2288
2372
  // marker's promise ("a fresh session will resume you") is fulfilled by a
2289
2373
  // manual resume exactly as by an automatic one. (staleEntry still re-marks
2290
2374
  // below — a resume inside a stale session is a NEW interrupt.)
2375
+ const storedCompletion = state.goal.pendingCompletion;
2291
2376
  updateGoal({ status: "active", pauseReason: undefined, pauseSuggestedAction: undefined, pauseKind: undefined, pauseOptions: undefined, pauseRecommended: undefined, pauseResumeAt: undefined, interruptedAt: undefined, interruptedReason: undefined, ...(staleEntry ? { interruptedAt: nowIso(), interruptedReason: "resumed in a stale session" } : {}), ...(usage ? { usage } : {}) }, ctx);
2292
2377
  if (staleEntry) return;
2378
+ // A stored completion claim is a direct-audit resume, not an agent turn.
2379
+ // Keeping the claim while merely scheduling a continuation left manual
2380
+ // pause/resume with an ACTIVE goal that no timer would ever consume.
2381
+ if (storedCompletion) {
2382
+ ctx.ui.notify("Resuming the stored completion claim — running the isolated auditor directly (no agent turn needed).", "info");
2383
+ void retryStoredCompletionAudit("manual");
2384
+ return;
2385
+ }
2293
2386
  // v0.22.5: say what was resumed — with a non-empty list this also resumes
2294
2387
  // the queue (the active goal IS the list's head item).
2295
2388
  // v0.22.7: name WHAT was resumed — list items resume through /list.
@@ -3068,7 +3161,20 @@ function sendLoopTurn(): void {
3068
3161
  }
3069
3162
 
3070
3163
  /** agent_end hook for loop 3: measure → judge → continue or stop. */
3071
- async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
3164
+ async function runLoopTick(initialCtx: ExtensionContext, event?: any): Promise<void> {
3165
+ // v0.34.20: measurement/git work is asynchronous. Rebind the local
3166
+ // context after every await or abandon the tick; never let a replacement
3167
+ // session inherit the agent_end context.
3168
+ const generation = sessionGeneration;
3169
+ const initial = freshCtxForGeneration(generation);
3170
+ if (!initial) return;
3171
+ let ctx: ExtensionContext = initial;
3172
+ const rebind = (): boolean => {
3173
+ const current = freshCtxForGeneration(generation);
3174
+ if (!current) return false;
3175
+ ctx = current;
3176
+ return true;
3177
+ };
3072
3178
  const loop = state.loop!;
3073
3179
  // v0.15.0: token budget is an arbitrary bound; accumulate orchestrator-side.
3074
3180
  if (event?.messages) {
@@ -3076,6 +3182,7 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
3076
3182
  }
3077
3183
  const metricless = !loop.measureCmd;
3078
3184
  const value = metricless ? null : await runMeasure(ctx, loop.measureCmd!);
3185
+ if (!rebind()) return;
3079
3186
  // Hypothesis line (pi-autoresearch's good idea): the agent's stated intent
3080
3187
  // for the turn goes into the ledger, making loop history auditable.
3081
3188
  let hypothesis: string | undefined;
@@ -3100,10 +3207,12 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
3100
3207
  const iterStartHead = loop.iterMetrics?.iterationStartHead;
3101
3208
  const iterStartAt = loop.iterMetrics?.iterationStartAt;
3102
3209
  const currentHeadRes = await runGit(ctx, ["rev-parse", "HEAD"]);
3210
+ if (!rebind()) return;
3103
3211
  const currentHead = currentHeadRes.ok ? currentHeadRes.stdout : undefined;
3104
3212
  let gitCommits = 0;
3105
3213
  if (iterStartHead && currentHead && iterStartHead !== currentHead) {
3106
3214
  const countRes = await runGit(ctx, ["rev-list", "--count", `${iterStartHead}..HEAD`]);
3215
+ if (!rebind()) return;
3107
3216
  const n = Number.parseInt(countRes.stdout, 10);
3108
3217
  if (countRes.ok && Number.isFinite(n) && n > 0) gitCommits = n;
3109
3218
  }
@@ -3212,10 +3321,13 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
3212
3321
  if (loop.branchName && outcome.kind === "continue") {
3213
3322
  if (metricless || outcome.improved) {
3214
3323
  await runGit(ctx, ["add", "-A"]);
3324
+ if (!rebind()) return;
3215
3325
  const committed = await runGit(ctx, ["commit", "-m", metricless ? `pi-glla-loop: iteration ${loop.iteration}` : `pi-glla-loop: iteration ${loop.iteration} (${loop.direction}=${loop.bestValue})`]);
3326
+ if (!rebind()) return;
3216
3327
  appendLedger(ctx.cwd, "loop_git", { action: "commit", iteration: loop.iteration, ok: committed.ok });
3217
3328
  } else {
3218
3329
  const reset = await runGit(ctx, ["reset", "--hard", "HEAD"]);
3330
+ if (!rebind()) return;
3219
3331
  appendLedger(ctx.cwd, "loop_git", { action: "reset", iteration: loop.iteration, ok: reset.ok });
3220
3332
  }
3221
3333
  persistState(ctx);
@@ -3229,6 +3341,7 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
3229
3341
  loop.stopReason = `stuck — ${loop.lastStuckReason} (${loop.consecutiveStuck} consecutive interventions)`;
3230
3342
  persistState(ctx);
3231
3343
  await finishLoopGit(ctx, loop);
3344
+ if (!rebind()) return;
3232
3345
  ctx.ui.notify(`Loop stopped: ${loop.stopReason}. ${loop.history.length} iterations recorded.`, "warning");
3233
3346
  appendLedger(ctx.cwd, "loop_stopped", { reason: loop.stopReason, iterations: loop.iteration, best: loop.bestValue });
3234
3347
  notifyExternal(ctx, `Loop stopped: ${loop.stopReason}`);
@@ -3265,6 +3378,7 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
3265
3378
  }
3266
3379
  }
3267
3380
  await finishLoopGit(ctx, loop);
3381
+ if (!rebind()) return;
3268
3382
  ctx.ui.notify(`Loop stopped: ${outcome.reason}. ${loop.history.length} iterations recorded.`, "info");
3269
3383
  appendLedger(ctx.cwd, "loop_stopped", { reason: outcome.reason, iterations: loop.iteration, best: loop.bestValue });
3270
3384
  notifyExternal(ctx, `Loop stopped: ${outcome.reason}`);
@@ -3277,10 +3391,17 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
3277
3391
  * where the work lives and how to merge it. Scratch branch is never deleted. */
3278
3392
  async function finishLoopGit(ctx: ExtensionContext, loop: LoopState): Promise<void> {
3279
3393
  if (!loop.branchName) return;
3394
+ const generation = sessionGeneration;
3280
3395
  // Uncommitted remnants (final stalled iterations were reset already, but be safe).
3281
3396
  await runGit(ctx, ["reset", "--hard", "HEAD"]);
3397
+ const afterReset = freshCtxForGeneration(generation);
3398
+ if (!afterReset) return;
3399
+ ctx = afterReset;
3282
3400
  if (loop.originalBranch) {
3283
3401
  await runGit(ctx, ["checkout", loop.originalBranch]);
3402
+ const afterCheckout = freshCtxForGeneration(generation);
3403
+ if (!afterCheckout) return;
3404
+ ctx = afterCheckout;
3284
3405
  }
3285
3406
  ctx.ui.notify(
3286
3407
  `Loop work is on branch ${loop.branchName} (${loop.iteration} iterations, best ${loop.bestValue ?? "n/a"}).\nMerge with: git merge ${loop.branchName} — or delete with: git branch -D ${loop.branchName}`,
@@ -3532,7 +3653,11 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
3532
3653
  clearLoopTimer();
3533
3654
  state.loop = { ...state.loop, active: false, stopReason: state.loop.stopReason ?? `stopped by user (/loop ${sub})` };
3534
3655
  persistState(ctx);
3656
+ const stopGeneration = sessionGeneration;
3535
3657
  await finishLoopGit(ctx, state.loop);
3658
+ const afterFinish = freshCtxForGeneration(stopGeneration);
3659
+ if (!afterFinish) return;
3660
+ ctx = afterFinish;
3536
3661
  appendLedger(ctx.cwd, "loop_stopped", { reason: "user", iterations: state.loop.iteration, best: state.loop.bestValue });
3537
3662
  ctx.ui.notify(
3538
3663
  `Loop stopped after ${state.loop.iteration} iterations. Best: ${state.loop.bestValue ?? "n/a"}.`,
@@ -3553,7 +3678,11 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
3553
3678
  const reason = loopFinishStopReason(rest);
3554
3679
  state.loop = { ...state.loop, active: false, stopReason: reason };
3555
3680
  persistState(ctx);
3681
+ const finishGeneration = sessionGeneration;
3556
3682
  await finishLoopGit(ctx, state.loop);
3683
+ const afterFinish = freshCtxForGeneration(finishGeneration);
3684
+ if (!afterFinish) return;
3685
+ ctx = afterFinish;
3557
3686
  appendLedger(ctx.cwd, "loop_stopped", { reason, iterations: state.loop.iteration, best: state.loop.bestValue });
3558
3687
  ctx.ui.notify(
3559
3688
  `Loop finished (${reason}) after ${state.loop.iteration} iterations. Best: ${state.loop.bestValue ?? "n/a"}.`,
@@ -3682,7 +3811,34 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
3682
3811
  // Tools exposed to the agent
3683
3812
  // =================================================================
3684
3813
 
3685
- function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3814
+ const STALE_TOOL_CONTEXT_MESSAGE =
3815
+ "This tool call crossed a session replacement before it could run. No stale context was used; wait for a fresh session_start and retry.";
3816
+
3817
+ function staleToolResult(): { content: Array<{ type: "text"; text: string }>; details: Record<string, never> } {
3818
+ return { content: [{ type: "text", text: STALE_TOOL_CONTEXT_MESSAGE }], details: {} };
3819
+ }
3820
+
3821
+ /**
3822
+ * v0.34.20: registerAgentTools runs once per extension instance, but pi
3823
+ * invokes the registered tool with the current event context. Never use the
3824
+ * context captured when the tools were registered after a reload/rebind.
3825
+ * Prefer the invocation context, validate it cheaply, and fall back only to
3826
+ * the current fresh context — never to the registration-time ctx.
3827
+ */
3828
+ function currentToolContext(execCtx: unknown): ExtensionContext | null {
3829
+ const candidate = execCtx as ExtensionContext | undefined;
3830
+ if (candidate) {
3831
+ try {
3832
+ candidate.isIdle();
3833
+ return candidate;
3834
+ } catch {
3835
+ // The invocation itself may be a late event; try the current binding.
3836
+ }
3837
+ }
3838
+ return freshCtx();
3839
+ }
3840
+
3841
+ function registerAgentTools(pi: any): void {
3686
3842
  pi.registerTool(defineTool({
3687
3843
  name: "complete_goal",
3688
3844
  label: "Complete goal",
@@ -3695,6 +3851,10 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3695
3851
  async execute(_id, params, signal, _onUpdate, execCtx) {
3696
3852
  const foreign0 = foreignToolGuard(execCtx);
3697
3853
  if (foreign0) return { content: [{ type: "text", text: foreign0 }], details: {} };
3854
+ const toolCtx = currentToolContext(execCtx);
3855
+ if (!toolCtx) return staleToolResult();
3856
+ let ctx: ExtensionContext = toolCtx;
3857
+ const auditGeneration = sessionGeneration;
3698
3858
  if (!state.goal || state.goal.status !== "active") {
3699
3859
  return { content: [{ type: "text", text: "No active goal." }], details: {} };
3700
3860
  }
@@ -3710,7 +3870,22 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3710
3870
  appendLedger(ctx.cwd, "goal_tweaked", { via: "complete_goal.newObjective", from: oldObjective.slice(0, 200), to: cleanObj.slice(0, 200) });
3711
3871
  ctx.ui.notify(`Objective updated (complete_goal newObjective): ${cleanObj.slice(0, 80)}`, "info");
3712
3872
  }
3713
- updateGoal({ status: "auditing", pendingTasks: undefined }, ctx);
3873
+ // v0.34.20: persist the completion claim BEFORE the isolated auditor
3874
+ // starts. If session replacement lands during the audit, a fresh
3875
+ // session can recover the exact claim instead of leaving an untracked
3876
+ // goal stuck in `auditing`.
3877
+ updateGoal({
3878
+ status: "auditing",
3879
+ pendingTasks: undefined,
3880
+ pendingCompletion: {
3881
+ completionSummary: p.completionSummary,
3882
+ verificationSummary: p.verificationSummary,
3883
+ at: nowIso(),
3884
+ },
3885
+ }, ctx);
3886
+ const auditGoal = state.goal;
3887
+ if (!auditGoal) return staleToolResult();
3888
+ const auditGoalId = auditGoal.id;
3714
3889
  const settings = loadSettings(ctx.cwd);
3715
3890
  const { model: auditorModel, error: modelError, via } = resolveAuditorModel(ctx, settings.auditorModel, settings.auditorModelFallback, settings.auditorSameSessionSwap !== false);
3716
3891
  if (modelError) {
@@ -3723,20 +3898,22 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3723
3898
  const runAudit = () =>
3724
3899
  runGoalCompletionAuditor({
3725
3900
  ctx,
3726
- goal: state.goal!,
3901
+ goal: auditGoal,
3727
3902
  completionSummary: p.completionSummary,
3728
3903
  verificationSummary: p.verificationSummary,
3729
3904
  model: auditorModel,
3730
3905
  thinkingLevel: (settings.auditorThinkingLevel ?? "high") as any, // may be "max" — pi ≥0.83 understands it; the dev-types predate it
3731
3906
  signal: signal ?? undefined,
3732
3907
  onProgress: (progress) => {
3908
+ const current = freshCtxForGeneration(auditGeneration);
3909
+ if (!current) return;
3733
3910
  latestAuditProgress = {
3734
3911
  currentTool: progress.currentTool,
3735
3912
  label: progress.label,
3736
3913
  elapsedMs: progress.elapsedMs,
3737
3914
  lastEventAt: Date.now(),
3738
3915
  };
3739
- refreshUI(ctx);
3916
+ refreshUI(current);
3740
3917
  },
3741
3918
  });
3742
3919
  // v0.25.4 (post-audit fix): a retriable infra failure (stream error,
@@ -3745,19 +3922,33 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3745
3922
  // (retried once)". Neither attempt is a verdict on the work.
3746
3923
  const auditStartMs = Date.now();
3747
3924
  completionAuditInFlight = true;
3925
+ completionAuditGeneration = auditGeneration;
3748
3926
  let result: Awaited<ReturnType<typeof runAudit>>;
3749
3927
  let retriedOnce = false;
3750
3928
  try {
3751
3929
  ({ result, retriedOnce } = await runWithInfraRetry(runAudit, {
3930
+ shouldRetry: () => freshCtxForGeneration(auditGeneration) !== null,
3752
3931
  onRetry: (err) => {
3932
+ const current = freshCtxForGeneration(auditGeneration);
3933
+ if (!current) return;
3753
3934
  latestAuditProgress = { label: `infra error (${err.slice(0, 40)}) — retrying once`, lastEventAt: Date.now() };
3754
- refreshUI(ctx);
3755
- appendLedger(ctx.cwd, "audit_infra_retry", { goalId: state.goal?.id, error: err.slice(0, 200) });
3935
+ refreshUI(current);
3936
+ appendLedger(current.cwd, "audit_infra_retry", { goalId: auditGoalId, error: err.slice(0, 200) });
3756
3937
  },
3757
3938
  }));
3758
3939
  } finally {
3759
- completionAuditInFlight = false;
3940
+ if (completionAuditGeneration === auditGeneration) {
3941
+ completionAuditInFlight = false;
3942
+ completionAuditGeneration = null;
3943
+ latestAuditProgress = null;
3944
+ }
3945
+ }
3946
+ const auditContextAfterRun = freshCtxForGeneration(auditGeneration);
3947
+ if (!auditContextAfterRun || !state.goal || state.goal.id !== auditGoalId) {
3948
+ if (completionAuditGeneration === auditGeneration) latestAuditProgress = null;
3949
+ return staleToolResult();
3760
3950
  }
3951
+ ctx = auditContextAfterRun;
3761
3952
  const auditDurationMs = Date.now() - auditStartMs;
3762
3953
  latestAuditProgress = null;
3763
3954
  // Audit history: record REAL verdicts only — a non-empty report is the
@@ -3826,7 +4017,10 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3826
4017
  // Escape hatch: the user aborted the audit (Esc). Offer the explicit
3827
4018
  // choice — complete WITHOUT audit, or keep working. (pi-goal-x parity.)
3828
4019
  if (result.error === "Auditor aborted.") {
3829
- updateGoal({ status: "active", auditHistory: history, pauseReason: "audit aborted by user (Esc)" }, ctx);
4020
+ updateGoal({ status: "active", auditHistory: history, pendingCompletion: undefined, pauseReason: "audit aborted by user (Esc)" }, ctx);
4021
+ const abortConfirmCtx = freshCtxForGeneration(auditGeneration);
4022
+ if (!abortConfirmCtx) return staleToolResult();
4023
+ ctx = abortConfirmCtx;
3830
4024
  let completeAnyway = false;
3831
4025
  try {
3832
4026
  completeAnyway = await ctx.ui.confirm(
@@ -3836,8 +4030,11 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3836
4030
  } catch {
3837
4031
  completeAnyway = false;
3838
4032
  }
4033
+ const afterAbortConfirmCtx = freshCtxForGeneration(auditGeneration);
4034
+ if (!afterAbortConfirmCtx) return staleToolResult();
4035
+ ctx = afterAbortConfirmCtx;
3839
4036
  if (completeAnyway) {
3840
- updateGoal({ auditHistory: history }, ctx);
4037
+ updateGoal({ auditHistory: history, pendingCompletion: undefined }, ctx);
3841
4038
  archiveCurrentGoal(ctx, "complete", "completed without audit (user choice after Esc)");
3842
4039
  return { content: [{ type: "text", text: "Goal marked complete without audit (user choice)." }], details: {} };
3843
4040
  }
@@ -3849,7 +4046,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3849
4046
  }
3850
4047
 
3851
4048
  if (result.approved) {
3852
- updateGoal({ auditHistory: history }, ctx);
4049
+ updateGoal({ auditHistory: history, pendingCompletion: undefined }, ctx);
3853
4050
  const objective = state.goal.objective;
3854
4051
  archiveCurrentGoal(ctx, "complete", `auditor ${result.model} approved`);
3855
4052
  notifyExternal(ctx, `Goal complete (auditor approved): ${objective.slice(0, 120)}`);
@@ -3871,6 +4068,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3871
4068
  updateGoal({
3872
4069
  status: "active",
3873
4070
  auditHistory: history,
4071
+ pendingCompletion: undefined,
3874
4072
  pauseReason: `auditor verdict: IMPOSSIBLE (partial) — ${reason}`,
3875
4073
  pauseSuggestedAction: "Narrow the objective past the impossible part (complete_goal newObjective or /goal tweak) and continue",
3876
4074
  }, ctx);
@@ -3888,6 +4086,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3888
4086
  updateGoal({
3889
4087
  status: "paused",
3890
4088
  auditHistory: history,
4089
+ pendingCompletion: undefined,
3891
4090
  pauseKind: "decision",
3892
4091
  pauseOptions: ["Tweak the objective — /goal tweak <new text>", "Cancel the goal (/goal cancel)"],
3893
4092
  pauseRecommended: 1,
@@ -3933,7 +4132,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3933
4132
  pauseSuggestedAction: `Quota auto-retry in ${retryMin}m — or /goal resume to retry now`,
3934
4133
  }, ctx);
3935
4134
  appendLedger(ctx.cwd, "goal_paused", { reason: `auditor quota: retry in ${quota.retryAfterSec}s (${quota.fromUpstream ? "upstream hint" : "default"})` });
3936
- scheduleQuotaRetry(ctx, quota.retryAfterSec, result.error, () => {
4135
+ scheduleQuotaRetryForSession(ctx, quota.retryAfterSec, result.error, (fresh) => {
3937
4136
  // Re-check: only auto-resume if STILL paused for the quota
3938
4137
  // reason (a user /goal pause during the window is not stomped).
3939
4138
  if (state.goal && state.goal.status === "paused" && (state.goal.pauseReason ?? "").startsWith("auditor quota:")) {
@@ -3941,15 +4140,15 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3941
4140
  // agent is not needed to re-submit an unchanged claim, and
3942
4141
  // re-engaging it produced hallucinated-closure loops.
3943
4142
  if (state.goal.pendingCompletion) {
3944
- void retryStoredCompletionAudit(ctx);
4143
+ void retryStoredCompletionAudit();
3945
4144
  return;
3946
4145
  }
3947
- updateGoal({ status: "active" }, ctx);
3948
- appendLedger(ctx.cwd, "goal_resumed", { via: "quota-retry" });
3949
- if (resolveEffectiveAggressiveSettings(loadSettings(ctx.cwd)).aggressiveMode) {
3950
- ctx.ui.notify("Auto-resume fired (event: auditor quota window elapsed). Continue working.", "info");
4146
+ updateGoal({ status: "active" }, fresh);
4147
+ appendLedger(fresh.cwd, "goal_resumed", { via: "quota-retry" });
4148
+ if (resolveEffectiveAggressiveSettings(loadSettings(fresh.cwd)).aggressiveMode) {
4149
+ fresh.ui.notify("Auto-resume fired (event: auditor quota window elapsed). Continue working.", "info");
3951
4150
  }
3952
- scheduleContinuation(ctx, true);
4151
+ scheduleContinuation(fresh, true);
3953
4152
  }
3954
4153
  });
3955
4154
  return {
@@ -3969,6 +4168,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3969
4168
  updateGoal({
3970
4169
  status: "paused",
3971
4170
  auditHistory: history,
4171
+ pendingCompletion: undefined,
3972
4172
  auditInfraStreak: infraStreak,
3973
4173
  pauseKind: "error",
3974
4174
  pauseReason: `auditor infrastructure failed ${infraStreak}× in a row — the auditor model is likely broken OR a verification command is hanging (ssh/sudo/long test runs stall the stream) (last: ${result.error.slice(0, 120)})`,
@@ -3988,6 +4188,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
3988
4188
  updateGoal({
3989
4189
  status: "active",
3990
4190
  auditHistory: history,
4191
+ pendingCompletion: undefined,
3991
4192
  auditInfraStreak: infraStreak,
3992
4193
  pauseReason: `auditor infrastructure${retriedOnce ? " (retried once)" : ""}: ${result.error}`,
3993
4194
  pauseSuggestedAction: "Fix the auditor model (/glla model=provider/id) and call complete_goal again — your work was NOT judged",
@@ -4012,6 +4213,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4012
4213
  updateGoal({
4013
4214
  status: "active",
4014
4215
  auditHistory: history,
4216
+ pendingCompletion: undefined,
4015
4217
  pauseReason: `regression shield: auditor approved, but evidence never referenced ${missing.length} contract item(s)`,
4016
4218
  pauseSuggestedAction: "call complete_goal again — the next auditor run is told exactly which items to quote evidence for",
4017
4219
  }, ctx);
@@ -4057,6 +4259,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4057
4259
  updateGoal({
4058
4260
  status: "active",
4059
4261
  auditHistory: history,
4262
+ pendingCompletion: undefined,
4060
4263
  pendingTasks,
4061
4264
  pauseReason: `auditor disapproved ${trailingDisapprovals}× consecutively (cap ${auditCap}) — aggressiveMode: continuing with TODOs`,
4062
4265
  }, ctx);
@@ -4077,6 +4280,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4077
4280
  updateGoal({
4078
4281
  status: "paused",
4079
4282
  auditHistory: history,
4283
+ pendingCompletion: undefined,
4080
4284
  pauseKind: "decision",
4081
4285
  pauseOptions: ["Fix the disapproval gap, then continue (/goal resume)", "Tweak the objective — /goal tweak <new text>", "Cancel the goal (/goal cancel)"],
4082
4286
  pauseRecommended: 1,
@@ -4098,6 +4302,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4098
4302
  updateGoal({
4099
4303
  status: "active",
4100
4304
  auditHistory: history,
4305
+ pendingCompletion: undefined,
4101
4306
  pauseReason: "auditor disapproved",
4102
4307
  pauseSuggestedAction: "Inspect auditor feedback and fix the actual gap before calling complete_goal again",
4103
4308
  }, ctx);
@@ -4127,6 +4332,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4127
4332
  async execute(_id, params, _signal, _onUpdate, execCtx) {
4128
4333
  const foreign1 = foreignToolGuard(execCtx);
4129
4334
  if (foreign1) return { content: [{ type: "text", text: foreign1 }], details: {} };
4335
+ const ctx = currentToolContext(execCtx);
4336
+ if (!ctx) return staleToolResult();
4130
4337
  const p = params as { reason: string; suggestedAction?: string; kind?: "decision" | "error" | "wait" | "blocked"; options?: string[]; recommended?: number; resumeAt?: string };
4131
4338
  if (!state.goal) return { content: [{ type: "text", text: "No active goal." }], details: {} };
4132
4339
  updateGoal({
@@ -4158,6 +4365,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4158
4365
  async execute(_id, params, _signal, _onUpdate, execCtx) {
4159
4366
  const foreign7 = foreignToolGuard(execCtx);
4160
4367
  if (foreign7) return { content: [{ type: "text", text: foreign7 }], details: {} };
4368
+ const ctx = currentToolContext(execCtx);
4369
+ if (!ctx) return staleToolResult();
4161
4370
  const p = params as { id: string };
4162
4371
  if (!state.goal || !state.goal.taskList) {
4163
4372
  return { content: [{ type: "text", text: "No task list in this goal." }], details: {} };
@@ -4188,6 +4397,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4188
4397
  async execute(_id, params, _signal, _onUpdate, execCtx) {
4189
4398
  const foreign8 = foreignToolGuard(execCtx);
4190
4399
  if (foreign8) return { content: [{ type: "text", text: foreign8 }], details: {} };
4400
+ const ctx = currentToolContext(execCtx);
4401
+ if (!ctx) return staleToolResult();
4191
4402
  const p = params as { id: string; status: "pending" | "in_progress" | "complete" };
4192
4403
  if (!state.goal || !state.goal.taskList) {
4193
4404
  return { content: [{ type: "text", text: "No task list in this goal." }], details: {} };
@@ -4220,13 +4431,14 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4220
4431
  const foreign2 = foreignToolGuard(execCtx);
4221
4432
  if (foreign2) return { content: [{ type: "text", text: foreign2 }], details: {} };
4222
4433
  const p = params as { objective: string; verificationContract?: string; items?: string[] };
4434
+ const liveCtx = currentToolContext(execCtx);
4435
+ if (!liveCtx) return staleToolResult();
4223
4436
  if (draftingTarget !== "goal" && draftingTarget !== "list") {
4224
4437
  return {
4225
4438
  content: [{ type: "text", text: "Not in goal drafting mode. The user starts drafting with /goal or /list add (no args), or activates directly with /goal <objective>." }],
4226
4439
  details: {},
4227
4440
  };
4228
4441
  }
4229
- const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
4230
4442
  // v0.28.14: one-active-thing EARLY guard — refuse the whole interview
4231
4443
  // when a loop is live (the post-confirm backstop below stays: state
4232
4444
  // can change mid-interview).
@@ -4408,6 +4620,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4408
4620
  const foreign3 = foreignToolGuard(execCtx);
4409
4621
  if (foreign3) return { content: [{ type: "text", text: foreign3 }], details: {} };
4410
4622
  const p = params as { target: string; measureCmd?: string; direction?: "min" | "max"; window?: number; max?: number; time?: number; tokens?: number; branch?: boolean };
4623
+ const liveCtx = currentToolContext(execCtx);
4624
+ if (!liveCtx) return staleToolResult();
4411
4625
  if (draftingTarget !== "loop") {
4412
4626
  return {
4413
4627
  content: [{ type: "text", text: "You cannot start or draft a loop — only the user can, from the slash bar (the Confirm is the product). Do NOT write draft files or wait for the user to say 'start' in chat; that dead-ends. Instead hand the user the exact command: /loop start \"<target>\" (bare = infinite metricless; add measure=\"<cmd>\" direction=min|max for a metric loop), or /loop respec to reconcile against the root spec, or /loop with no args to draft interactively." }],
@@ -4433,7 +4647,6 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4433
4647
  if (!metricless && p.direction !== "min" && p.direction !== "max") {
4434
4648
  return { content: [{ type: "text", text: 'direction=min|max is required for a measured loop (omit measureCmd or pass "none" for a metricless spec loop).' }], details: {} };
4435
4649
  }
4436
- const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
4437
4650
  // v0.28.14: one-active-thing — refuse to even test-run a loop measure
4438
4651
  // while a goal/list-item is active (the /loop start COMMAND guards
4439
4652
  // this; the tool path used to skip it and stack a loop over a goal).
@@ -4531,7 +4744,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4531
4744
  const foreign4 = foreignToolGuard(execCtx);
4532
4745
  if (foreign4) return { content: [{ type: "text", text: foreign4 }], details: {} };
4533
4746
  const p = params as { target?: string; measureCmd?: string; specText?: string; specAppend?: string; rationale: string };
4534
- const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
4747
+ const liveCtx = currentToolContext(execCtx);
4748
+ if (!liveCtx) return staleToolResult();
4535
4749
  const loop = state.loop;
4536
4750
  if (!loop?.active) {
4537
4751
  return { content: [{ type: "text", text: "No active loop to refine. propose_loop_refine is only valid while a loop is running." }], details: {} };
@@ -4627,6 +4841,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4627
4841
  const foreign5 = foreignToolGuard(execCtx);
4628
4842
  if (foreign5) return { content: [{ type: "text", text: foreign5 }], details: {} };
4629
4843
  const p = params as { items: string[] };
4844
+ const liveCtx = currentToolContext(execCtx);
4845
+ if (!liveCtx) return staleToolResult();
4630
4846
  if (listMutationBlocked(draftingTarget)) {
4631
4847
  return { content: [{ type: "text", text: LIST_DRAFTING_BLOCK_MESSAGE }], details: {} };
4632
4848
  }
@@ -4634,7 +4850,6 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4634
4850
  return { content: [{ type: "text", text: "No items given." }], details: {} };
4635
4851
  }
4636
4852
  const clean = p.items.map((t) => t.trim()).filter((t) => t.length > 0);
4637
- const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
4638
4853
  const wasIdle = !state.goal || state.goal.status === "complete" || state.goal.status === "aborted";
4639
4854
  const n = enqueueItems(liveCtx, clean, "agent list_add");
4640
4855
  return {
@@ -4660,6 +4875,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4660
4875
  const foreign6 = foreignToolGuard(execCtx);
4661
4876
  if (foreign6) return { content: [{ type: "text", text: foreign6 }], details: {} };
4662
4877
  const p = params as { n: number };
4878
+ const liveCtx = currentToolContext(execCtx);
4879
+ if (!liveCtx) return staleToolResult();
4663
4880
  if (listMutationBlocked(draftingTarget)) {
4664
4881
  return { content: [{ type: "text", text: LIST_DRAFTING_BLOCK_MESSAGE }], details: {} };
4665
4882
  }
@@ -4667,7 +4884,6 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4667
4884
  if (!Number.isInteger(n) || n < 1) {
4668
4885
  return { content: [{ type: "text", text: "n must be a positive integer (1-based position)." }], details: {} };
4669
4886
  }
4670
- const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
4671
4887
  // v0.28.14: one-active-thing — a list item must not jump a live loop.
4672
4888
  if (isLoopActive()) {
4673
4889
  return { content: [{ type: "text", text: "A loop is active — one active thing at a time. The user must /loop stop it before a list item can activate." }], details: {} };
@@ -4729,11 +4945,12 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
4729
4945
  return { content: [{ type: "text", text: "A task list already exists. Use update_task_status / complete_task to work it." }], details: {} };
4730
4946
  }
4731
4947
  const p = params as { tasks: TaskProposal[] };
4948
+ const liveCtx = currentToolContext(execCtx);
4949
+ if (!liveCtx) return staleToolResult();
4732
4950
  const invalid = validateTaskProposal(p.tasks);
4733
4951
  if (invalid) {
4734
4952
  return { content: [{ type: "text", text: invalid }], details: {} };
4735
4953
  }
4736
- const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
4737
4954
  const preview = p.tasks.map((t, i) => {
4738
4955
  const subs = (t.subtasks ?? []).map((s, j) => ` ${i + 1}.${j + 1} ${s}`).join("\n");
4739
4956
  return `${i + 1}. ${t.title}` + (subs ? `\n${subs}` : "");
@@ -5533,7 +5750,11 @@ async function cmdGllaWipe(ctx: ExtensionContext): Promise<void> {
5533
5750
  if (loop) {
5534
5751
  clearLoopTimer();
5535
5752
  state.loop = undefined;
5753
+ const wipeGeneration = sessionGeneration;
5536
5754
  await finishLoopGit(ctx, loop);
5755
+ const afterFinish = freshCtxForGeneration(wipeGeneration);
5756
+ if (!afterFinish) return;
5757
+ ctx = afterFinish;
5537
5758
  appendLedger(ctx.cwd, "loop_stopped", { reason: "user wipe (/glla wipe)", iterations: loop.iteration, best: loop.bestValue });
5538
5759
  }
5539
5760
  persistState(ctx);
@@ -6207,7 +6428,9 @@ export default function (pi: ExtensionAPI): void {
6207
6428
  // Tool registration is lazy: done on the first session event, when a
6208
6429
  // context exists. Tools show even without an active goal (and return
6209
6430
  // "no active goal" if called).
6210
- let registeredCtx: ExtensionContext | null = null;
6431
+ // Tool definitions are re-registered at lifecycle boundaries; the current
6432
+ // invocation context is resolved inside each execute handler.
6433
+ let toolsRegistered = false;
6211
6434
 
6212
6435
  // v0.24.5 tool-visibility self-heal: surface the notify exactly once
6213
6436
  // per session so the user learns about an external allowlist once and
@@ -6326,7 +6549,7 @@ export default function (pi: ExtensionAPI): void {
6326
6549
 
6327
6550
  // v0.15.1: ask_user_question answers arrive as tool results, not chat
6328
6551
  // messages — count answered (non-cancelled) questionnaires as replies too.
6329
- pi.on("tool_result", async (event: any) => {
6552
+ pi.on("tool_result", async (event: any, eventCtx: ExtensionContext) => {
6330
6553
  if (sessionHandoffPending || extensionApiStale || staleTerminalDone || zombieStoodDown) return;
6331
6554
  noteToolResult(event); // v0.33.0: slim widget "last action" feed
6332
6555
  // v0.24.0: roll loop tool-result fingerprints (same-tool-same-result
@@ -6364,11 +6587,14 @@ export default function (pi: ExtensionAPI): void {
6364
6587
  // HIT QUOTA ERRORS section carries the full guidance.
6365
6588
  if (isSubagentQuotaResult(String(event?.toolName ?? ""), Boolean(event?.isError ?? event?.error), event?.output ?? event?.result ?? event?.details ?? "")) {
6366
6589
  const errText = typeof (event?.output ?? event?.result) === "string" ? (event?.output ?? event?.result) : JSON.stringify(event?.output ?? event?.result ?? event?.details ?? "");
6367
- appendLedger(registeredCtx?.cwd ?? process.cwd(), "subagent_quota_error", { error: String(errText).slice(0, 200) });
6368
- registeredCtx?.ui.notify(
6369
- "Subagent hit a quota error (403/limit). Repair: re-spawn with an explicit model= on your quota pool, or do the work inline — see the continuation prompt's WHEN SUBAGENTS HIT QUOTA ERRORS. Explore's upstream haiku pin is the usual cause (pi-subagents#175); glla's inherit-parent strategy removes it for NEW sessions.",
6370
- "warning",
6371
- );
6590
+ const current = currentToolContext(eventCtx);
6591
+ if (current) {
6592
+ appendLedger(current.cwd, "subagent_quota_error", { error: String(errText).slice(0, 200) });
6593
+ current.ui.notify(
6594
+ "Subagent hit a quota error (403/limit). Repair: re-spawn with an explicit model= on your quota pool, or do the work inline — see the continuation prompt's WHEN SUBAGENTS HIT QUOTA ERRORS. Explore's upstream haiku pin is the usual cause (pi-subagents#175); glla's inherit-parent strategy removes it for NEW sessions.",
6595
+ "warning",
6596
+ );
6597
+ }
6372
6598
  }
6373
6599
  if (draftingTarget === null) return;
6374
6600
  if (askUserQuestionAnswered(String(event?.toolName ?? ""), event?.details)) {
@@ -6389,7 +6615,7 @@ export default function (pi: ExtensionAPI): void {
6389
6615
  writeSessionHandoff(ctx, shutdownReason);
6390
6616
  sessionReplacementUntil = Date.now() + SESSION_REBIND_GRACE_MS;
6391
6617
  clearSessionOwnedTimers();
6392
- registeredCtx = null;
6618
+ toolsRegistered = false;
6393
6619
  toolHealNotified = false;
6394
6620
  });
6395
6621
 
@@ -6405,6 +6631,11 @@ export default function (pi: ExtensionAPI): void {
6405
6631
  staleTerminalDone = false; // v0.33.1: a rebound session can go terminal again
6406
6632
  zombieStoodDown = false;
6407
6633
  sessionGeneration++;
6634
+ // An auditor belonging to the disposed generation cannot block the fresh
6635
+ // session's recovery gate; its finally block is generation-guarded too.
6636
+ completionAuditInFlight = false;
6637
+ completionAuditGeneration = null;
6638
+ latestAuditProgress = null;
6408
6639
  // Ephemeral watchdog counters belong to the old session, not the
6409
6640
  // persisted goal. Reset them so a stale boundary cannot make the next
6410
6641
  // fresh session inherit a false stall count.
@@ -6446,9 +6677,9 @@ export default function (pi: ExtensionAPI): void {
6446
6677
  heldLoop: state.loop && (state.loop.active || state.loop.stopReason === HELD_ON_RESTORE) ? state.loop.target.slice(0, 60) : undefined,
6447
6678
  };
6448
6679
  carryoverResolved = !(carryoverSnapshot.pausedGoal || carryoverSnapshot.listCount > 0 || carryoverSnapshot.heldLoop);
6449
- if (!registeredCtx) {
6450
- registerAgentTools(pi, ctx);
6451
- registeredCtx = ctx;
6680
+ if (!toolsRegistered) {
6681
+ registerAgentTools(pi);
6682
+ toolsRegistered = true;
6452
6683
  }
6453
6684
  ensureAgentToolsActive(pi, ctx);
6454
6685
  warnOnCommandCollision(ctx);
@@ -6704,9 +6935,9 @@ export default function (pi: ExtensionAPI): void {
6704
6935
  t.turns++;
6705
6936
  state.goal.telemetry = t;
6706
6937
  }
6707
- if (!registeredCtx) {
6708
- registerAgentTools(pi, ctx);
6709
- registeredCtx = ctx;
6938
+ if (!toolsRegistered) {
6939
+ registerAgentTools(pi);
6940
+ toolsRegistered = true;
6710
6941
  }
6711
6942
  ensureAgentToolsActive(pi, ctx);
6712
6943
  // v0.27.3: nudge accounting — substantive analytical turns (long, novel
@@ -6894,16 +7125,16 @@ export default function (pi: ExtensionAPI): void {
6894
7125
  notifyExternal(ctx, `${goalNoun()} parked: provider erroring across 6 error-brake cycles — hourly top-of-hour probes scheduled.`);
6895
7126
  appendLedger(ctx.cwd, "error_brake_capped", { streak: brakeStreak, reason });
6896
7127
  const probeMs = msUntilNextHourBoundary(Date.now());
6897
- scheduleQuotaRetry(ctx, probeMs / 1000, reason, () => {
7128
+ scheduleQuotaRetryForSession(ctx, probeMs / 1000, reason, (fresh) => {
6898
7129
  // Re-check: only probe if STILL parked by the error-brake cap —
6899
7130
  // a user pause/resume/cancel meanwhile is never stomped.
6900
7131
  if (state.goal && state.goal.status === "paused" && state.goal.pauseKind === "error"
6901
7132
  && (state.goal.pauseReason ?? "").includes("error-brakes in a row")) {
6902
- appendLedger(ctx.cwd, "hourly_rate_probe", { goalId: state.goal.id, streak: state.goal.errorBrakeStreak ?? 0 });
6903
- updateGoal({ status: "active" }, ctx);
6904
- appendLedger(ctx.cwd, "goal_resumed", { via: "hourly-rate-probe" });
6905
- ctx.ui.notify("Hourly probe: resuming (rate-limit windows typically expire at the top of the hour).", "info");
6906
- scheduleContinuation(ctx, true);
7133
+ appendLedger(fresh.cwd, "hourly_rate_probe", { goalId: state.goal.id, streak: state.goal.errorBrakeStreak ?? 0 });
7134
+ updateGoal({ status: "active" }, fresh);
7135
+ appendLedger(fresh.cwd, "goal_resumed", { via: "hourly-rate-probe" });
7136
+ fresh.ui.notify("Hourly probe: resuming (rate-limit windows typically expire at the top of the hour).", "info");
7137
+ scheduleContinuation(fresh, true);
6907
7138
  }
6908
7139
  }, "Hourly rate-limit probe");
6909
7140
  return;
@@ -6925,14 +7156,14 @@ export default function (pi: ExtensionAPI): void {
6925
7156
  ctx.ui.notify(`Goal paused: ${reason}.${quotaWall ? " Quota/rate-limit wall — resuming won't help until the window resets; switch /model to continue now." : ""}`, "warning");
6926
7157
  notifyExternal(ctx, `Goal paused: ${reason}.`);
6927
7158
  appendLedger(ctx.cwd, "goal_paused", { reason });
6928
- scheduleQuotaRetry(ctx, cooldownMs / 1000, reason, () => {
7159
+ scheduleQuotaRetryForSession(ctx, cooldownMs / 1000, reason, (fresh) => {
6929
7160
  // Re-check: only auto-resume if STILL paused for the error brake
6930
7161
  // (a user /goal pause during the window is not stomped).
6931
7162
  if (state.goal && state.goal.status === "paused" && (state.goal.pauseReason ?? "").startsWith("5 consecutive errors")) {
6932
- updateGoal({ status: "active" }, ctx);
6933
- appendLedger(ctx.cwd, "goal_resumed", { via: "error-brake-retry" });
6934
- ctx.ui.notify("Auto-resumed after the 5-error brake (cooldown elapsed).", "info");
6935
- scheduleContinuation(ctx, true);
7163
+ updateGoal({ status: "active" }, fresh);
7164
+ appendLedger(fresh.cwd, "goal_resumed", { via: "error-brake-retry" });
7165
+ fresh.ui.notify("Auto-resumed after the 5-error brake (cooldown elapsed).", "info");
7166
+ scheduleContinuation(fresh, true);
6936
7167
  }
6937
7168
  }, "5 consecutive errors — auto-retry");
6938
7169
  return;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.34.19",
3
+ "version": "0.34.20",
4
4
  "description": "Mission control for autonomous pi: interview-drafted goals, an audited task queue, and forever-loops (metric, spec, project-audit) that run for hours. An isolated extension-less auditor re-verifies every completion with raw evidence; confirmed drafts, decision pauses and consent gates keep you in charge.",
5
5
  "license": "MIT",
6
6
  "author": "dracon",