pi-goal-list-loop-audit 0.38.67 → 0.38.68

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,43 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.38.68 — Relentless auto-continue: heat-routed exhaustion, quota retries, truthful recovering (2026-09-19)
4
+
5
+ ### Length-exhaustion wedge fixed at the root: heat, not budget (field 162348)
6
+
7
+ Repeated output-token truncation at near-full context is a heat problem —
8
+ the prompt no longer fits the model — so growing the truncation budget
9
+ only ping-pongs into the same wall. `decideLengthExhaustion` now routes on
10
+ `contextPercent`: roomy context restarts the truncation budget relentlessly
11
+ (episode 1 of a bounded 2-episode budget, episode 2 parks durably),
12
+ near-full context with a larger-context fallback rotates models, and
13
+ near-full context without one yields to pi auto-compaction and lets the
14
+ settle path own the resume. The compact-defer path no longer kicks a fresh
15
+ turn into the known-hot window, and a manual `/goal resume` between
16
+ episodes restarts the cycle instead of parking one round early.
17
+
18
+ ### Quota-identical auditor failures hammer bounded retries (field 150821)
19
+
20
+ Rate-limit and plan-quota walls are transient, so they are exempt from the
21
+ identical-failure park: the auditor keeps hammering its bounded retry plan
22
+ instead of parking with "check the auditor/model setup". Billing walls
23
+ and non-quota infra failures still park truthfully. A resumed or recovered
24
+ cycle reseeds the dead candidate chain via `freshAuditorCycleClaim`
25
+ instead of re-walking it, and no longer renders the previous cycle's dead
26
+ chain or diagnostic one round late.
27
+
28
+ ### Armed retry renders recovering, not dead-paused (field 151113)
29
+
30
+ A supervised auditor wait with an armed retry now renders `recovering`
31
+ with an auto-retrying status line. Bare user waits with no pending claim
32
+ still render `paused` — no false recovering labels.
33
+
34
+ ### glla-delegate skill states its drafting precondition (field 125302)
35
+
36
+ The skill doc now says `propose_goal_draft` only runs inside an already-open
37
+ drafting session (bare `/goal`, `/list`, or `/list add`), refuses from
38
+ normal chat, and falls back to `list_add` — the refusal is correct
39
+ behavior, so this is a doc fix with a regression pin, no prod change.
40
+
3
41
  ## 0.38.67 — Stop clobbering other extensions' compactions (2026-09-19)
4
42
 
5
43
  ### session_before_compact returns nothing (issue #56, ezoushen)
package/docs/INDEX.md CHANGED
@@ -16,7 +16,7 @@ Policy contracts and recent changes live in the `audit/` directory of the
16
16
  failback; v0.35.9 hardened cross-version npm tarball checks; v0.35.10
17
17
  handles multi-entry npm dry-run reports; v0.35.11 accepts both npm report
18
18
  shapes; v0.35.12 supports npm 12's keyed pack reports; v0.35.13 fixes stale-API recovery loops.
19
- v0.35.14–v0.38.67 continue through the supervisor freeze (`/glla pause`),
19
+ v0.35.14–v0.38.68 continue through the supervisor freeze (`/glla pause`),
20
20
  load hold, auditor picker parity, Windows launch fix, zombie-watchdog
21
21
  subagent carve-out, due-wait backstop, the `/glla agents` visibility panel,
22
22
  durable state-root selection, blank-until-resume auditor context, frozen
@@ -113,7 +113,7 @@ export interface CommandDeps {
113
113
  releaseContinuationDispatchStandDown: () => void;
114
114
  releaseInitialSessionLoadBarrier: () => void;
115
115
  resolveCarryover: (ctx: ExtensionContext, trigger: "goal" | "loop" | "list") => boolean;
116
- safeSteerUser: (ctx: ExtensionContext, text: string) => boolean;
116
+ resetLengthExhaustionEpisodes: () => void; safeSteerUser: (ctx: ExtensionContext, text: string) => boolean;
117
117
  scheduleContinuation: (ctx: ExtensionContext, force?: boolean, delayMs?: number) => void;
118
118
  scheduleSessionTimeout: (callback: () => void, delayMs: number) => NodeJS.Timeout;
119
119
  createGoal: (objective: string, ctx: ExtensionContext, policy?: "goal" | "list") => Goal;
@@ -138,7 +138,7 @@ let listQueue: CommandDeps["listQueue"], notifyExternal: CommandDeps["notifyExte
138
138
  archiveCurrentGoal: CommandDeps["archiveCurrentGoal"], healGoalPolicy: CommandDeps["healGoalPolicy"], startDrafting: CommandDeps["startDrafting"], warnIfStaleAtEntry: CommandDeps["warnIfStaleAtEntry"], queuePendingListOperation: CommandDeps["queuePendingListOperation"], freshCtx: CommandDeps["freshCtx"],
139
139
  freshCtxForGeneration: CommandDeps["freshCtxForGeneration"], goStaleTerminal: CommandDeps["goStaleTerminal"], groupOpenChildren: CommandDeps["groupOpenChildren"], activateNextListItem: CommandDeps["activateNextListItem"], clearMainModelRecoveryTimer: CommandDeps["clearMainModelRecoveryTimer"], mainModelRecoveryTimerActive: CommandDeps["mainModelRecoveryTimerActive"], continuationDispatchPending: CommandDeps["continuationDispatchPending"], resetContinuationDispatchState: CommandDeps["resetContinuationDispatchState"],
140
140
  isCompletionAuditRecoveryPending: CommandDeps["isCompletionAuditRecoveryPending"], markCompletionAuditRecoveryPending: CommandDeps["markCompletionAuditRecoveryPending"], retryStoredCompletionAudit: CommandDeps["retryStoredCompletionAudit"], probeMainModelRecovery: CommandDeps["probeMainModelRecovery"], releaseContinuationDispatchStandDown: CommandDeps["releaseContinuationDispatchStandDown"],
141
- releaseInitialSessionLoadBarrier: CommandDeps["releaseInitialSessionLoadBarrier"], resolveCarryover: CommandDeps["resolveCarryover"], safeSteerUser: CommandDeps["safeSteerUser"], scheduleContinuation: CommandDeps["scheduleContinuation"], scheduleSessionTimeout: CommandDeps["scheduleSessionTimeout"],
141
+ releaseInitialSessionLoadBarrier: CommandDeps["releaseInitialSessionLoadBarrier"], resolveCarryover: CommandDeps["resolveCarryover"], resetLengthExhaustionEpisodes: CommandDeps["resetLengthExhaustionEpisodes"], safeSteerUser: CommandDeps["safeSteerUser"], scheduleContinuation: CommandDeps["scheduleContinuation"], scheduleSessionTimeout: CommandDeps["scheduleSessionTimeout"],
142
142
  createGoal: CommandDeps["createGoal"], fireReviewer: CommandDeps["fireReviewer"], openSettingsUI: CommandDeps["openSettingsUI"], manuallyResumeMainModelRecovery: CommandDeps["manuallyResumeMainModelRecovery"], activeGoalCommand: CommandDeps["activeGoalCommand"],
143
143
  activeGoalStatusCommand: CommandDeps["activeGoalStatusCommand"], activeGoalSurfaceCommand: CommandDeps["activeGoalSurfaceCommand"], goalNoun: CommandDeps["goalNoun"], displaySlice: CommandDeps["displaySlice"], shortObj: CommandDeps["shortObj"];
144
144
 
@@ -148,7 +148,7 @@ export function createGoalCommands(d: CommandDeps): void {
148
148
  archiveCurrentGoal = d.archiveCurrentGoal; healGoalPolicy = d.healGoalPolicy; startDrafting = d.startDrafting; warnIfStaleAtEntry = d.warnIfStaleAtEntry; queuePendingListOperation = d.queuePendingListOperation; freshCtx = d.freshCtx;
149
149
  freshCtxForGeneration = d.freshCtxForGeneration; goStaleTerminal = d.goStaleTerminal; groupOpenChildren = d.groupOpenChildren; activateNextListItem = d.activateNextListItem; clearMainModelRecoveryTimer = d.clearMainModelRecoveryTimer; mainModelRecoveryTimerActive = d.mainModelRecoveryTimerActive; continuationDispatchPending = d.continuationDispatchPending; resetContinuationDispatchState = d.resetContinuationDispatchState;
150
150
  isCompletionAuditRecoveryPending = d.isCompletionAuditRecoveryPending; markCompletionAuditRecoveryPending = d.markCompletionAuditRecoveryPending; retryStoredCompletionAudit = d.retryStoredCompletionAudit; probeMainModelRecovery = d.probeMainModelRecovery; releaseContinuationDispatchStandDown = d.releaseContinuationDispatchStandDown;
151
- releaseInitialSessionLoadBarrier = d.releaseInitialSessionLoadBarrier; resolveCarryover = d.resolveCarryover; safeSteerUser = d.safeSteerUser; scheduleContinuation = d.scheduleContinuation; scheduleSessionTimeout = d.scheduleSessionTimeout;
151
+ releaseInitialSessionLoadBarrier = d.releaseInitialSessionLoadBarrier; resolveCarryover = d.resolveCarryover; resetLengthExhaustionEpisodes = d.resetLengthExhaustionEpisodes; safeSteerUser = d.safeSteerUser; scheduleContinuation = d.scheduleContinuation; scheduleSessionTimeout = d.scheduleSessionTimeout;
152
152
  createGoal = d.createGoal; fireReviewer = d.fireReviewer; openSettingsUI = d.openSettingsUI; manuallyResumeMainModelRecovery = d.manuallyResumeMainModelRecovery; activeGoalCommand = d.activeGoalCommand;
153
153
  activeGoalStatusCommand = d.activeGoalStatusCommand; activeGoalSurfaceCommand = d.activeGoalSurfaceCommand; goalNoun = d.goalNoun; displaySlice = d.displaySlice; shortObj = d.shortObj;
154
154
  agentsSnapshot = d.agentsSnapshot;
@@ -589,6 +589,10 @@ async function cmdResume(ctx: ExtensionContext): Promise<void> {
589
589
  updateGoal({ status: "active", pauseReason: undefined, pauseSuggestedAction: undefined, pauseKind: undefined, pauseOptions: undefined, pauseRecommended: undefined, pauseResumeAt: undefined, interruptedAt: undefined, interruptedReason: undefined, autoResumedAt: undefined, autoResumedEvent: undefined, ...(staleEntry ? { interruptedAt: nowIso(), interruptedReason: "resumed in a stale session" } : {}), ...(usage ? { usage } : {}) }, ctx);
590
590
  if (staleEntry) return;
591
591
  releaseAuditorSurface();
592
+ // A manual resume starts a fresh relentless cycle: a user pause between
593
+ // an episode-1 exhaustion wedge and the next one must not make that
594
+ // wedge park one cycle early.
595
+ resetLengthExhaustionEpisodes();
592
596
  // A stored completion claim is a direct-audit resume, not an agent turn.
593
597
  // Keeping the claim while merely scheduling a continuation left manual
594
598
  // pause/resume with an ACTIVE goal that no timer would ever consume.
@@ -11,7 +11,7 @@ import * as fs from "node:fs";
11
11
  import * as path from "node:path";
12
12
  import { createHash } from "node:crypto";
13
13
  import { execSync } from "node:child_process";
14
- import { normalizeProviderErrorText, providerErrorFingerprint, providerErrorPresentation, sanitizeProviderAuditReport, sanitizeProviderDisplayText, type QuotaSignal } from "./quota-retry.js";
14
+ import { normalizeProviderErrorText, providerErrorFingerprint, providerErrorPresentation, quotaSignal, sanitizeProviderAuditReport, sanitizeProviderDisplayText, type QuotaSignal } from "./quota-retry.js";
15
15
  import { MAX_AUDITOR_CANDIDATE_REFS, MAX_MAIN_MODEL_FALLBACKS, normalizeBoundedModelRefs } from "./main-model-recovery.js";
16
16
  import { resolveGllaStateDir, stateRootPending } from "./glla-state-root.js";
17
17
  export { normalizeProviderErrorText, providerErrorFingerprint, providerErrorPresentation, sanitizeProviderAuditReport, sanitizeProviderDisplayText } from "./quota-retry.js";
@@ -4076,6 +4076,51 @@ export function trackAuditorIdenticalFailure(
4076
4076
  };
4077
4077
  }
4078
4078
 
4079
+ /** v0.38.68 (relentless goal, field 150821): quota walls are transient —
4080
+ the fix is hammering bounded retries, not parking with "check the
4081
+ auditor/model setup". Rate-limit and plan-quota fingerprints are exempt
4082
+ from the identical park; the retry-plan horizon keeps governing them.
4083
+ Billing is NOT exempt (retries cannot fix a paywall) and unknown/non-quota
4084
+ failures keep the park. Pure. */
4085
+ export function isQuotaIdenticalParkExempt(diagnostic: string | undefined): boolean {
4086
+ const signal = quotaSignal(diagnostic);
4087
+ return signal === "rate-limit" || signal === "plan-quota";
4088
+ }
4089
+
4090
+ /** v0.38.68 (relentless goal, field 150821): reseed a dead auditor chain
4091
+ instead of re-walking it. The 3 -> 4 count bump came from a recovery path
4092
+ re-firing an identical-parked claim with the burned cursor intact. Clearing
4093
+ the candidate chain, the failure streak, and the retry window restarts a
4094
+ fresh bounded window with re-resolved models. Pure — callers spread the
4095
+ result over the stored claim. */
4096
+ const FRESH_AUDITOR_CYCLE_CLEARED_KEYS = [
4097
+ "auditorCandidateRefs",
4098
+ "auditorCandidateRef",
4099
+ "auditorRetryCandidateRef",
4100
+ "auditorRetryAttemptStartedAt",
4101
+ "auditorAttemptedRefs",
4102
+ "auditorEvictedRefs",
4103
+ "auditorLastFailureFingerprint",
4104
+ "auditorConsecutiveIdenticalFailures",
4105
+ "auditorFailureCount",
4106
+ "auditorFailureClass",
4107
+ "auditorFallbackExhausted",
4108
+ "auditorFailureAt",
4109
+ "retryAttempts",
4110
+ "retryFirstAt",
4111
+ "retryUntil",
4112
+ // v0.38.68 reviewer P2: a fresh cycle must not render the previous
4113
+ // cycle's dead chain or wall text one round late.
4114
+ "exhaustedChain",
4115
+ "providerErrorDiagnostic",
4116
+ ] as const;
4117
+
4118
+ export function freshAuditorCycleClaim<T extends object>(claim: T): T {
4119
+ const reseeded: Record<string, unknown> = {};
4120
+ Object.assign(reseeded, claim);
4121
+ for (const key of FRESH_AUDITOR_CYCLE_CLEARED_KEYS) delete reseeded[key];
4122
+ return reseeded as T;
4123
+ }
4079
4124
  /** v0.38.63: append a dead ref to the eviction list when the error proves it
4080
4125
  * unresolvable; otherwise return the list untouched (absent stays absent).
4081
4126
  * Bounded to MAX_AUDITOR_CANDIDATE_REFS like every other cursor ref list. */
@@ -1826,7 +1826,21 @@ function goalLines(g: Goal, state: State, audit: AuditDisplayProgress | null | u
1826
1826
  // 090343 "working while displaying paused here"; the rearm storm streak
1827
1827
  // 19 was firing while the head chip said ⏸ paused). The status line already
1828
1828
  // renders ⏳ auto-retrying for these; the widget head must not contradict it.
1829
- const recovering = g.status === "paused" && !!state.mainModelRecovery && (!!state.mainModelRecovery.retryAt || !!state.mainModelRecovery.pendingModelSwitch);
1829
+ // v0.38.68 (relentless, field 151113): a supervised auditor wait with an
1830
+ // ARMED retry (retry-waiting claim + a finite retry time) is recovering
1831
+ // too — the card read "paused" while the glla recovery timer was
1832
+ // actively auto-retrying. A bare user wait with no retry evidence stays
1833
+ // paused (absent stays absent).
1834
+ const pendingRetryAt = g.pendingCompletion?.phase === "retry-waiting" && typeof g.pendingCompletion.recoveryRetryAt === "string"
1835
+ ? Date.parse(g.pendingCompletion.recoveryRetryAt)
1836
+ : Number.NaN;
1837
+ const waitResumeAt = typeof g.pauseResumeAt === "string" ? Date.parse(g.pauseResumeAt) : Number.NaN;
1838
+ const auditorRecovering = g.status === "paused"
1839
+ && pauseKind(g) === "wait"
1840
+ && !!g.pendingCompletion
1841
+ && (Number.isFinite(pendingRetryAt) || Number.isFinite(waitResumeAt));
1842
+ const recovering = (g.status === "paused" && !!state.mainModelRecovery && (!!state.mainModelRecovery.retryAt || !!state.mainModelRecovery.pendingModelSwitch))
1843
+ || auditorRecovering;
1830
1844
  const icon =
1831
1845
  interrupted
1832
1846
  ? paint(theme, "error", "⚠")
@@ -123,6 +123,43 @@ export function isContextStarvedLengthStop(
123
123
  return false;
124
124
  }
125
125
 
126
+ export type LengthExhaustionDecision = "rotate-fallback" | "compact-defer" | "fresh-budget";
127
+
128
+ /** v0.38.68 (relentless goal, field 162348): answer "compact hiccup or
129
+ over-context" by routing on context heat. Hot context (at or above the
130
+ starvation boundary) means the prompt no longer fits the model — a blind
131
+ fresh truncation budget would burn quota re-emitting a giant artifact into
132
+ a full window, so rotate to a larger-context fallback when one exists and
133
+ otherwise defer to pi auto-compaction. Roomy or unknown context means the
134
+ model simply will not chunk: grant one fresh budget (fresh eyes, split
135
+ instructions) and only park for manual action when that budget exhausts
136
+ too. Pure — the agent_end site owns episodes, notify, and rotation. */
137
+ export function decideLengthExhaustion(args: {
138
+ contextPercent?: number | null;
139
+ fallbackRefsAvailable: boolean;
140
+ }): LengthExhaustionDecision {
141
+ const percent = typeof args.contextPercent === "number" ? args.contextPercent : null;
142
+ const hot = percent !== null && Number.isFinite(percent) && percent >= LENGTH_CONTINUE_CONTEXT_STARVED_PERCENT;
143
+ if (hot) return args.fallbackRefsAvailable ? "rotate-fallback" : "compact-defer";
144
+ return "fresh-budget";
145
+ }
146
+
147
+ export const LENGTH_EXHAUSTION_MAX_EPISODES = 2;
148
+
149
+ /** v0.38.68 (relentless goal): bounded exhaustion episodes. The first
150
+ truncation-budget exhaustion stays relentless (rotate / compact-defer /
151
+ fresh-budget per decideLengthExhaustion); when THAT budget exhausts too,
152
+ park for manual action instead of burning quota forever. Pure — the
153
+ agent_end site holds the counter and resets it on clean turns,
154
+ session_start, and manual parks. */
155
+ export function nextLengthExhaustionEpisode(priorEpisodes: number): { episodes: number; parkNow: boolean } {
156
+ const base = typeof priorEpisodes === "number" && Number.isFinite(priorEpisodes) && priorEpisodes > 0
157
+ ? Math.trunc(priorEpisodes)
158
+ : 0;
159
+ const episodes = base + 1;
160
+ return { episodes, parkNow: episodes >= LENGTH_EXHAUSTION_MAX_EPISODES };
161
+ }
162
+
126
163
  export function makeLengthContinueTracker(max: number = LENGTH_CONTINUE_MAX) {
127
164
  let consecutive = 0;
128
165
  let gaveUp = false;
@@ -182,9 +182,13 @@ export { __testOnlySetContinuationStartTimeout, __testOnlySetContinuationRetryBa
182
182
  import {
183
183
  LENGTH_CONTINUE_MAX,
184
184
  LENGTH_CONTINUE_TEXT,
185
+ LENGTH_EXHAUSTION_MAX_EPISODES,
186
+ decideLengthExhaustion,
185
187
  isContextStarvedLengthStop,
188
+ nextLengthExhaustionEpisode,
186
189
  resetLengthContinue,
187
190
  tickLengthContinue,
191
+ type LengthExhaustionDecision,
188
192
  } from "../length-continue.js";
189
193
  import { isSubagentProviderFailure } from "../quota-retry.js";
190
194
  import { captureProviderTokenUsage } from "../context-growth.js";
@@ -480,6 +484,24 @@ let inBandProviderFailureRaw: string | null = null;
480
484
  function clearInBandProviderFailure(): void {
481
485
  inBandProviderFailureRaw = null;
482
486
  }
487
+ // v0.38.68 (relentless): bounded length-exhaustion episodes. Episode 1
488
+ // stays relentless per context heat (rotate / compact-defer / fresh
489
+ // budget); episode 2 parks for manual action. Reset on clean turns,
490
+ // session_start, and manual parks — never on starved stops (the wedge
491
+ // persists) so a hot context cannot lap the budget forever.
492
+ let lengthExhaustionEpisodes = 0;
493
+ /** A manual resume starts a fresh relentless cycle: a user pause between
494
+ an episode-1 wedge and the next one must not make that wedge park one
495
+ cycle early. */
496
+ export function resetLengthExhaustionEpisodes(): void {
497
+ lengthExhaustionEpisodes = 0;
498
+ }
499
+ /** Test-only: reset the exhaustion-episode counter without firing turns.
500
+ Module state would otherwise leak across behavioral tests sharing one
501
+ process (an episode-1 leftover makes the next test's first wedge park). */
502
+ export function __testOnlyResetLengthExhaustionEpisodes(): void {
503
+ resetLengthExhaustionEpisodes();
504
+ }
483
505
 
484
506
  /** Arm the next automatic re-dispatch after a successful zombie abort.
485
507
  * The configured retry budget keeps repeated recovery finite; the caller gets
@@ -1333,6 +1355,7 @@ export function registerGoalRuntime(pi: ExtensionAPI): void {
1333
1355
  // Session-scoped resources are reset only after this context has passed
1334
1356
  // the host-admission gate; child factories never touch host state.
1335
1357
  resetLengthContinue();
1358
+ lengthExhaustionEpisodes = 0;
1336
1359
  sessionHandoffPending = false;
1337
1360
  // Reset terminal ownership before rememberCtx: this is the only event
1338
1361
  // allowed to bind a context after a stale/shutdown handoff.
@@ -2016,6 +2039,45 @@ export function registerGoalRuntime(pi: ExtensionAPI): void {
2016
2039
  replayApprovalSummariesOnContact(ctx);
2017
2040
  });
2018
2041
 
2042
+ /** v0.38.68 (relentless, field 162348): shared hot-exhaustion core for the
2043
+ goal and loop give-up branches. Rotation and compaction-defer are identical
2044
+ for both; only the fresh-budget copy, the park action, and the kick differ
2045
+ and stay at the call sites. Returns true when the episode was handled
2046
+ (rotated or deferred) so the caller only owns the fresh-budget path. */
2047
+ async function handleHotLengthExhaustion(
2048
+ ctx: ExtensionContext,
2049
+ decision: LengthExhaustionDecision,
2050
+ percent: number | null,
2051
+ consecutive: number,
2052
+ recentCompact: boolean,
2053
+ ): Promise<"kick" | "settle" | "fresh"> {
2054
+ if (decision === "rotate-fallback") {
2055
+ const switched = await recoverFromContextOverflow(ctx, `output-token limit — ${LENGTH_CONTINUE_MAX}× truncated at ${percent !== null ? `${percent.toFixed(1)}%` : "near-full"} context; rotating to a larger-context model`);
2056
+ if (switched) {
2057
+ appendLedger(ctx.cwd, "length_exhausted_rotated", { consecutive, contextPercent: percent });
2058
+ ctx.ui.notify(`glla: response truncated ${LENGTH_CONTINUE_MAX}× at ${percent !== null ? `${percent.toFixed(1)}%` : "near-full"} context — rotated to a larger-context backup model with a fresh truncation budget.`, "info");
2059
+ return "kick";
2060
+ }
2061
+ // Rotation refused (chain burned between decision and attempt) — fall
2062
+ // through to compaction-defer below rather than parking.
2063
+ }
2064
+ if (decision !== "fresh-budget") {
2065
+ noteContextPercent(percent);
2066
+ const yielded = noteContextStarvedYield();
2067
+ appendLedger(ctx.cwd, "length_exhausted_compact_pending", { consecutive, contextPercent: percent, starvedStreak: yielded.streak, recentCompact });
2068
+ ctx.ui.notify(`glla: response truncated ${LENGTH_CONTINUE_MAX}× at ${percent !== null ? `${percent.toFixed(1)}%` : "near-full"} context — the prompt no longer fits this model. Yielding to pi auto-compaction; work stays active and resumes with a fresh truncation budget after compaction lands.${recentCompact ? " A compact-and-retry already failed within the last 90s, so the fallback rotation above is the next recourse." : ""}`, "info");
2069
+ void runEmergencyCompactorIfDue(ctx, yielded.shouldRefuse, {
2070
+ notify: (message) => ctx.ui.notify(message, "info"),
2071
+ page: (message) => notifyExternal(ctx, message),
2072
+ });
2073
+ // v0.38.68 reviewer P1: settle owns the resume — kicking a turn here
2074
+ // dispatches into a known-hot window before the starvation choke
2075
+ // engages (refuse bites at streak>=2; first exhaustion yields streak=1).
2076
+ return "settle";
2077
+ }
2078
+ return "fresh";
2079
+ }
2080
+
2019
2081
  pi.on("agent_end", async (event: any, ctx: ExtensionContext) => {
2020
2082
  rememberCtx(ctx);
2021
2083
  signalSupervisionEvent({ plane: state.mainModelRecovery ? "provider-recovery" : state.loop?.active ? "loop" : state.goal?.policy === "list" ? "list" : state.goal ? "goal" : "queue", kind: "progress", source: "agent_end" });
@@ -2098,7 +2160,12 @@ export function registerGoalRuntime(pi: ExtensionAPI): void {
2098
2160
  });
2099
2161
  }
2100
2162
  const contextStarvedLength = isContextStarvedLengthStop(rawLastA, contextUsage);
2101
- const lc = tickLengthContinue(lastA?.stopReason === "length" && !contextStarvedLength);
2163
+ const lengthStopped = lastA?.stopReason === "length" && !contextStarvedLength;
2164
+ // v0.38.68: a genuinely clean turn closes the exhaustion episode. A
2165
+ // starved stop does NOT (the wedge persists); error stops route to the
2166
+ // recovery paths which own the next turn.
2167
+ if (lastA?.stopReason !== "length" && lastA?.stopReason !== "error" && !contextStarvedLength) lengthExhaustionEpisodes = 0;
2168
+ const lc = tickLengthContinue(lengthStopped);
2102
2169
  if (contextStarvedLength) {
2103
2170
  const starved = noteContextStarvedYield();
2104
2171
  appendLedger(ctx.cwd, "length_continue_deferred_context_full", {
@@ -2141,26 +2208,64 @@ export function registerGoalRuntime(pi: ExtensionAPI): void {
2141
2208
  appendLedger(ctx.cwd, "length_continue_exhausted", { consecutive: lc.consecutive });
2142
2209
  ctx.ui.notify(`glla: response hit the output-token cap ${LENGTH_CONTINUE_MAX}× in a row — stepping aside. Ask the model to split the work into smaller pieces.`, "warning");
2143
2210
  if (state.goal && state.goal.status === "active") {
2144
- notifyExternal(ctx, `Response truncated ${LENGTH_CONTINUE_MAX}× in a row — ${goalNoun()} paused; split the work, then ${activeGoalSurfaceCommand("resume")}.`);
2145
- updateGoal({
2146
- status: "paused",
2147
- pauseKind: "error",
2148
- pauseReason: `output-token limit — ${LENGTH_CONTINUE_MAX} responses in a row were truncated mid-artifact; auto-continue exhausted`,
2149
- pauseSuggestedAction: `Re-scope the current artifact into smaller pieces (several smaller write/edit calls across turns instead of one giant response), then ${activeGoalSurfaceCommand("resume")} — the truncation budget restarts fresh.`,
2150
- }, ctx);
2151
- // The pause is the durable record; an explicit recovery gets a full
2152
- // fresh truncation budget (otherwise the sticky gaveUp flag would
2153
- // make the resumed turn silently dead on the first truncation).
2154
- resetLengthContinue();
2211
+ // v0.38.68 (relentless, field 162348): bounded exhaustion episodes
2212
+ // instead of an immediate manual park. Episode 1 stays relentless
2213
+ // per context heat; episode 2 parks — the model proved it will not
2214
+ // chunk and further budgets would only burn quota.
2215
+ const episode = nextLengthExhaustionEpisode(lengthExhaustionEpisodes);
2216
+ lengthExhaustionEpisodes = episode.episodes;
2217
+ if (episode.parkNow) {
2218
+ notifyExternal(ctx, `Response truncated ${LENGTH_CONTINUE_MAX}× in a row — ${goalNoun()} paused; split the work, then ${activeGoalSurfaceCommand("resume")}.`);
2219
+ updateGoal({
2220
+ status: "paused",
2221
+ pauseKind: "error",
2222
+ pauseReason: `output-token limit — ${LENGTH_CONTINUE_MAX} responses in a row were truncated mid-artifact; auto-continue exhausted`,
2223
+ pauseSuggestedAction: `Re-scope the current artifact into smaller pieces (several smaller write/edit calls across turns instead of one giant response), then ${activeGoalSurfaceCommand("resume")} — the truncation budget restarts fresh.`,
2224
+ }, ctx);
2225
+ // The pause is the durable record; an explicit recovery gets a full
2226
+ // fresh truncation budget (otherwise the sticky gaveUp flag would
2227
+ // make the resumed turn silently dead on the first truncation).
2228
+ // Episodes reset so the manual resume starts a fresh relentless cycle.
2229
+ lengthExhaustionEpisodes = 0;
2230
+ resetLengthContinue();
2231
+ } else {
2232
+ const percent = typeof contextUsage?.percent === "number" && Number.isFinite(contextUsage.percent) ? contextUsage.percent : null;
2233
+ const decision = decideLengthExhaustion({ contextPercent: percent, fallbackRefsAvailable: mainModelFallbackRefs(ctx).length > 0 });
2234
+ const sinceLastCompactMs = state.lastCompactionAt ? Date.now() - state.lastCompactionAt : Number.POSITIVE_INFINITY;
2235
+ const handled = await handleHotLengthExhaustion(ctx, decision, percent, lc.consecutive, sinceLastCompactMs < COMPACTION_GRACE_MS);
2236
+ if (handled === "fresh") {
2237
+ appendLedger(ctx.cwd, "length_exhausted_fresh_budget", { consecutive: lc.consecutive, contextPercent: percent, episode: episode.episodes });
2238
+ ctx.ui.notify(`glla: response hit the output-token cap ${LENGTH_CONTINUE_MAX}× in a row — truncation budget restarted (relentless episode ${episode.episodes} of ${LENGTH_EXHAUSTION_MAX_EPISODES - 1}). Split the work into smaller pieces across turns; a repeat parks for manual action.`, "warning");
2239
+ }
2240
+ resetLengthContinue();
2241
+ if (handled !== "settle") scheduleContinuation(ctx);
2242
+ }
2155
2243
  } else if (state.loop?.active) {
2156
- state.loop.active = false;
2157
- state.loop.stopReason = `output-token limit — ${LENGTH_CONTINUE_MAX} consecutive truncated responses (iteration ${state.loop.iteration} preserved; /loop resume after re-scoping the work into smaller pieces)`;
2158
- persistState(ctx);
2159
- const recap = compactLoopCompletionSummary({ ...state.loop, historyLength: state.loop.history.length });
2160
- ctx.ui.notify(`glla: response hit the output-token cap ${LENGTH_CONTINUE_MAX}× in a row — the loop stopped. Ask the model to split the work into smaller pieces.\nRecap: ${recap}`, "warning");
2161
- notifyExternal(ctx, `Response truncated ${LENGTH_CONTINUE_MAX}× in a row — loop stopped; /loop resume after re-scoping. Recap: ${recap}`);
2162
- appendLedger(ctx.cwd, "loop_stopped", { reason: state.loop.stopReason, iterations: state.loop.iteration, best: state.loop.bestValue, recap });
2163
- resetLengthContinue();
2244
+ // v0.38.68 (relentless): same bounded episodes as the goal branch.
2245
+ const episode = nextLengthExhaustionEpisode(lengthExhaustionEpisodes);
2246
+ lengthExhaustionEpisodes = episode.episodes;
2247
+ if (episode.parkNow) {
2248
+ state.loop.active = false;
2249
+ state.loop.stopReason = `output-token limit — ${LENGTH_CONTINUE_MAX} consecutive truncated responses (iteration ${state.loop.iteration} preserved; /loop resume after re-scoping the work into smaller pieces)`;
2250
+ persistState(ctx);
2251
+ const recap = compactLoopCompletionSummary({ ...state.loop, historyLength: state.loop.history.length });
2252
+ ctx.ui.notify(`glla: response hit the output-token cap ${LENGTH_CONTINUE_MAX}× in a row — the loop stopped. Ask the model to split the work into smaller pieces.\nRecap: ${recap}`, "warning");
2253
+ notifyExternal(ctx, `Response truncated ${LENGTH_CONTINUE_MAX}× in a row — loop stopped; /loop resume after re-scoping. Recap: ${recap}`);
2254
+ appendLedger(ctx.cwd, "loop_stopped", { reason: state.loop.stopReason, iterations: state.loop.iteration, best: state.loop.bestValue, recap });
2255
+ lengthExhaustionEpisodes = 0;
2256
+ resetLengthContinue();
2257
+ } else {
2258
+ const percent = typeof contextUsage?.percent === "number" && Number.isFinite(contextUsage.percent) ? contextUsage.percent : null;
2259
+ const decision = decideLengthExhaustion({ contextPercent: percent, fallbackRefsAvailable: mainModelFallbackRefs(ctx).length > 0 });
2260
+ const sinceLastCompactMs = state.lastCompactionAt ? Date.now() - state.lastCompactionAt : Number.POSITIVE_INFINITY;
2261
+ const handled = await handleHotLengthExhaustion(ctx, decision, percent, lc.consecutive, sinceLastCompactMs < COMPACTION_GRACE_MS);
2262
+ if (handled === "fresh") {
2263
+ appendLedger(ctx.cwd, "length_exhausted_fresh_budget", { consecutive: lc.consecutive, contextPercent: percent, episode: episode.episodes });
2264
+ ctx.ui.notify(`glla: response hit the output-token cap ${LENGTH_CONTINUE_MAX}× in a row — truncation budget restarted (relentless episode ${episode.episodes} of ${LENGTH_EXHAUSTION_MAX_EPISODES - 1}). Split the work into smaller pieces across turns; a repeat stops the loop.`, "warning");
2265
+ }
2266
+ resetLengthContinue();
2267
+ if (handled !== "settle") scheduleLoopTick(ctx);
2268
+ }
2164
2269
  } else {
2165
2270
  notifyExternal(ctx, "Response truncated 3× in a row — giving up auto-continue.");
2166
2271
  }
@@ -116,6 +116,8 @@ import {
116
116
  modelSwitch,
117
117
  isForbiddenModel,
118
118
  filterEvictedAuditorRefs,
119
+ freshAuditorCycleClaim,
120
+ isQuotaIdenticalParkExempt,
119
121
  withEvictedAuditorRef,
120
122
  trackAuditorIdenticalFailure,
121
123
  isGoalRevisionCurrent,
@@ -653,29 +655,24 @@ function beginCompletionAudit(ctx: ExtensionContext, claim: PendingCompletion, o
653
655
  // An agent-tool resume carries the same in-conversation user
654
656
  // authorization as a typed /goal resume, so an exhausted claim also
655
657
  // starts a fresh cycle instead of re-walking a dead cursor.
656
- const freshAuditorCycle = (origin === "manual" || origin === "agent") && claim.auditorFallbackExhausted === true;
658
+ // v0.38.68 (relentless, field 150821): the single automatic
659
+ // session-recovery retry reseeds too — re-walking a burned cursor is what
660
+ // bumped an identical-parked claim from 3 to 4 failures. Every origin now
661
+ // starts a fresh bounded window with re-resolved models via
662
+ // freshAuditorCycleClaim; the automatic attempt stays bounded by
663
+ // automaticRecoveryAttempted.
664
+ const freshAuditorCycle = (origin === "manual" || origin === "agent" || origin === "session-recovery") && claim.auditorFallbackExhausted === true;
657
665
  const claimForAttempt = (origin === "manual" || origin === "agent")
658
666
  ? {
659
667
  ...claim,
660
668
  retryAttempts: undefined,
661
669
  retryFirstAt: undefined,
662
670
  retryUntil: undefined,
663
- ...(freshAuditorCycle ? {
664
- auditorCandidateRefs: undefined,
665
- auditorCandidateRef: undefined,
666
- auditorRetryCandidateRef: undefined,
667
- auditorRetryAttemptStartedAt: undefined,
668
- auditorAttemptedRefs: undefined,
669
- auditorEvictedRefs: undefined,
670
- auditorLastFailureFingerprint: undefined,
671
- auditorConsecutiveIdenticalFailures: undefined,
672
- auditorFailureCount: undefined,
673
- auditorFailureClass: undefined,
674
- auditorFallbackExhausted: undefined,
675
- auditorFailureAt: undefined,
676
- } : {}),
671
+ ...(freshAuditorCycle ? freshAuditorCycleClaim(claim) : {}),
677
672
  }
678
- : claim;
673
+ : origin === "session-recovery" && freshAuditorCycle
674
+ ? { ...claim, ...freshAuditorCycleClaim(claim) }
675
+ : claim;
679
676
  const pending: PendingCompletion = {
680
677
  ...claimForAttempt,
681
678
  phase: "running",
@@ -1787,7 +1784,11 @@ async function retryStoredCompletionAudit(origin: CompletionAuditOrigin = "provi
1787
1784
  const plan = auditorRetryPlan(durableClaim, undefined, undefined, aggressive);
1788
1785
  // v0.38.63 (audit-stuck batch): same identical-park gate as the
1789
1786
  // complete_goal ladder path — the loop terminates visibly here too.
1787
+ // v0.38.68 (relentless, field 150821): quota walls are transient —
1788
+ // hammer the bounded retry plan instead of parking with "check the
1789
+ // setup". Only identical NON-quota failures park.
1790
1790
  const identical = trackAuditorIdenticalFailure(durableClaim, failureCopy.fingerprint);
1791
+ const quotaIdenticalExempt = isQuotaIdenticalParkExempt(failureCopy.diagnostic);
1791
1792
  const pending = {
1792
1793
  ...durableClaim,
1793
1794
  phase: "retry-waiting" as const,
@@ -1805,7 +1806,7 @@ async function retryStoredCompletionAudit(origin: CompletionAuditOrigin = "provi
1805
1806
  retryFirstAt: plan.firstAt,
1806
1807
  retryUntil: plan.autoRetryUntil,
1807
1808
  };
1808
- if (identical.identicalParkDue) {
1809
+ if (identical.identicalParkDue && !quotaIdenticalExempt) {
1809
1810
  const deadChain = (durableClaim.exhaustedChain
1810
1811
  ?? (durableClaim.auditorCandidateRefs ?? durableClaim.auditorAttemptedRefs ?? []).join(" → ")
1811
1812
  ?? "").slice(0, 300) || "unknown chain";
@@ -124,6 +124,7 @@ import {
124
124
  modelSwitch,
125
125
  isForbiddenModel,
126
126
  filterEvictedAuditorRefs,
127
+ isQuotaIdenticalParkExempt,
127
128
  withEvictedAuditorRef,
128
129
  trackAuditorIdenticalFailure,
129
130
  isGoalRevisionCurrent,
@@ -1791,7 +1792,11 @@ function registerAgentTools(pi: any): void {
1791
1792
  // failures terminate the loop visibly — park blocked naming the
1792
1793
  // dead chain, schedule no further retry. Manual/agent resume
1793
1794
  // still opens a fresh cycle with re-resolved models.
1795
+ // v0.38.68 (relentless, field 150821): quota walls are transient —
1796
+ // hammer the bounded retry plan instead of parking with "check the
1797
+ // setup". Only identical NON-quota failures park.
1794
1798
  const identical = trackAuditorIdenticalFailure(durableCompletionClaim, failureCopy.fingerprint);
1799
+ const quotaIdenticalExempt = isQuotaIdenticalParkExempt(failureCopy.diagnostic);
1795
1800
  const pending = {
1796
1801
  ...durableCompletionClaim,
1797
1802
  phase: "retry-waiting" as const,
@@ -1809,7 +1814,7 @@ function registerAgentTools(pi: any): void {
1809
1814
  retryFirstAt: plan.firstAt,
1810
1815
  retryUntil: plan.autoRetryUntil,
1811
1816
  };
1812
- if (identical.identicalParkDue) {
1817
+ if (identical.identicalParkDue && !quotaIdenticalExempt) {
1813
1818
  const deadChain = (durableCompletionClaim.exhaustedChain
1814
1819
  ?? (durableCompletionClaim.auditorCandidateRefs ?? durableCompletionClaim.auditorAttemptedRefs ?? []).join(" → ")
1815
1820
  ?? "").slice(0, 300) || "unknown chain";
@@ -16,7 +16,7 @@ import "./goal-auditor-hooks.js";
16
16
  import "./goal-list-queue.js";
17
17
  import "./goal-tools.js";
18
18
  import "./goal-settings-ui.js";
19
- import { abortZombieRun, enqueueFaultRepairTask, registerGoalRuntime } from "./goal-activation.js";
19
+ import { abortZombieRun, enqueueFaultRepairTask, registerGoalRuntime, resetLengthExhaustionEpisodes } from "./goal-activation.js";
20
20
 
21
21
  import {
22
22
  createGoalContinuation,
@@ -178,6 +178,7 @@ const commandDeps: CommandDeps = {
178
178
  releaseContinuationDispatchStandDown,
179
179
  releaseInitialSessionLoadBarrier,
180
180
  resolveCarryover,
181
+ resetLengthExhaustionEpisodes,
181
182
  safeSteerUser,
182
183
  scheduleContinuation,
183
184
  scheduleSessionTimeout,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.38.67",
3
+ "version": "0.38.68",
4
4
  "description": "Mission control for autonomous pi: interview-drafted goals, an audited task queue, and forever-loops (metric, spec, project-audit) that run for hours. A detached extension-less auditor process re-verifies every completion with raw evidence without holding the main pi turn; confirmed drafts, decision pauses and consent gates keep you in charge.",
5
5
  "license": "AGPL-3.0-only",
6
6
  "author": "dracon",
@@ -17,9 +17,14 @@ tools in the main session only:
17
17
  current request and offer to queue it first. Queue it without another
18
18
  question only when the user has given standing permission such as
19
19
  “queue follow-ups as you find them.”
20
- - If one durable, multi-hour objective is warranted, interview only what is
21
- genuinely unknown, then call `propose_goal_draft`; the user's Confirm
22
- dialog is the activation gate. Never activate a speculative raw seed.
20
+ - If one durable, multi-hour objective is warranted, `propose_goal_draft`
21
+ works only while a drafting session is already open (the user ran bare
22
+ `/goal`, `/list`, or `/list add` with no args): interview only what is
23
+ genuinely unknown, then propose; the user's Confirm dialog is the
24
+ activation gate. From normal chat with no draft open, do NOT call
25
+ `propose_goal_draft` — it refuses outside drafting mode. Ask the user to
26
+ open drafting with bare `/goal`, or use `list_add` for straight queueing.
27
+ Never activate a speculative raw seed.
23
28
  (Unless Auto-accept drafts is on in /glla settings, which skips the
24
29
  Confirm — never promise a dialog that setting suppresses.)
25
30
  - Never call `list_activate` for an item the user has not selected or clearly