pi-goal-list-loop-audit 0.34.18 → 0.34.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -1313,15 +1313,42 @@ export interface InfraRetryOutcome<T> {
|
|
|
1313
1313
|
* (retried once)". The failed pair is never a verdict on the work. */
|
|
1314
1314
|
export async function runWithInfraRetry<T extends { error?: string; approved: boolean; disapproved: boolean }>(
|
|
1315
1315
|
run: () => Promise<T>,
|
|
1316
|
-
opts: {
|
|
1316
|
+
opts: {
|
|
1317
|
+
backoffMs?: number;
|
|
1318
|
+
sleep?: (ms: number) => Promise<void>;
|
|
1319
|
+
onRetry?: (error: string) => void;
|
|
1320
|
+
/**
|
|
1321
|
+
* v0.34.20: delayed retry callers can fail closed across a session
|
|
1322
|
+
* replacement. The first attempt may finish after its ExtensionContext
|
|
1323
|
+
* was invalidated; never launch the second attempt unless the caller can
|
|
1324
|
+
* prove that its session/generation is still live.
|
|
1325
|
+
*/
|
|
1326
|
+
shouldRetry?: () => boolean;
|
|
1327
|
+
} = {},
|
|
1317
1328
|
): Promise<InfraRetryOutcome<T>> {
|
|
1318
1329
|
const sleep = opts.sleep ?? ((ms: number) => new Promise<void>((r) => setTimeout(r, ms)));
|
|
1319
1330
|
const first = await run();
|
|
1320
1331
|
if (first.approved || first.disapproved || !isRetriableInfraError(first.error)) {
|
|
1321
1332
|
return { result: first, retriedOnce: false };
|
|
1322
1333
|
}
|
|
1334
|
+
if (opts.shouldRetry) {
|
|
1335
|
+
try {
|
|
1336
|
+
if (!opts.shouldRetry()) return { result: first, retriedOnce: false };
|
|
1337
|
+
} catch {
|
|
1338
|
+
// A lifecycle probe that cannot establish liveness is a hard stop, not
|
|
1339
|
+
// permission to retry an old session.
|
|
1340
|
+
return { result: first, retriedOnce: false };
|
|
1341
|
+
}
|
|
1342
|
+
}
|
|
1323
1343
|
opts.onRetry?.(first.error!);
|
|
1324
1344
|
await sleep(opts.backoffMs ?? 5000);
|
|
1345
|
+
if (opts.shouldRetry) {
|
|
1346
|
+
try {
|
|
1347
|
+
if (!opts.shouldRetry()) return { result: first, retriedOnce: false };
|
|
1348
|
+
} catch {
|
|
1349
|
+
return { result: first, retriedOnce: false };
|
|
1350
|
+
}
|
|
1351
|
+
}
|
|
1325
1352
|
const second = await run();
|
|
1326
1353
|
return { result: second, retriedOnce: true };
|
|
1327
1354
|
}
|
|
@@ -310,17 +310,24 @@ function heldLoopLines(l: LoopState, now: number, theme?: DisplayTheme, width?:
|
|
|
310
310
|
function goalLines(g: Goal, state: State, audit: AuditDisplayProgress | null | undefined, now: number, theme?: DisplayTheme, width?: number, extras?: WidgetExtras): string[] {
|
|
311
311
|
// Head glyph is ● (not ◆): U+25C6 renders as a color-emoji diamond in some
|
|
312
312
|
// terminal fonts and ignores ANSI color; ● takes the paint everywhere.
|
|
313
|
+
const interrupted = g.status === "active" && !!g.interruptedAt;
|
|
313
314
|
const icon =
|
|
314
|
-
|
|
315
|
-
? paint(theme,
|
|
316
|
-
: g.status === "
|
|
317
|
-
? paint(theme, "
|
|
318
|
-
:
|
|
315
|
+
interrupted
|
|
316
|
+
? paint(theme, "error", "⚠")
|
|
317
|
+
: g.status === "paused"
|
|
318
|
+
? paint(theme, pauseIsError(g) ? "error" : "warning", "⏸")
|
|
319
|
+
: g.status === "auditing"
|
|
320
|
+
? paint(theme, "accent", "⟡")
|
|
321
|
+
: paint(theme, "success", "●");
|
|
319
322
|
// v0.24.7: a list item is named as such and points at /list — before,
|
|
320
323
|
// the widget called it "active" and hinted "/goal status", reading as if
|
|
321
324
|
// queue work were a standalone goal.
|
|
322
325
|
const isList = g.policy === "list";
|
|
323
|
-
const statusWord =
|
|
326
|
+
const statusWord = interrupted
|
|
327
|
+
? paint(theme, "error", "interrupted")
|
|
328
|
+
: g.status === "active"
|
|
329
|
+
? paint(theme, "success", "active")
|
|
330
|
+
: g.status;
|
|
324
331
|
// v0.33.0: slim card — status folds INTO the head line as middot segments
|
|
325
332
|
// (filter(Boolean).join, the universal CLI idiom). Line 2 is the live
|
|
326
333
|
// "last action · next task" line; the footer stays the hint line.
|
|
@@ -351,6 +358,12 @@ function goalLines(g: Goal, state: State, audit: AuditDisplayProgress | null | u
|
|
|
351
358
|
const objBudget = width && width > 0 ? Math.max(16, width - 1 - 2 - 3 - visibleLen(segsText)) : 48;
|
|
352
359
|
const head = `${icon} ${truncate(g.objective.replace(/\s+/g, " "), objBudget)} ${paint(theme, "dim", "·")} ${segsText}`;
|
|
353
360
|
const lines = [head];
|
|
361
|
+
if (interrupted) {
|
|
362
|
+
const resumeCmd = isList ? "/list resume" : "/goal resume";
|
|
363
|
+
lines.push(`├─ ${paint(theme, "error", "host session lost — waiting for fresh session_start")}`);
|
|
364
|
+
lines.push(`└─ ${paint(theme, "warning", `/reload to rebind · ${resumeCmd} if it does not resume`)}`);
|
|
365
|
+
return lines;
|
|
366
|
+
}
|
|
354
367
|
if (g.status === "auditing") {
|
|
355
368
|
lines.push(`├─ auditor: ${audit?.label ?? "running"}${audit?.currentTool ? ` · ${truncate(audit.currentTool, 30)}` : ""}`);
|
|
356
369
|
// v0.25.4: auditor-quiet stall — progress events stopped arriving
|
|
@@ -35,6 +35,48 @@ export interface LengthContinueTick {
|
|
|
35
35
|
consecutive: number;
|
|
36
36
|
}
|
|
37
37
|
|
|
38
|
+
export interface ContextUsageLike {
|
|
39
|
+
tokens?: number | null;
|
|
40
|
+
contextWindow?: number;
|
|
41
|
+
percent?: number | null;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export interface AssistantLengthMessageLike {
|
|
45
|
+
stopReason?: string;
|
|
46
|
+
usage?: {
|
|
47
|
+
output?: number;
|
|
48
|
+
};
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
export const LENGTH_CONTINUE_CONTEXT_STARVED_PERCENT = 90;
|
|
52
|
+
export const LENGTH_CONTINUE_CONTEXT_STARVED_MAX_OUTPUT = 8;
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* v0.34.19: distinguish a REAL overlong assistant response from pi's
|
|
56
|
+
* context-safety clamp. Near the configured context ceiling,
|
|
57
|
+
* pi-ai's clampMaxTokensToContext() can reduce max_tokens to 1; MiniMax then
|
|
58
|
+
* returns stopReason "length" with ~1 output token. That is context
|
|
59
|
+
* starvation: auto-compaction must own recovery. Sending LENGTH_CONTINUE_TEXT
|
|
60
|
+
* here queues another 1-token request before pi's post-agent_end compaction
|
|
61
|
+
* check and delays the actual cure (field: darklord 2026-08-02, 198,116 /
|
|
62
|
+
* 198,179 total tokens of a 200,000 window, output=1 twice).
|
|
63
|
+
*/
|
|
64
|
+
export function isContextStarvedLengthStop(
|
|
65
|
+
message: AssistantLengthMessageLike | null | undefined,
|
|
66
|
+
contextUsage: ContextUsageLike | null | undefined,
|
|
67
|
+
): boolean {
|
|
68
|
+
if (message?.stopReason !== "length") return false;
|
|
69
|
+
const output = message.usage?.output;
|
|
70
|
+
if (typeof output !== "number" || !Number.isFinite(output)) return false;
|
|
71
|
+
if (output > LENGTH_CONTINUE_CONTEXT_STARVED_MAX_OUTPUT) return false;
|
|
72
|
+
const percent = typeof contextUsage?.percent === "number"
|
|
73
|
+
? contextUsage.percent
|
|
74
|
+
: typeof contextUsage?.tokens === "number" && typeof contextUsage?.contextWindow === "number" && contextUsage.contextWindow > 0
|
|
75
|
+
? (contextUsage.tokens / contextUsage.contextWindow) * 100
|
|
76
|
+
: null;
|
|
77
|
+
return percent !== null && Number.isFinite(percent) && percent >= LENGTH_CONTINUE_CONTEXT_STARVED_PERCENT;
|
|
78
|
+
}
|
|
79
|
+
|
|
38
80
|
export function makeLengthContinueTracker(max: number = LENGTH_CONTINUE_MAX) {
|
|
39
81
|
let consecutive = 0;
|
|
40
82
|
let gaveUp = false;
|
package/extensions/loops/goal.ts
CHANGED
|
@@ -92,6 +92,7 @@ import {
|
|
|
92
92
|
import {
|
|
93
93
|
LENGTH_CONTINUE_MAX,
|
|
94
94
|
LENGTH_CONTINUE_TEXT,
|
|
95
|
+
isContextStarvedLengthStop,
|
|
95
96
|
resetLengthContinue,
|
|
96
97
|
tickLengthContinue,
|
|
97
98
|
} from "../length-continue.js";
|
|
@@ -231,6 +232,11 @@ let extensionApiStale = false;
|
|
|
231
232
|
// path must still ledger the stale handle, stop stale work, and preserve the
|
|
232
233
|
// interrupt marker so a later fresh lifecycle can restore it.
|
|
233
234
|
let staleTerminalDone = false;
|
|
235
|
+
// v0.34.19: delayed session-owned callbacks capture this generation. A
|
|
236
|
+
// clearTimeout can race a callback already queued by Node; without a
|
|
237
|
+
// generation check, an old compaction/refire callback can run after /reload
|
|
238
|
+
// and schedule work against the fresh session.
|
|
239
|
+
let sessionGeneration = 0;
|
|
234
240
|
|
|
235
241
|
/** v0.26.7: a stale api is terminal for this process — go loudly with
|
|
236
242
|
* restart guidance instead of retrying sends that can never land.
|
|
@@ -309,6 +315,10 @@ function goStaleTerminal(ctx: ExtensionContext, where: string): void {
|
|
|
309
315
|
} else if (state.goal && state.goal.status === "active") {
|
|
310
316
|
updateGoal({ interruptedAt: nowIso(), interruptedReason: `extension api stale (${where})` }, ctx);
|
|
311
317
|
}
|
|
318
|
+
// The stale process loses its ticker immediately, so paint the durable
|
|
319
|
+
// interrupted state synchronously while the old UI handle can still accept
|
|
320
|
+
// updates. The next session_start paints it again from disk.
|
|
321
|
+
refreshUI(ctx);
|
|
312
322
|
ctx.ui.notify(`glla: ${guidance}`, "warning");
|
|
313
323
|
notifyExternal(ctx, `glla: extension api stale — waiting for a fresh session_start; restart pi normally only if no replacement arrives. (${where})`);
|
|
314
324
|
}
|
|
@@ -622,6 +632,10 @@ function releaseInitialSessionLoadBarrier(): void {
|
|
|
622
632
|
}
|
|
623
633
|
|
|
624
634
|
function rememberCtx(ctx: ExtensionContext): void {
|
|
635
|
+
// Late events from a disposed session must never reclaim lastCtx after the
|
|
636
|
+
// lifecycle handoff has been declared. Only session_start clears these
|
|
637
|
+
// gates and may bind a fresh context.
|
|
638
|
+
if (sessionHandoffPending || staleTerminalDone || zombieStoodDown) return;
|
|
625
639
|
let ownerLive = false;
|
|
626
640
|
if (ownerSession && lastCtx) {
|
|
627
641
|
try { lastCtx.isIdle(); ownerLive = true; } catch { /* owner went stale (session replaced) */ }
|
|
@@ -692,6 +706,9 @@ let carryoverResolved = true;
|
|
|
692
706
|
// set while complete_goal's isolated audit runs, so the heartbeat never
|
|
693
707
|
// refires into an in-flight completion.
|
|
694
708
|
let completionAuditInFlight = false;
|
|
709
|
+
// v0.34.20: an old auditor's finally block must not clear the in-flight
|
|
710
|
+
// marker belonging to a fresh lifecycle generation.
|
|
711
|
+
let completionAuditGeneration: number | null = null;
|
|
695
712
|
// v0.32.0: consecutive stored-claim quota retries (capped at 5, then hold).
|
|
696
713
|
let quotaRetryStreak = 0;
|
|
697
714
|
let heartbeatTimer: NodeJS.Timeout | null = null;
|
|
@@ -1065,7 +1082,7 @@ function heartbeatTick(): void {
|
|
|
1065
1082
|
appendLedger(ctx.cwd, "stranded_audit_recovered", { goalId: state.goal.id, via: state.goal.pendingCompletion ? "stored-claim" : "resume-active" });
|
|
1066
1083
|
if (state.goal.pendingCompletion) {
|
|
1067
1084
|
ctx.ui.notify("Recovering a completion audit whose result never landed — re-running the auditor with the stored claim.", "info");
|
|
1068
|
-
void retryStoredCompletionAudit(
|
|
1085
|
+
void retryStoredCompletionAudit("quota-retry");
|
|
1069
1086
|
} else {
|
|
1070
1087
|
updateGoal({ status: "active" }, ctx);
|
|
1071
1088
|
ctx.ui.notify("A completion audit was interrupted (its result never landed). Resuming — re-call complete_goal when the deliverable still stands.", "warning");
|
|
@@ -1210,9 +1227,19 @@ function clearContinuationTimer(): void {
|
|
|
1210
1227
|
}
|
|
1211
1228
|
|
|
1212
1229
|
function scheduleSessionTimeout(callback: () => void, delayMs: number): NodeJS.Timeout {
|
|
1230
|
+
const generation = sessionGeneration;
|
|
1213
1231
|
let timer: NodeJS.Timeout;
|
|
1214
1232
|
timer = setTimeout(() => {
|
|
1215
1233
|
sessionTimeouts.delete(timer);
|
|
1234
|
+
// clearTimeout is not enough when the callback is already queued. Do not
|
|
1235
|
+
// let an old session's callback re-arm work after stale/shutdown/reload.
|
|
1236
|
+
if (
|
|
1237
|
+
generation !== sessionGeneration ||
|
|
1238
|
+
sessionHandoffPending ||
|
|
1239
|
+
extensionApiStale ||
|
|
1240
|
+
staleTerminalDone ||
|
|
1241
|
+
zombieStoodDown
|
|
1242
|
+
) return;
|
|
1216
1243
|
callback();
|
|
1217
1244
|
}, delayMs);
|
|
1218
1245
|
sessionTimeouts.add(timer);
|
|
@@ -1222,6 +1249,7 @@ function scheduleSessionTimeout(callback: () => void, delayMs: number): NodeJS.T
|
|
|
1222
1249
|
|
|
1223
1250
|
function clearSessionOwnedTimers(): void {
|
|
1224
1251
|
sessionHandoffPending = true;
|
|
1252
|
+
sessionGeneration++;
|
|
1225
1253
|
initialSessionLoadPending = false;
|
|
1226
1254
|
clearContinuationTimer();
|
|
1227
1255
|
clearLoopTimer();
|
|
@@ -1253,6 +1281,53 @@ function freshCtx(): ExtensionContext | null {
|
|
|
1253
1281
|
}
|
|
1254
1282
|
}
|
|
1255
1283
|
|
|
1284
|
+
/**
|
|
1285
|
+
* v0.34.20: a timer can already be queued when clearSessionOwnedTimers()
|
|
1286
|
+
* runs, and an async audit can finish after a replacement without a queued
|
|
1287
|
+
* timer at all. Delayed work must prove both facts before touching pi:
|
|
1288
|
+
* generation identity is unchanged and the context probe succeeds. A null
|
|
1289
|
+
* result is a normal fail-closed handoff, not a reason to use the caller's
|
|
1290
|
+
* captured context as a fallback.
|
|
1291
|
+
*/
|
|
1292
|
+
function freshCtxForGeneration(generation: number): ExtensionContext | null {
|
|
1293
|
+
if (
|
|
1294
|
+
generation !== sessionGeneration ||
|
|
1295
|
+
sessionHandoffPending ||
|
|
1296
|
+
initialSessionLoadPending ||
|
|
1297
|
+
extensionApiStale ||
|
|
1298
|
+
staleTerminalDone ||
|
|
1299
|
+
zombieStoodDown
|
|
1300
|
+
) return null;
|
|
1301
|
+
return freshCtx();
|
|
1302
|
+
}
|
|
1303
|
+
|
|
1304
|
+
/**
|
|
1305
|
+
* v0.34.20: the generic quota helper owns only the wall-clock timer and the
|
|
1306
|
+
* immediate notification. This adapter owns the session boundary: callbacks
|
|
1307
|
+
* receive a context proven fresh at fire time and may not close over the
|
|
1308
|
+
* scheduling event's ctx.
|
|
1309
|
+
*/
|
|
1310
|
+
function scheduleQuotaRetryForSession(
|
|
1311
|
+
ctx: ExtensionContext,
|
|
1312
|
+
retryAfterSec: number,
|
|
1313
|
+
reason: string,
|
|
1314
|
+
fire: (ctx: ExtensionContext) => void | Promise<void>,
|
|
1315
|
+
label?: string,
|
|
1316
|
+
): void {
|
|
1317
|
+
const generation = sessionGeneration;
|
|
1318
|
+
scheduleQuotaRetry(ctx, retryAfterSec, reason, () => {
|
|
1319
|
+
const current = freshCtxForGeneration(generation);
|
|
1320
|
+
if (!current) return;
|
|
1321
|
+
try {
|
|
1322
|
+
void Promise.resolve(fire(current)).catch((err) => {
|
|
1323
|
+
if (isStaleApiError(err)) extensionApiStale = true;
|
|
1324
|
+
});
|
|
1325
|
+
} catch (err) {
|
|
1326
|
+
if (isStaleApiError(err)) extensionApiStale = true;
|
|
1327
|
+
}
|
|
1328
|
+
}, label);
|
|
1329
|
+
}
|
|
1330
|
+
|
|
1256
1331
|
// v0.34.15 (hegemon 2026-08-01): pi ACCEPTED the continuation — footer showed
|
|
1257
1332
|
// "1 queued" — but the turn trigger was dead, so the message sat queued while
|
|
1258
1333
|
// pi idled. The 0.34.11 watchdog gates on "pi reported NO pending" and the
|
|
@@ -1285,7 +1360,7 @@ function armQueueStuckProbe(sentAt: number): void {
|
|
|
1285
1360
|
}
|
|
1286
1361
|
|
|
1287
1362
|
function scheduleContinuation(ctx: ExtensionContext, force = false, delayMs?: number): void {
|
|
1288
|
-
if (sessionHandoffPending || initialSessionLoadPending) return;
|
|
1363
|
+
if (sessionHandoffPending || initialSessionLoadPending || extensionApiStale || staleTerminalDone || zombieStoodDown) return;
|
|
1289
1364
|
abortedStandDown = false; // v0.29.5: any explicit schedule ends the stand-down
|
|
1290
1365
|
if (!isActionableGoal()) return;
|
|
1291
1366
|
rememberCtx(ctx);
|
|
@@ -1303,7 +1378,7 @@ function scheduleContinuation(ctx: ExtensionContext, force = false, delayMs?: nu
|
|
|
1303
1378
|
}
|
|
1304
1379
|
|
|
1305
1380
|
function sendContinuation(goalId: string): void {
|
|
1306
|
-
if (sessionHandoffPending || initialSessionLoadPending) return;
|
|
1381
|
+
if (sessionHandoffPending || initialSessionLoadPending || extensionApiStale || staleTerminalDone || zombieStoodDown) return;
|
|
1307
1382
|
continuationTimer = null;
|
|
1308
1383
|
continuationScheduledFor = null;
|
|
1309
1384
|
if (!isActionableGoal()) return;
|
|
@@ -1578,11 +1653,13 @@ function autoArbitrateStackedState(ctx: ExtensionContext): void {
|
|
|
1578
1653
|
* against the live queue), present DECIDE findings without queueing them.
|
|
1579
1654
|
* Confirm-gated like every bulk import (v0.23.7: the user reads what lands
|
|
1580
1655
|
* in the queue); a decline leaves the findings open for a later re-run.
|
|
1656
|
+
* v0.34.20: this detached operation retains only cwd + generation. Every
|
|
1657
|
+
* context use after the confirmation await must come from the fresh session.
|
|
1581
1658
|
*/
|
|
1582
|
-
async function fanOutListAuditFindings(
|
|
1659
|
+
async function fanOutListAuditFindings(cwd: string, generation: number): Promise<void> {
|
|
1583
1660
|
let md = "";
|
|
1584
1661
|
try {
|
|
1585
|
-
md = fs.readFileSync(path.join(
|
|
1662
|
+
md = fs.readFileSync(path.join(cwd, AUDIT_FINDINGS_REL), "utf-8");
|
|
1586
1663
|
} catch {
|
|
1587
1664
|
/* no findings file — the audit was clean or never wrote */
|
|
1588
1665
|
}
|
|
@@ -1594,6 +1671,8 @@ async function fanOutListAuditFindings(ctx: ExtensionContext): Promise<void> {
|
|
|
1594
1671
|
// hundreds of items on a single Confirm.
|
|
1595
1672
|
const fresh = open.filter((f) => !queuedText.includes(f.text.slice(0, 60))).slice(0, 50);
|
|
1596
1673
|
const alreadyQueued = open.length - fresh.length;
|
|
1674
|
+
const current = freshCtxForGeneration(generation);
|
|
1675
|
+
if (!current) return;
|
|
1597
1676
|
// v0.33.3: DECIDE findings are RAISED TO THE USER as real questions
|
|
1598
1677
|
// (hegemon 2026-07-31: a truncated notify left the user typing "decide
|
|
1599
1678
|
// what" into the void). The orchestrator can't call ask_user_question —
|
|
@@ -1603,41 +1682,48 @@ async function fanOutListAuditFindings(ctx: ExtensionContext): Promise<void> {
|
|
|
1603
1682
|
// queued or the fan-out was declined.
|
|
1604
1683
|
if (decisions.length > 0) {
|
|
1605
1684
|
const decList = decisions.slice(0, 8).map((d, i) => `${i + 1}. ${d.slice(0, 500)}`).join("\n");
|
|
1606
|
-
if (safeSteerUser(
|
|
1685
|
+
if (safeSteerUser(current,
|
|
1607
1686
|
`[DECIDE FINDINGS — user decisions needed] The audit surfaced ${decisions.length} DECIDE finding(s) — direction calls only the user can make (a decision is not a task, so they were NOT queued):\n${decList}\nRaise them to the user NOW with ask_user_question — one question per finding, options from the finding's own two sides plus "Defer" (prose numbered list if ask_user_question is unavailable; Esc = Defer). Then record every answer in ${AUDIT_FINDINGS_REL}: replace the "- [?]" line with "- [x] DECIDED: <what was chosen> (<date>)" (or "- [x] DEFERRED") so it stops re-surfacing, and queue any chosen work with list_add — do NOT start the work inline.`))
|
|
1608
|
-
appendLedger(
|
|
1687
|
+
appendLedger(cwd, "list_audit_decisions_raised", { decisions: decisions.length });
|
|
1609
1688
|
}
|
|
1610
1689
|
const decideNote =
|
|
1611
1690
|
decisions.length > 0
|
|
1612
1691
|
? ` ${decisions.length} DECIDE finding(s) need YOU — raising them as questions now (not queued — a decision is not a task).`
|
|
1613
1692
|
: "";
|
|
1614
1693
|
if (fresh.length === 0) {
|
|
1615
|
-
|
|
1694
|
+
const afterDecision = freshCtxForGeneration(generation);
|
|
1695
|
+
if (!afterDecision) return;
|
|
1696
|
+
afterDecision.ui.notify(
|
|
1616
1697
|
open.length > 0
|
|
1617
1698
|
? `Audit collected ${open.length} open finding(s) — all already queued.${decideNote}`
|
|
1618
1699
|
: `Audit complete — no open findings; the project is clean, nothing to queue.${decideNote}`,
|
|
1619
1700
|
"info",
|
|
1620
1701
|
);
|
|
1621
|
-
appendLedger(
|
|
1702
|
+
appendLedger(cwd, "list_audit_fanout_empty", { open: open.length, decisions: decisions.length });
|
|
1622
1703
|
return;
|
|
1623
1704
|
}
|
|
1624
1705
|
const preview = fresh.map((f, i) => ` ${i + 1}. ${f.text.slice(0, 110)}`).join("\n");
|
|
1625
1706
|
let confirmed = true;
|
|
1626
|
-
|
|
1707
|
+
const beforeConfirm = freshCtxForGeneration(generation);
|
|
1708
|
+
if (!beforeConfirm) return;
|
|
1709
|
+
if (beforeConfirm.hasUI) {
|
|
1627
1710
|
try {
|
|
1628
|
-
confirmed = await
|
|
1711
|
+
confirmed = await beforeConfirm.ui.confirm(`Queue ${fresh.length} audit finding(s) as list items?`, preview);
|
|
1629
1712
|
} catch {
|
|
1630
1713
|
confirmed = false;
|
|
1631
1714
|
}
|
|
1632
1715
|
}
|
|
1716
|
+
// A confirm result from an old session is not consent for the replacement.
|
|
1717
|
+
const afterConfirm = freshCtxForGeneration(generation);
|
|
1718
|
+
if (!afterConfirm) return;
|
|
1633
1719
|
if (!confirmed) {
|
|
1634
|
-
appendLedger(
|
|
1635
|
-
|
|
1720
|
+
appendLedger(cwd, "list_audit_fanout_declined", { findings: fresh.length });
|
|
1721
|
+
afterConfirm.ui.notify(`Fan-out declined — the findings stay open in ${AUDIT_FINDINGS_REL}; /list audit re-queues them any time.`, "info");
|
|
1636
1722
|
return;
|
|
1637
1723
|
}
|
|
1638
|
-
const n = enqueueItems(
|
|
1639
|
-
appendLedger(
|
|
1640
|
-
|
|
1724
|
+
const n = enqueueItems(afterConfirm, fresh.map((f) => listAuditFanoutItemText(f.text)), "list audit fan-out");
|
|
1725
|
+
appendLedger(cwd, "list_audit_fanout", { queued: n, alreadyQueued, decisions: decisions.length });
|
|
1726
|
+
afterConfirm.ui.notify(
|
|
1641
1727
|
`Queued ${n} finding(s) — the list drains them fix by fix, each with its own audited commit.${alreadyQueued > 0 ? ` (${alreadyQueued} already queued.)` : ""}${decideNote}`,
|
|
1642
1728
|
"info",
|
|
1643
1729
|
);
|
|
@@ -1677,10 +1763,13 @@ function archiveCurrentGoal(ctx: ExtensionContext, status: Status, stopReason?:
|
|
|
1677
1763
|
const isListAuditCollect = goal.objective.includes(LIST_AUDIT_COLLECT_MARKER);
|
|
1678
1764
|
// v0.34.7: the float gets a catch — ANY rejection here used to become
|
|
1679
1765
|
// an uncaughtException and kill pi (darklord 2026-08-01).
|
|
1680
|
-
if (isListAuditCollect)
|
|
1681
|
-
|
|
1682
|
-
|
|
1766
|
+
if (isListAuditCollect) {
|
|
1767
|
+
const fanoutCwd = ctx.cwd;
|
|
1768
|
+
const fanoutGeneration = sessionGeneration;
|
|
1769
|
+
void fanOutListAuditFindings(fanoutCwd, fanoutGeneration).catch((err) => {
|
|
1770
|
+
appendLedger(fanoutCwd, "list_audit_fanout_error", { error: String(err).slice(0, 200) });
|
|
1683
1771
|
});
|
|
1772
|
+
}
|
|
1684
1773
|
const advanced = activateNextListItem(ctx);
|
|
1685
1774
|
// v0.26.0: the queue just EMPTIED on a completion → list-complete.
|
|
1686
1775
|
if (!advanced && !isListAuditCollect) {
|
|
@@ -1713,11 +1802,17 @@ function archiveCurrentGoal(ctx: ExtensionContext, status: Status, stopReason?:
|
|
|
1713
1802
|
* infra) → hand back to the agent: resume active + continuation, verdict
|
|
1714
1803
|
* durable in auditHistory.
|
|
1715
1804
|
*/
|
|
1716
|
-
async function retryStoredCompletionAudit(
|
|
1805
|
+
async function retryStoredCompletionAudit(origin: "quota-retry" | "manual" = "quota-retry"): Promise<void> {
|
|
1717
1806
|
const goal = state.goal;
|
|
1718
1807
|
if (!goal?.pendingCompletion) return;
|
|
1808
|
+
const goalId = goal.id;
|
|
1719
1809
|
if (completionAuditInFlight) return;
|
|
1720
|
-
const
|
|
1810
|
+
const generation = sessionGeneration;
|
|
1811
|
+
// Delayed audit recovery has no safe fallback: if the current generation
|
|
1812
|
+
// cannot be proven live, the fresh session must rehydrate the durable claim.
|
|
1813
|
+
const initialCtx = freshCtxForGeneration(generation);
|
|
1814
|
+
if (!initialCtx) return;
|
|
1815
|
+
let liveCtx: ExtensionContext = initialCtx;
|
|
1721
1816
|
const claim = goal.pendingCompletion;
|
|
1722
1817
|
updateGoal({ status: "auditing" }, liveCtx);
|
|
1723
1818
|
appendLedger(liveCtx.cwd, "goal_resumed", { via: origin === "manual" ? "manual-audit" : "quota-retry-direct-audit" });
|
|
@@ -1729,6 +1824,7 @@ async function retryStoredCompletionAudit(ctx: ExtensionContext, origin: "quota-
|
|
|
1729
1824
|
if (modelError) liveCtx.ui.notify(`Auditor model issue: ${modelError}`, "warning");
|
|
1730
1825
|
latestAuditProgress = { label: "quota-retry", lastEventAt: Date.now() };
|
|
1731
1826
|
completionAuditInFlight = true;
|
|
1827
|
+
completionAuditGeneration = generation;
|
|
1732
1828
|
const auditStartMs = Date.now();
|
|
1733
1829
|
let result: Awaited<ReturnType<typeof runGoalCompletionAuditor>>;
|
|
1734
1830
|
try {
|
|
@@ -1736,23 +1832,36 @@ async function retryStoredCompletionAudit(ctx: ExtensionContext, origin: "quota-
|
|
|
1736
1832
|
() =>
|
|
1737
1833
|
runGoalCompletionAuditor({
|
|
1738
1834
|
ctx: liveCtx,
|
|
1739
|
-
goal
|
|
1835
|
+
goal,
|
|
1740
1836
|
completionSummary: claim.completionSummary,
|
|
1741
1837
|
verificationSummary: claim.verificationSummary,
|
|
1742
1838
|
model: auditorModel,
|
|
1743
1839
|
thinkingLevel: (settings.auditorThinkingLevel ?? "high") as any, // may be "max" — pi ≥0.83 understands it; the dev-types predate it
|
|
1744
1840
|
onProgress: (progress) => {
|
|
1841
|
+
const current = freshCtxForGeneration(generation);
|
|
1842
|
+
if (!current) return;
|
|
1745
1843
|
latestAuditProgress = { currentTool: progress.currentTool, label: progress.label, elapsedMs: progress.elapsedMs, lastEventAt: Date.now() };
|
|
1746
|
-
refreshUI(
|
|
1844
|
+
refreshUI(current);
|
|
1747
1845
|
},
|
|
1748
1846
|
}),
|
|
1749
|
-
{
|
|
1847
|
+
{
|
|
1848
|
+
shouldRetry: () => freshCtxForGeneration(generation) !== null,
|
|
1849
|
+
onRetry: (err) => {
|
|
1850
|
+
const current = freshCtxForGeneration(generation);
|
|
1851
|
+
if (current) appendLedger(current.cwd, "audit_infra_retry", { goalId, error: err.slice(0, 200) });
|
|
1852
|
+
},
|
|
1853
|
+
},
|
|
1750
1854
|
));
|
|
1751
1855
|
} finally {
|
|
1752
|
-
|
|
1753
|
-
|
|
1856
|
+
if (completionAuditGeneration === generation) {
|
|
1857
|
+
completionAuditInFlight = false;
|
|
1858
|
+
completionAuditGeneration = null;
|
|
1859
|
+
latestAuditProgress = null;
|
|
1860
|
+
}
|
|
1754
1861
|
}
|
|
1755
|
-
|
|
1862
|
+
const currentAfterAudit = freshCtxForGeneration(generation);
|
|
1863
|
+
if (!currentAfterAudit || !state.goal || state.goal.id !== goalId) return; // replacement/stale/goal boundary — fresh session rebinds durable state
|
|
1864
|
+
liveCtx = currentAfterAudit;
|
|
1756
1865
|
|
|
1757
1866
|
// Record the run in history (same compact shape as the tool path).
|
|
1758
1867
|
const auditorRan = result.output.trim().length > 0;
|
|
@@ -1809,9 +1918,9 @@ async function retryStoredCompletionAudit(ctx: ExtensionContext, origin: "quota-
|
|
|
1809
1918
|
return;
|
|
1810
1919
|
}
|
|
1811
1920
|
liveCtx.ui.notify(`Auditor still quota-limited — next auto-retry in ${retryMin}m (your completion claim is stored; no action needed).`, "warning");
|
|
1812
|
-
|
|
1921
|
+
scheduleQuotaRetryForSession(liveCtx, quota.retryAfterSec, result.error, (fresh) => {
|
|
1813
1922
|
if (state.goal && state.goal.status === "paused" && (state.goal.pauseReason ?? "").startsWith("auditor quota:") && state.goal.pendingCompletion) {
|
|
1814
|
-
void retryStoredCompletionAudit(
|
|
1923
|
+
void retryStoredCompletionAudit(origin);
|
|
1815
1924
|
}
|
|
1816
1925
|
});
|
|
1817
1926
|
return;
|
|
@@ -2111,7 +2220,7 @@ async function cmdGoal(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
2111
2220
|
},
|
|
2112
2221
|
}, ctx);
|
|
2113
2222
|
appendLedger(ctx.cwd, "manual_audit_requested", { goalId: state.goal.id });
|
|
2114
|
-
void retryStoredCompletionAudit(
|
|
2223
|
+
void retryStoredCompletionAudit("manual");
|
|
2115
2224
|
return;
|
|
2116
2225
|
}
|
|
2117
2226
|
if (route.name === "tweak") return cmdTweak(route.rest, ctx);
|
|
@@ -2263,8 +2372,17 @@ async function cmdResume(ctx: ExtensionContext): Promise<void> {
|
|
|
2263
2372
|
// marker's promise ("a fresh session will resume you") is fulfilled by a
|
|
2264
2373
|
// manual resume exactly as by an automatic one. (staleEntry still re-marks
|
|
2265
2374
|
// below — a resume inside a stale session is a NEW interrupt.)
|
|
2375
|
+
const storedCompletion = state.goal.pendingCompletion;
|
|
2266
2376
|
updateGoal({ status: "active", pauseReason: undefined, pauseSuggestedAction: undefined, pauseKind: undefined, pauseOptions: undefined, pauseRecommended: undefined, pauseResumeAt: undefined, interruptedAt: undefined, interruptedReason: undefined, ...(staleEntry ? { interruptedAt: nowIso(), interruptedReason: "resumed in a stale session" } : {}), ...(usage ? { usage } : {}) }, ctx);
|
|
2267
2377
|
if (staleEntry) return;
|
|
2378
|
+
// A stored completion claim is a direct-audit resume, not an agent turn.
|
|
2379
|
+
// Keeping the claim while merely scheduling a continuation left manual
|
|
2380
|
+
// pause/resume with an ACTIVE goal that no timer would ever consume.
|
|
2381
|
+
if (storedCompletion) {
|
|
2382
|
+
ctx.ui.notify("Resuming the stored completion claim — running the isolated auditor directly (no agent turn needed).", "info");
|
|
2383
|
+
void retryStoredCompletionAudit("manual");
|
|
2384
|
+
return;
|
|
2385
|
+
}
|
|
2268
2386
|
// v0.22.5: say what was resumed — with a non-empty list this also resumes
|
|
2269
2387
|
// the queue (the active goal IS the list's head item).
|
|
2270
2388
|
// v0.22.7: name WHAT was resumed — list items resume through /list.
|
|
@@ -2910,7 +3028,7 @@ function loopPrompt(loop: LoopState, regressionNote: string, strategyNote: strin
|
|
|
2910
3028
|
}
|
|
2911
3029
|
|
|
2912
3030
|
function scheduleLoopTick(ctx: ExtensionContext): void {
|
|
2913
|
-
if (sessionHandoffPending || initialSessionLoadPending || !isLoopActive()) return;
|
|
3031
|
+
if (sessionHandoffPending || initialSessionLoadPending || extensionApiStale || staleTerminalDone || zombieStoodDown || !isLoopActive()) return;
|
|
2914
3032
|
rememberCtx(ctx);
|
|
2915
3033
|
clearLoopTimer();
|
|
2916
3034
|
let delay = 0;
|
|
@@ -2923,7 +3041,7 @@ function scheduleLoopTick(ctx: ExtensionContext): void {
|
|
|
2923
3041
|
}
|
|
2924
3042
|
|
|
2925
3043
|
function sendLoopTurn(): void {
|
|
2926
|
-
if (sessionHandoffPending || initialSessionLoadPending) return;
|
|
3044
|
+
if (sessionHandoffPending || initialSessionLoadPending || extensionApiStale || staleTerminalDone || zombieStoodDown) return;
|
|
2927
3045
|
loopTimer = null;
|
|
2928
3046
|
if (!isLoopActive() || !extensionApi) return;
|
|
2929
3047
|
const ctx = freshCtx();
|
|
@@ -3043,7 +3161,20 @@ function sendLoopTurn(): void {
|
|
|
3043
3161
|
}
|
|
3044
3162
|
|
|
3045
3163
|
/** agent_end hook for loop 3: measure → judge → continue or stop. */
|
|
3046
|
-
async function runLoopTick(
|
|
3164
|
+
async function runLoopTick(initialCtx: ExtensionContext, event?: any): Promise<void> {
|
|
3165
|
+
// v0.34.20: measurement/git work is asynchronous. Rebind the local
|
|
3166
|
+
// context after every await or abandon the tick; never let a replacement
|
|
3167
|
+
// session inherit the agent_end context.
|
|
3168
|
+
const generation = sessionGeneration;
|
|
3169
|
+
const initial = freshCtxForGeneration(generation);
|
|
3170
|
+
if (!initial) return;
|
|
3171
|
+
let ctx: ExtensionContext = initial;
|
|
3172
|
+
const rebind = (): boolean => {
|
|
3173
|
+
const current = freshCtxForGeneration(generation);
|
|
3174
|
+
if (!current) return false;
|
|
3175
|
+
ctx = current;
|
|
3176
|
+
return true;
|
|
3177
|
+
};
|
|
3047
3178
|
const loop = state.loop!;
|
|
3048
3179
|
// v0.15.0: token budget is an arbitrary bound; accumulate orchestrator-side.
|
|
3049
3180
|
if (event?.messages) {
|
|
@@ -3051,6 +3182,7 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
|
|
|
3051
3182
|
}
|
|
3052
3183
|
const metricless = !loop.measureCmd;
|
|
3053
3184
|
const value = metricless ? null : await runMeasure(ctx, loop.measureCmd!);
|
|
3185
|
+
if (!rebind()) return;
|
|
3054
3186
|
// Hypothesis line (pi-autoresearch's good idea): the agent's stated intent
|
|
3055
3187
|
// for the turn goes into the ledger, making loop history auditable.
|
|
3056
3188
|
let hypothesis: string | undefined;
|
|
@@ -3075,10 +3207,12 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
|
|
|
3075
3207
|
const iterStartHead = loop.iterMetrics?.iterationStartHead;
|
|
3076
3208
|
const iterStartAt = loop.iterMetrics?.iterationStartAt;
|
|
3077
3209
|
const currentHeadRes = await runGit(ctx, ["rev-parse", "HEAD"]);
|
|
3210
|
+
if (!rebind()) return;
|
|
3078
3211
|
const currentHead = currentHeadRes.ok ? currentHeadRes.stdout : undefined;
|
|
3079
3212
|
let gitCommits = 0;
|
|
3080
3213
|
if (iterStartHead && currentHead && iterStartHead !== currentHead) {
|
|
3081
3214
|
const countRes = await runGit(ctx, ["rev-list", "--count", `${iterStartHead}..HEAD`]);
|
|
3215
|
+
if (!rebind()) return;
|
|
3082
3216
|
const n = Number.parseInt(countRes.stdout, 10);
|
|
3083
3217
|
if (countRes.ok && Number.isFinite(n) && n > 0) gitCommits = n;
|
|
3084
3218
|
}
|
|
@@ -3187,10 +3321,13 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
|
|
|
3187
3321
|
if (loop.branchName && outcome.kind === "continue") {
|
|
3188
3322
|
if (metricless || outcome.improved) {
|
|
3189
3323
|
await runGit(ctx, ["add", "-A"]);
|
|
3324
|
+
if (!rebind()) return;
|
|
3190
3325
|
const committed = await runGit(ctx, ["commit", "-m", metricless ? `pi-glla-loop: iteration ${loop.iteration}` : `pi-glla-loop: iteration ${loop.iteration} (${loop.direction}=${loop.bestValue})`]);
|
|
3326
|
+
if (!rebind()) return;
|
|
3191
3327
|
appendLedger(ctx.cwd, "loop_git", { action: "commit", iteration: loop.iteration, ok: committed.ok });
|
|
3192
3328
|
} else {
|
|
3193
3329
|
const reset = await runGit(ctx, ["reset", "--hard", "HEAD"]);
|
|
3330
|
+
if (!rebind()) return;
|
|
3194
3331
|
appendLedger(ctx.cwd, "loop_git", { action: "reset", iteration: loop.iteration, ok: reset.ok });
|
|
3195
3332
|
}
|
|
3196
3333
|
persistState(ctx);
|
|
@@ -3204,6 +3341,7 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
|
|
|
3204
3341
|
loop.stopReason = `stuck — ${loop.lastStuckReason} (${loop.consecutiveStuck} consecutive interventions)`;
|
|
3205
3342
|
persistState(ctx);
|
|
3206
3343
|
await finishLoopGit(ctx, loop);
|
|
3344
|
+
if (!rebind()) return;
|
|
3207
3345
|
ctx.ui.notify(`Loop stopped: ${loop.stopReason}. ${loop.history.length} iterations recorded.`, "warning");
|
|
3208
3346
|
appendLedger(ctx.cwd, "loop_stopped", { reason: loop.stopReason, iterations: loop.iteration, best: loop.bestValue });
|
|
3209
3347
|
notifyExternal(ctx, `Loop stopped: ${loop.stopReason}`);
|
|
@@ -3240,6 +3378,7 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
|
|
|
3240
3378
|
}
|
|
3241
3379
|
}
|
|
3242
3380
|
await finishLoopGit(ctx, loop);
|
|
3381
|
+
if (!rebind()) return;
|
|
3243
3382
|
ctx.ui.notify(`Loop stopped: ${outcome.reason}. ${loop.history.length} iterations recorded.`, "info");
|
|
3244
3383
|
appendLedger(ctx.cwd, "loop_stopped", { reason: outcome.reason, iterations: loop.iteration, best: loop.bestValue });
|
|
3245
3384
|
notifyExternal(ctx, `Loop stopped: ${outcome.reason}`);
|
|
@@ -3252,10 +3391,17 @@ async function runLoopTick(ctx: ExtensionContext, event?: any): Promise<void> {
|
|
|
3252
3391
|
* where the work lives and how to merge it. Scratch branch is never deleted. */
|
|
3253
3392
|
async function finishLoopGit(ctx: ExtensionContext, loop: LoopState): Promise<void> {
|
|
3254
3393
|
if (!loop.branchName) return;
|
|
3394
|
+
const generation = sessionGeneration;
|
|
3255
3395
|
// Uncommitted remnants (final stalled iterations were reset already, but be safe).
|
|
3256
3396
|
await runGit(ctx, ["reset", "--hard", "HEAD"]);
|
|
3397
|
+
const afterReset = freshCtxForGeneration(generation);
|
|
3398
|
+
if (!afterReset) return;
|
|
3399
|
+
ctx = afterReset;
|
|
3257
3400
|
if (loop.originalBranch) {
|
|
3258
3401
|
await runGit(ctx, ["checkout", loop.originalBranch]);
|
|
3402
|
+
const afterCheckout = freshCtxForGeneration(generation);
|
|
3403
|
+
if (!afterCheckout) return;
|
|
3404
|
+
ctx = afterCheckout;
|
|
3259
3405
|
}
|
|
3260
3406
|
ctx.ui.notify(
|
|
3261
3407
|
`Loop work is on branch ${loop.branchName} (${loop.iteration} iterations, best ${loop.bestValue ?? "n/a"}).\nMerge with: git merge ${loop.branchName} — or delete with: git branch -D ${loop.branchName}`,
|
|
@@ -3507,7 +3653,11 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
3507
3653
|
clearLoopTimer();
|
|
3508
3654
|
state.loop = { ...state.loop, active: false, stopReason: state.loop.stopReason ?? `stopped by user (/loop ${sub})` };
|
|
3509
3655
|
persistState(ctx);
|
|
3656
|
+
const stopGeneration = sessionGeneration;
|
|
3510
3657
|
await finishLoopGit(ctx, state.loop);
|
|
3658
|
+
const afterFinish = freshCtxForGeneration(stopGeneration);
|
|
3659
|
+
if (!afterFinish) return;
|
|
3660
|
+
ctx = afterFinish;
|
|
3511
3661
|
appendLedger(ctx.cwd, "loop_stopped", { reason: "user", iterations: state.loop.iteration, best: state.loop.bestValue });
|
|
3512
3662
|
ctx.ui.notify(
|
|
3513
3663
|
`Loop stopped after ${state.loop.iteration} iterations. Best: ${state.loop.bestValue ?? "n/a"}.`,
|
|
@@ -3528,7 +3678,11 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
3528
3678
|
const reason = loopFinishStopReason(rest);
|
|
3529
3679
|
state.loop = { ...state.loop, active: false, stopReason: reason };
|
|
3530
3680
|
persistState(ctx);
|
|
3681
|
+
const finishGeneration = sessionGeneration;
|
|
3531
3682
|
await finishLoopGit(ctx, state.loop);
|
|
3683
|
+
const afterFinish = freshCtxForGeneration(finishGeneration);
|
|
3684
|
+
if (!afterFinish) return;
|
|
3685
|
+
ctx = afterFinish;
|
|
3532
3686
|
appendLedger(ctx.cwd, "loop_stopped", { reason, iterations: state.loop.iteration, best: state.loop.bestValue });
|
|
3533
3687
|
ctx.ui.notify(
|
|
3534
3688
|
`Loop finished (${reason}) after ${state.loop.iteration} iterations. Best: ${state.loop.bestValue ?? "n/a"}.`,
|
|
@@ -3657,7 +3811,34 @@ async function cmdLoop(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
3657
3811
|
// Tools exposed to the agent
|
|
3658
3812
|
// =================================================================
|
|
3659
3813
|
|
|
3660
|
-
|
|
3814
|
+
const STALE_TOOL_CONTEXT_MESSAGE =
|
|
3815
|
+
"This tool call crossed a session replacement before it could run. No stale context was used; wait for a fresh session_start and retry.";
|
|
3816
|
+
|
|
3817
|
+
function staleToolResult(): { content: Array<{ type: "text"; text: string }>; details: Record<string, never> } {
|
|
3818
|
+
return { content: [{ type: "text", text: STALE_TOOL_CONTEXT_MESSAGE }], details: {} };
|
|
3819
|
+
}
|
|
3820
|
+
|
|
3821
|
+
/**
|
|
3822
|
+
* v0.34.20: registerAgentTools runs once per extension instance, but pi
|
|
3823
|
+
* invokes the registered tool with the current event context. Never use the
|
|
3824
|
+
* context captured when the tools were registered after a reload/rebind.
|
|
3825
|
+
* Prefer the invocation context, validate it cheaply, and fall back only to
|
|
3826
|
+
* the current fresh context — never to the registration-time ctx.
|
|
3827
|
+
*/
|
|
3828
|
+
function currentToolContext(execCtx: unknown): ExtensionContext | null {
|
|
3829
|
+
const candidate = execCtx as ExtensionContext | undefined;
|
|
3830
|
+
if (candidate) {
|
|
3831
|
+
try {
|
|
3832
|
+
candidate.isIdle();
|
|
3833
|
+
return candidate;
|
|
3834
|
+
} catch {
|
|
3835
|
+
// The invocation itself may be a late event; try the current binding.
|
|
3836
|
+
}
|
|
3837
|
+
}
|
|
3838
|
+
return freshCtx();
|
|
3839
|
+
}
|
|
3840
|
+
|
|
3841
|
+
function registerAgentTools(pi: any): void {
|
|
3661
3842
|
pi.registerTool(defineTool({
|
|
3662
3843
|
name: "complete_goal",
|
|
3663
3844
|
label: "Complete goal",
|
|
@@ -3670,6 +3851,10 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3670
3851
|
async execute(_id, params, signal, _onUpdate, execCtx) {
|
|
3671
3852
|
const foreign0 = foreignToolGuard(execCtx);
|
|
3672
3853
|
if (foreign0) return { content: [{ type: "text", text: foreign0 }], details: {} };
|
|
3854
|
+
const toolCtx = currentToolContext(execCtx);
|
|
3855
|
+
if (!toolCtx) return staleToolResult();
|
|
3856
|
+
let ctx: ExtensionContext = toolCtx;
|
|
3857
|
+
const auditGeneration = sessionGeneration;
|
|
3673
3858
|
if (!state.goal || state.goal.status !== "active") {
|
|
3674
3859
|
return { content: [{ type: "text", text: "No active goal." }], details: {} };
|
|
3675
3860
|
}
|
|
@@ -3685,7 +3870,22 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3685
3870
|
appendLedger(ctx.cwd, "goal_tweaked", { via: "complete_goal.newObjective", from: oldObjective.slice(0, 200), to: cleanObj.slice(0, 200) });
|
|
3686
3871
|
ctx.ui.notify(`Objective updated (complete_goal newObjective): ${cleanObj.slice(0, 80)}`, "info");
|
|
3687
3872
|
}
|
|
3688
|
-
|
|
3873
|
+
// v0.34.20: persist the completion claim BEFORE the isolated auditor
|
|
3874
|
+
// starts. If session replacement lands during the audit, a fresh
|
|
3875
|
+
// session can recover the exact claim instead of leaving an untracked
|
|
3876
|
+
// goal stuck in `auditing`.
|
|
3877
|
+
updateGoal({
|
|
3878
|
+
status: "auditing",
|
|
3879
|
+
pendingTasks: undefined,
|
|
3880
|
+
pendingCompletion: {
|
|
3881
|
+
completionSummary: p.completionSummary,
|
|
3882
|
+
verificationSummary: p.verificationSummary,
|
|
3883
|
+
at: nowIso(),
|
|
3884
|
+
},
|
|
3885
|
+
}, ctx);
|
|
3886
|
+
const auditGoal = state.goal;
|
|
3887
|
+
if (!auditGoal) return staleToolResult();
|
|
3888
|
+
const auditGoalId = auditGoal.id;
|
|
3689
3889
|
const settings = loadSettings(ctx.cwd);
|
|
3690
3890
|
const { model: auditorModel, error: modelError, via } = resolveAuditorModel(ctx, settings.auditorModel, settings.auditorModelFallback, settings.auditorSameSessionSwap !== false);
|
|
3691
3891
|
if (modelError) {
|
|
@@ -3698,20 +3898,22 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3698
3898
|
const runAudit = () =>
|
|
3699
3899
|
runGoalCompletionAuditor({
|
|
3700
3900
|
ctx,
|
|
3701
|
-
goal:
|
|
3901
|
+
goal: auditGoal,
|
|
3702
3902
|
completionSummary: p.completionSummary,
|
|
3703
3903
|
verificationSummary: p.verificationSummary,
|
|
3704
3904
|
model: auditorModel,
|
|
3705
3905
|
thinkingLevel: (settings.auditorThinkingLevel ?? "high") as any, // may be "max" — pi ≥0.83 understands it; the dev-types predate it
|
|
3706
3906
|
signal: signal ?? undefined,
|
|
3707
3907
|
onProgress: (progress) => {
|
|
3908
|
+
const current = freshCtxForGeneration(auditGeneration);
|
|
3909
|
+
if (!current) return;
|
|
3708
3910
|
latestAuditProgress = {
|
|
3709
3911
|
currentTool: progress.currentTool,
|
|
3710
3912
|
label: progress.label,
|
|
3711
3913
|
elapsedMs: progress.elapsedMs,
|
|
3712
3914
|
lastEventAt: Date.now(),
|
|
3713
3915
|
};
|
|
3714
|
-
refreshUI(
|
|
3916
|
+
refreshUI(current);
|
|
3715
3917
|
},
|
|
3716
3918
|
});
|
|
3717
3919
|
// v0.25.4 (post-audit fix): a retriable infra failure (stream error,
|
|
@@ -3720,19 +3922,33 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3720
3922
|
// (retried once)". Neither attempt is a verdict on the work.
|
|
3721
3923
|
const auditStartMs = Date.now();
|
|
3722
3924
|
completionAuditInFlight = true;
|
|
3925
|
+
completionAuditGeneration = auditGeneration;
|
|
3723
3926
|
let result: Awaited<ReturnType<typeof runAudit>>;
|
|
3724
3927
|
let retriedOnce = false;
|
|
3725
3928
|
try {
|
|
3726
3929
|
({ result, retriedOnce } = await runWithInfraRetry(runAudit, {
|
|
3930
|
+
shouldRetry: () => freshCtxForGeneration(auditGeneration) !== null,
|
|
3727
3931
|
onRetry: (err) => {
|
|
3932
|
+
const current = freshCtxForGeneration(auditGeneration);
|
|
3933
|
+
if (!current) return;
|
|
3728
3934
|
latestAuditProgress = { label: `infra error (${err.slice(0, 40)}) — retrying once`, lastEventAt: Date.now() };
|
|
3729
|
-
refreshUI(
|
|
3730
|
-
appendLedger(
|
|
3935
|
+
refreshUI(current);
|
|
3936
|
+
appendLedger(current.cwd, "audit_infra_retry", { goalId: auditGoalId, error: err.slice(0, 200) });
|
|
3731
3937
|
},
|
|
3732
3938
|
}));
|
|
3733
3939
|
} finally {
|
|
3734
|
-
|
|
3940
|
+
if (completionAuditGeneration === auditGeneration) {
|
|
3941
|
+
completionAuditInFlight = false;
|
|
3942
|
+
completionAuditGeneration = null;
|
|
3943
|
+
latestAuditProgress = null;
|
|
3944
|
+
}
|
|
3735
3945
|
}
|
|
3946
|
+
const auditContextAfterRun = freshCtxForGeneration(auditGeneration);
|
|
3947
|
+
if (!auditContextAfterRun || !state.goal || state.goal.id !== auditGoalId) {
|
|
3948
|
+
if (completionAuditGeneration === auditGeneration) latestAuditProgress = null;
|
|
3949
|
+
return staleToolResult();
|
|
3950
|
+
}
|
|
3951
|
+
ctx = auditContextAfterRun;
|
|
3736
3952
|
const auditDurationMs = Date.now() - auditStartMs;
|
|
3737
3953
|
latestAuditProgress = null;
|
|
3738
3954
|
// Audit history: record REAL verdicts only — a non-empty report is the
|
|
@@ -3801,7 +4017,10 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3801
4017
|
// Escape hatch: the user aborted the audit (Esc). Offer the explicit
|
|
3802
4018
|
// choice — complete WITHOUT audit, or keep working. (pi-goal-x parity.)
|
|
3803
4019
|
if (result.error === "Auditor aborted.") {
|
|
3804
|
-
updateGoal({ status: "active", auditHistory: history, pauseReason: "audit aborted by user (Esc)" }, ctx);
|
|
4020
|
+
updateGoal({ status: "active", auditHistory: history, pendingCompletion: undefined, pauseReason: "audit aborted by user (Esc)" }, ctx);
|
|
4021
|
+
const abortConfirmCtx = freshCtxForGeneration(auditGeneration);
|
|
4022
|
+
if (!abortConfirmCtx) return staleToolResult();
|
|
4023
|
+
ctx = abortConfirmCtx;
|
|
3805
4024
|
let completeAnyway = false;
|
|
3806
4025
|
try {
|
|
3807
4026
|
completeAnyway = await ctx.ui.confirm(
|
|
@@ -3811,8 +4030,11 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3811
4030
|
} catch {
|
|
3812
4031
|
completeAnyway = false;
|
|
3813
4032
|
}
|
|
4033
|
+
const afterAbortConfirmCtx = freshCtxForGeneration(auditGeneration);
|
|
4034
|
+
if (!afterAbortConfirmCtx) return staleToolResult();
|
|
4035
|
+
ctx = afterAbortConfirmCtx;
|
|
3814
4036
|
if (completeAnyway) {
|
|
3815
|
-
updateGoal({ auditHistory: history }, ctx);
|
|
4037
|
+
updateGoal({ auditHistory: history, pendingCompletion: undefined }, ctx);
|
|
3816
4038
|
archiveCurrentGoal(ctx, "complete", "completed without audit (user choice after Esc)");
|
|
3817
4039
|
return { content: [{ type: "text", text: "Goal marked complete without audit (user choice)." }], details: {} };
|
|
3818
4040
|
}
|
|
@@ -3824,7 +4046,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3824
4046
|
}
|
|
3825
4047
|
|
|
3826
4048
|
if (result.approved) {
|
|
3827
|
-
updateGoal({ auditHistory: history }, ctx);
|
|
4049
|
+
updateGoal({ auditHistory: history, pendingCompletion: undefined }, ctx);
|
|
3828
4050
|
const objective = state.goal.objective;
|
|
3829
4051
|
archiveCurrentGoal(ctx, "complete", `auditor ${result.model} approved`);
|
|
3830
4052
|
notifyExternal(ctx, `Goal complete (auditor approved): ${objective.slice(0, 120)}`);
|
|
@@ -3846,6 +4068,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3846
4068
|
updateGoal({
|
|
3847
4069
|
status: "active",
|
|
3848
4070
|
auditHistory: history,
|
|
4071
|
+
pendingCompletion: undefined,
|
|
3849
4072
|
pauseReason: `auditor verdict: IMPOSSIBLE (partial) — ${reason}`,
|
|
3850
4073
|
pauseSuggestedAction: "Narrow the objective past the impossible part (complete_goal newObjective or /goal tweak) and continue",
|
|
3851
4074
|
}, ctx);
|
|
@@ -3863,6 +4086,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3863
4086
|
updateGoal({
|
|
3864
4087
|
status: "paused",
|
|
3865
4088
|
auditHistory: history,
|
|
4089
|
+
pendingCompletion: undefined,
|
|
3866
4090
|
pauseKind: "decision",
|
|
3867
4091
|
pauseOptions: ["Tweak the objective — /goal tweak <new text>", "Cancel the goal (/goal cancel)"],
|
|
3868
4092
|
pauseRecommended: 1,
|
|
@@ -3908,7 +4132,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3908
4132
|
pauseSuggestedAction: `Quota auto-retry in ${retryMin}m — or /goal resume to retry now`,
|
|
3909
4133
|
}, ctx);
|
|
3910
4134
|
appendLedger(ctx.cwd, "goal_paused", { reason: `auditor quota: retry in ${quota.retryAfterSec}s (${quota.fromUpstream ? "upstream hint" : "default"})` });
|
|
3911
|
-
|
|
4135
|
+
scheduleQuotaRetryForSession(ctx, quota.retryAfterSec, result.error, (fresh) => {
|
|
3912
4136
|
// Re-check: only auto-resume if STILL paused for the quota
|
|
3913
4137
|
// reason (a user /goal pause during the window is not stomped).
|
|
3914
4138
|
if (state.goal && state.goal.status === "paused" && (state.goal.pauseReason ?? "").startsWith("auditor quota:")) {
|
|
@@ -3916,15 +4140,15 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3916
4140
|
// agent is not needed to re-submit an unchanged claim, and
|
|
3917
4141
|
// re-engaging it produced hallucinated-closure loops.
|
|
3918
4142
|
if (state.goal.pendingCompletion) {
|
|
3919
|
-
void retryStoredCompletionAudit(
|
|
4143
|
+
void retryStoredCompletionAudit();
|
|
3920
4144
|
return;
|
|
3921
4145
|
}
|
|
3922
|
-
updateGoal({ status: "active" },
|
|
3923
|
-
appendLedger(
|
|
3924
|
-
if (resolveEffectiveAggressiveSettings(loadSettings(
|
|
3925
|
-
|
|
4146
|
+
updateGoal({ status: "active" }, fresh);
|
|
4147
|
+
appendLedger(fresh.cwd, "goal_resumed", { via: "quota-retry" });
|
|
4148
|
+
if (resolveEffectiveAggressiveSettings(loadSettings(fresh.cwd)).aggressiveMode) {
|
|
4149
|
+
fresh.ui.notify("Auto-resume fired (event: auditor quota window elapsed). Continue working.", "info");
|
|
3926
4150
|
}
|
|
3927
|
-
scheduleContinuation(
|
|
4151
|
+
scheduleContinuation(fresh, true);
|
|
3928
4152
|
}
|
|
3929
4153
|
});
|
|
3930
4154
|
return {
|
|
@@ -3944,6 +4168,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3944
4168
|
updateGoal({
|
|
3945
4169
|
status: "paused",
|
|
3946
4170
|
auditHistory: history,
|
|
4171
|
+
pendingCompletion: undefined,
|
|
3947
4172
|
auditInfraStreak: infraStreak,
|
|
3948
4173
|
pauseKind: "error",
|
|
3949
4174
|
pauseReason: `auditor infrastructure failed ${infraStreak}× in a row — the auditor model is likely broken OR a verification command is hanging (ssh/sudo/long test runs stall the stream) (last: ${result.error.slice(0, 120)})`,
|
|
@@ -3963,6 +4188,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3963
4188
|
updateGoal({
|
|
3964
4189
|
status: "active",
|
|
3965
4190
|
auditHistory: history,
|
|
4191
|
+
pendingCompletion: undefined,
|
|
3966
4192
|
auditInfraStreak: infraStreak,
|
|
3967
4193
|
pauseReason: `auditor infrastructure${retriedOnce ? " (retried once)" : ""}: ${result.error}`,
|
|
3968
4194
|
pauseSuggestedAction: "Fix the auditor model (/glla model=provider/id) and call complete_goal again — your work was NOT judged",
|
|
@@ -3987,6 +4213,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
3987
4213
|
updateGoal({
|
|
3988
4214
|
status: "active",
|
|
3989
4215
|
auditHistory: history,
|
|
4216
|
+
pendingCompletion: undefined,
|
|
3990
4217
|
pauseReason: `regression shield: auditor approved, but evidence never referenced ${missing.length} contract item(s)`,
|
|
3991
4218
|
pauseSuggestedAction: "call complete_goal again — the next auditor run is told exactly which items to quote evidence for",
|
|
3992
4219
|
}, ctx);
|
|
@@ -4032,6 +4259,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4032
4259
|
updateGoal({
|
|
4033
4260
|
status: "active",
|
|
4034
4261
|
auditHistory: history,
|
|
4262
|
+
pendingCompletion: undefined,
|
|
4035
4263
|
pendingTasks,
|
|
4036
4264
|
pauseReason: `auditor disapproved ${trailingDisapprovals}× consecutively (cap ${auditCap}) — aggressiveMode: continuing with TODOs`,
|
|
4037
4265
|
}, ctx);
|
|
@@ -4052,6 +4280,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4052
4280
|
updateGoal({
|
|
4053
4281
|
status: "paused",
|
|
4054
4282
|
auditHistory: history,
|
|
4283
|
+
pendingCompletion: undefined,
|
|
4055
4284
|
pauseKind: "decision",
|
|
4056
4285
|
pauseOptions: ["Fix the disapproval gap, then continue (/goal resume)", "Tweak the objective — /goal tweak <new text>", "Cancel the goal (/goal cancel)"],
|
|
4057
4286
|
pauseRecommended: 1,
|
|
@@ -4073,6 +4302,7 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4073
4302
|
updateGoal({
|
|
4074
4303
|
status: "active",
|
|
4075
4304
|
auditHistory: history,
|
|
4305
|
+
pendingCompletion: undefined,
|
|
4076
4306
|
pauseReason: "auditor disapproved",
|
|
4077
4307
|
pauseSuggestedAction: "Inspect auditor feedback and fix the actual gap before calling complete_goal again",
|
|
4078
4308
|
}, ctx);
|
|
@@ -4102,6 +4332,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4102
4332
|
async execute(_id, params, _signal, _onUpdate, execCtx) {
|
|
4103
4333
|
const foreign1 = foreignToolGuard(execCtx);
|
|
4104
4334
|
if (foreign1) return { content: [{ type: "text", text: foreign1 }], details: {} };
|
|
4335
|
+
const ctx = currentToolContext(execCtx);
|
|
4336
|
+
if (!ctx) return staleToolResult();
|
|
4105
4337
|
const p = params as { reason: string; suggestedAction?: string; kind?: "decision" | "error" | "wait" | "blocked"; options?: string[]; recommended?: number; resumeAt?: string };
|
|
4106
4338
|
if (!state.goal) return { content: [{ type: "text", text: "No active goal." }], details: {} };
|
|
4107
4339
|
updateGoal({
|
|
@@ -4133,6 +4365,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4133
4365
|
async execute(_id, params, _signal, _onUpdate, execCtx) {
|
|
4134
4366
|
const foreign7 = foreignToolGuard(execCtx);
|
|
4135
4367
|
if (foreign7) return { content: [{ type: "text", text: foreign7 }], details: {} };
|
|
4368
|
+
const ctx = currentToolContext(execCtx);
|
|
4369
|
+
if (!ctx) return staleToolResult();
|
|
4136
4370
|
const p = params as { id: string };
|
|
4137
4371
|
if (!state.goal || !state.goal.taskList) {
|
|
4138
4372
|
return { content: [{ type: "text", text: "No task list in this goal." }], details: {} };
|
|
@@ -4163,6 +4397,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4163
4397
|
async execute(_id, params, _signal, _onUpdate, execCtx) {
|
|
4164
4398
|
const foreign8 = foreignToolGuard(execCtx);
|
|
4165
4399
|
if (foreign8) return { content: [{ type: "text", text: foreign8 }], details: {} };
|
|
4400
|
+
const ctx = currentToolContext(execCtx);
|
|
4401
|
+
if (!ctx) return staleToolResult();
|
|
4166
4402
|
const p = params as { id: string; status: "pending" | "in_progress" | "complete" };
|
|
4167
4403
|
if (!state.goal || !state.goal.taskList) {
|
|
4168
4404
|
return { content: [{ type: "text", text: "No task list in this goal." }], details: {} };
|
|
@@ -4195,13 +4431,14 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4195
4431
|
const foreign2 = foreignToolGuard(execCtx);
|
|
4196
4432
|
if (foreign2) return { content: [{ type: "text", text: foreign2 }], details: {} };
|
|
4197
4433
|
const p = params as { objective: string; verificationContract?: string; items?: string[] };
|
|
4434
|
+
const liveCtx = currentToolContext(execCtx);
|
|
4435
|
+
if (!liveCtx) return staleToolResult();
|
|
4198
4436
|
if (draftingTarget !== "goal" && draftingTarget !== "list") {
|
|
4199
4437
|
return {
|
|
4200
4438
|
content: [{ type: "text", text: "Not in goal drafting mode. The user starts drafting with /goal or /list add (no args), or activates directly with /goal <objective>." }],
|
|
4201
4439
|
details: {},
|
|
4202
4440
|
};
|
|
4203
4441
|
}
|
|
4204
|
-
const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
|
|
4205
4442
|
// v0.28.14: one-active-thing EARLY guard — refuse the whole interview
|
|
4206
4443
|
// when a loop is live (the post-confirm backstop below stays: state
|
|
4207
4444
|
// can change mid-interview).
|
|
@@ -4383,6 +4620,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4383
4620
|
const foreign3 = foreignToolGuard(execCtx);
|
|
4384
4621
|
if (foreign3) return { content: [{ type: "text", text: foreign3 }], details: {} };
|
|
4385
4622
|
const p = params as { target: string; measureCmd?: string; direction?: "min" | "max"; window?: number; max?: number; time?: number; tokens?: number; branch?: boolean };
|
|
4623
|
+
const liveCtx = currentToolContext(execCtx);
|
|
4624
|
+
if (!liveCtx) return staleToolResult();
|
|
4386
4625
|
if (draftingTarget !== "loop") {
|
|
4387
4626
|
return {
|
|
4388
4627
|
content: [{ type: "text", text: "You cannot start or draft a loop — only the user can, from the slash bar (the Confirm is the product). Do NOT write draft files or wait for the user to say 'start' in chat; that dead-ends. Instead hand the user the exact command: /loop start \"<target>\" (bare = infinite metricless; add measure=\"<cmd>\" direction=min|max for a metric loop), or /loop respec to reconcile against the root spec, or /loop with no args to draft interactively." }],
|
|
@@ -4408,7 +4647,6 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4408
4647
|
if (!metricless && p.direction !== "min" && p.direction !== "max") {
|
|
4409
4648
|
return { content: [{ type: "text", text: 'direction=min|max is required for a measured loop (omit measureCmd or pass "none" for a metricless spec loop).' }], details: {} };
|
|
4410
4649
|
}
|
|
4411
|
-
const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
|
|
4412
4650
|
// v0.28.14: one-active-thing — refuse to even test-run a loop measure
|
|
4413
4651
|
// while a goal/list-item is active (the /loop start COMMAND guards
|
|
4414
4652
|
// this; the tool path used to skip it and stack a loop over a goal).
|
|
@@ -4506,7 +4744,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4506
4744
|
const foreign4 = foreignToolGuard(execCtx);
|
|
4507
4745
|
if (foreign4) return { content: [{ type: "text", text: foreign4 }], details: {} };
|
|
4508
4746
|
const p = params as { target?: string; measureCmd?: string; specText?: string; specAppend?: string; rationale: string };
|
|
4509
|
-
const liveCtx = (execCtx
|
|
4747
|
+
const liveCtx = currentToolContext(execCtx);
|
|
4748
|
+
if (!liveCtx) return staleToolResult();
|
|
4510
4749
|
const loop = state.loop;
|
|
4511
4750
|
if (!loop?.active) {
|
|
4512
4751
|
return { content: [{ type: "text", text: "No active loop to refine. propose_loop_refine is only valid while a loop is running." }], details: {} };
|
|
@@ -4602,6 +4841,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4602
4841
|
const foreign5 = foreignToolGuard(execCtx);
|
|
4603
4842
|
if (foreign5) return { content: [{ type: "text", text: foreign5 }], details: {} };
|
|
4604
4843
|
const p = params as { items: string[] };
|
|
4844
|
+
const liveCtx = currentToolContext(execCtx);
|
|
4845
|
+
if (!liveCtx) return staleToolResult();
|
|
4605
4846
|
if (listMutationBlocked(draftingTarget)) {
|
|
4606
4847
|
return { content: [{ type: "text", text: LIST_DRAFTING_BLOCK_MESSAGE }], details: {} };
|
|
4607
4848
|
}
|
|
@@ -4609,7 +4850,6 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4609
4850
|
return { content: [{ type: "text", text: "No items given." }], details: {} };
|
|
4610
4851
|
}
|
|
4611
4852
|
const clean = p.items.map((t) => t.trim()).filter((t) => t.length > 0);
|
|
4612
|
-
const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
|
|
4613
4853
|
const wasIdle = !state.goal || state.goal.status === "complete" || state.goal.status === "aborted";
|
|
4614
4854
|
const n = enqueueItems(liveCtx, clean, "agent list_add");
|
|
4615
4855
|
return {
|
|
@@ -4635,6 +4875,8 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4635
4875
|
const foreign6 = foreignToolGuard(execCtx);
|
|
4636
4876
|
if (foreign6) return { content: [{ type: "text", text: foreign6 }], details: {} };
|
|
4637
4877
|
const p = params as { n: number };
|
|
4878
|
+
const liveCtx = currentToolContext(execCtx);
|
|
4879
|
+
if (!liveCtx) return staleToolResult();
|
|
4638
4880
|
if (listMutationBlocked(draftingTarget)) {
|
|
4639
4881
|
return { content: [{ type: "text", text: LIST_DRAFTING_BLOCK_MESSAGE }], details: {} };
|
|
4640
4882
|
}
|
|
@@ -4642,7 +4884,6 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4642
4884
|
if (!Number.isInteger(n) || n < 1) {
|
|
4643
4885
|
return { content: [{ type: "text", text: "n must be a positive integer (1-based position)." }], details: {} };
|
|
4644
4886
|
}
|
|
4645
|
-
const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
|
|
4646
4887
|
// v0.28.14: one-active-thing — a list item must not jump a live loop.
|
|
4647
4888
|
if (isLoopActive()) {
|
|
4648
4889
|
return { content: [{ type: "text", text: "A loop is active — one active thing at a time. The user must /loop stop it before a list item can activate." }], details: {} };
|
|
@@ -4704,11 +4945,12 @@ function registerAgentTools(pi: any, ctx: ExtensionContext): void {
|
|
|
4704
4945
|
return { content: [{ type: "text", text: "A task list already exists. Use update_task_status / complete_task to work it." }], details: {} };
|
|
4705
4946
|
}
|
|
4706
4947
|
const p = params as { tasks: TaskProposal[] };
|
|
4948
|
+
const liveCtx = currentToolContext(execCtx);
|
|
4949
|
+
if (!liveCtx) return staleToolResult();
|
|
4707
4950
|
const invalid = validateTaskProposal(p.tasks);
|
|
4708
4951
|
if (invalid) {
|
|
4709
4952
|
return { content: [{ type: "text", text: invalid }], details: {} };
|
|
4710
4953
|
}
|
|
4711
|
-
const liveCtx = (execCtx as ExtensionContext | undefined) ?? ctx;
|
|
4712
4954
|
const preview = p.tasks.map((t, i) => {
|
|
4713
4955
|
const subs = (t.subtasks ?? []).map((s, j) => ` ${i + 1}.${j + 1} ${s}`).join("\n");
|
|
4714
4956
|
return `${i + 1}. ${t.title}` + (subs ? `\n${subs}` : "");
|
|
@@ -5508,7 +5750,11 @@ async function cmdGllaWipe(ctx: ExtensionContext): Promise<void> {
|
|
|
5508
5750
|
if (loop) {
|
|
5509
5751
|
clearLoopTimer();
|
|
5510
5752
|
state.loop = undefined;
|
|
5753
|
+
const wipeGeneration = sessionGeneration;
|
|
5511
5754
|
await finishLoopGit(ctx, loop);
|
|
5755
|
+
const afterFinish = freshCtxForGeneration(wipeGeneration);
|
|
5756
|
+
if (!afterFinish) return;
|
|
5757
|
+
ctx = afterFinish;
|
|
5512
5758
|
appendLedger(ctx.cwd, "loop_stopped", { reason: "user wipe (/glla wipe)", iterations: loop.iteration, best: loop.bestValue });
|
|
5513
5759
|
}
|
|
5514
5760
|
persistState(ctx);
|
|
@@ -6182,7 +6428,9 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6182
6428
|
// Tool registration is lazy: done on the first session event, when a
|
|
6183
6429
|
// context exists. Tools show even without an active goal (and return
|
|
6184
6430
|
// "no active goal" if called).
|
|
6185
|
-
|
|
6431
|
+
// Tool definitions are re-registered at lifecycle boundaries; the current
|
|
6432
|
+
// invocation context is resolved inside each execute handler.
|
|
6433
|
+
let toolsRegistered = false;
|
|
6186
6434
|
|
|
6187
6435
|
// v0.24.5 tool-visibility self-heal: surface the notify exactly once
|
|
6188
6436
|
// per session so the user learns about an external allowlist once and
|
|
@@ -6234,6 +6482,9 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6234
6482
|
// 60s heartbeat notices. Re-arm it as soon as pi settles post-compact.
|
|
6235
6483
|
pi.on("session_compact", async (_event: any, ctx: ExtensionContext) => {
|
|
6236
6484
|
if (isForeignCtx(ctx)) return;
|
|
6485
|
+
// A late compact event can arrive after pi has already invalidated this
|
|
6486
|
+
// extension. It must not reclaim the old ctx or schedule settle refires.
|
|
6487
|
+
if (sessionHandoffPending || extensionApiStale || staleTerminalDone || zombieStoodDown) return;
|
|
6237
6488
|
rememberCtx(ctx);
|
|
6238
6489
|
if (!isSupervising()) return;
|
|
6239
6490
|
appendLedger(ctx.cwd, "session_compact", {});
|
|
@@ -6298,7 +6549,8 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6298
6549
|
|
|
6299
6550
|
// v0.15.1: ask_user_question answers arrive as tool results, not chat
|
|
6300
6551
|
// messages — count answered (non-cancelled) questionnaires as replies too.
|
|
6301
|
-
pi.on("tool_result", async (event: any) => {
|
|
6552
|
+
pi.on("tool_result", async (event: any, eventCtx: ExtensionContext) => {
|
|
6553
|
+
if (sessionHandoffPending || extensionApiStale || staleTerminalDone || zombieStoodDown) return;
|
|
6302
6554
|
noteToolResult(event); // v0.33.0: slim widget "last action" feed
|
|
6303
6555
|
// v0.24.0: roll loop tool-result fingerprints (same-tool-same-result
|
|
6304
6556
|
// detection) — recorded for ANY tool result while a loop is active.
|
|
@@ -6335,11 +6587,14 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6335
6587
|
// HIT QUOTA ERRORS section carries the full guidance.
|
|
6336
6588
|
if (isSubagentQuotaResult(String(event?.toolName ?? ""), Boolean(event?.isError ?? event?.error), event?.output ?? event?.result ?? event?.details ?? "")) {
|
|
6337
6589
|
const errText = typeof (event?.output ?? event?.result) === "string" ? (event?.output ?? event?.result) : JSON.stringify(event?.output ?? event?.result ?? event?.details ?? "");
|
|
6338
|
-
|
|
6339
|
-
|
|
6340
|
-
"
|
|
6341
|
-
|
|
6342
|
-
|
|
6590
|
+
const current = currentToolContext(eventCtx);
|
|
6591
|
+
if (current) {
|
|
6592
|
+
appendLedger(current.cwd, "subagent_quota_error", { error: String(errText).slice(0, 200) });
|
|
6593
|
+
current.ui.notify(
|
|
6594
|
+
"Subagent hit a quota error (403/limit). Repair: re-spawn with an explicit model= on your quota pool, or do the work inline — see the continuation prompt's WHEN SUBAGENTS HIT QUOTA ERRORS. Explore's upstream haiku pin is the usual cause (pi-subagents#175); glla's inherit-parent strategy removes it for NEW sessions.",
|
|
6595
|
+
"warning",
|
|
6596
|
+
);
|
|
6597
|
+
}
|
|
6343
6598
|
}
|
|
6344
6599
|
if (draftingTarget === null) return;
|
|
6345
6600
|
if (askUserQuestionAnswered(String(event?.toolName ?? ""), event?.details)) {
|
|
@@ -6360,7 +6615,7 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6360
6615
|
writeSessionHandoff(ctx, shutdownReason);
|
|
6361
6616
|
sessionReplacementUntil = Date.now() + SESSION_REBIND_GRACE_MS;
|
|
6362
6617
|
clearSessionOwnedTimers();
|
|
6363
|
-
|
|
6618
|
+
toolsRegistered = false;
|
|
6364
6619
|
toolHealNotified = false;
|
|
6365
6620
|
});
|
|
6366
6621
|
|
|
@@ -6371,6 +6626,21 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6371
6626
|
if (isForeignCtx(ctx)) return;
|
|
6372
6627
|
extensionApi = pi;
|
|
6373
6628
|
sessionHandoffPending = false;
|
|
6629
|
+
// Reset terminal ownership before rememberCtx: this is the only event
|
|
6630
|
+
// allowed to bind a context after a stale/shutdown handoff.
|
|
6631
|
+
staleTerminalDone = false; // v0.33.1: a rebound session can go terminal again
|
|
6632
|
+
zombieStoodDown = false;
|
|
6633
|
+
sessionGeneration++;
|
|
6634
|
+
// An auditor belonging to the disposed generation cannot block the fresh
|
|
6635
|
+
// session's recovery gate; its finally block is generation-guarded too.
|
|
6636
|
+
completionAuditInFlight = false;
|
|
6637
|
+
completionAuditGeneration = null;
|
|
6638
|
+
latestAuditProgress = null;
|
|
6639
|
+
// Ephemeral watchdog counters belong to the old session, not the
|
|
6640
|
+
// persisted goal. Reset them so a stale boundary cannot make the next
|
|
6641
|
+
// fresh session inherit a false stall count.
|
|
6642
|
+
heartbeatNudges = 0;
|
|
6643
|
+
consecutiveStalls = 0;
|
|
6374
6644
|
const startReason = typeof event?.reason === "string" ? event.reason : "unknown";
|
|
6375
6645
|
initialSessionLoadPending = isBlankInitialStartup(ctx, startReason);
|
|
6376
6646
|
rememberCtx(ctx);
|
|
@@ -6384,8 +6654,6 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6384
6654
|
// confirm the new handle actually works.
|
|
6385
6655
|
writeOwnerFile(ctx.cwd);
|
|
6386
6656
|
sessionReplacementUntil = 0;
|
|
6387
|
-
zombieStoodDown = false;
|
|
6388
|
-
staleTerminalDone = false; // v0.33.1: a rebound session must be able to go terminal AGAIN (was: one-shot for the process lifetime)
|
|
6389
6657
|
postCompactResumeOwed = false; // v0.33.1: a compact from a previous session must not resync THIS one
|
|
6390
6658
|
postCompactResyncPending = false;
|
|
6391
6659
|
appendLedger(ctx.cwd, "session_rebound", { reason: startReason });
|
|
@@ -6409,9 +6677,9 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6409
6677
|
heldLoop: state.loop && (state.loop.active || state.loop.stopReason === HELD_ON_RESTORE) ? state.loop.target.slice(0, 60) : undefined,
|
|
6410
6678
|
};
|
|
6411
6679
|
carryoverResolved = !(carryoverSnapshot.pausedGoal || carryoverSnapshot.listCount > 0 || carryoverSnapshot.heldLoop);
|
|
6412
|
-
if (!
|
|
6413
|
-
registerAgentTools(pi
|
|
6414
|
-
|
|
6680
|
+
if (!toolsRegistered) {
|
|
6681
|
+
registerAgentTools(pi);
|
|
6682
|
+
toolsRegistered = true;
|
|
6415
6683
|
}
|
|
6416
6684
|
ensureAgentToolsActive(pi, ctx);
|
|
6417
6685
|
warnOnCommandCollision(ctx);
|
|
@@ -6605,6 +6873,10 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6605
6873
|
});
|
|
6606
6874
|
|
|
6607
6875
|
pi.on("agent_end", async (event: any, ctx: ExtensionContext) => {
|
|
6876
|
+
// A late agent_end from the disposed session is not a fresh turn. Do not
|
|
6877
|
+
// account it, run length continuation, or schedule another send after a
|
|
6878
|
+
// stale terminal/handoff.
|
|
6879
|
+
if (sessionHandoffPending || extensionApiStale || staleTerminalDone || zombieStoodDown) return;
|
|
6608
6880
|
rememberCtx(ctx);
|
|
6609
6881
|
// v0.23.8: a subagent finishing must not drive the main session's
|
|
6610
6882
|
// continuation loop.
|
|
@@ -6625,7 +6897,30 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6625
6897
|
const rawPriorA = assistants.length >= 2 ? assistants[assistants.length - 2] : null;
|
|
6626
6898
|
const extractText = (m: any): string => (m && Array.isArray(m.content)) ? m.content.filter((p: any) => p.type === "text").map((p: any) => p.text).join("\n") : "";
|
|
6627
6899
|
const lastA = rawLastA ? { stopReason: rawLastA.stopReason, text: extractText(rawLastA), priorText: extractText(rawPriorA) } : null;
|
|
6628
|
-
|
|
6900
|
+
// v0.34.19: pi-ai clamps max_tokens to the remaining context before the
|
|
6901
|
+
// provider call. At ~99% context that clamp can be 1 token, which the
|
|
6902
|
+
// provider reports as stopReason "length" — but this is NOT an overlong
|
|
6903
|
+
// assistant response. Extension agent_end runs BEFORE pi's own
|
|
6904
|
+
// auto-compaction check (agent-session.js _handlePostAgentRun), so sending
|
|
6905
|
+
// LENGTH_CONTINUE_TEXT here queues another 1-token request and delays the
|
|
6906
|
+
// real cure. Defer to pi compaction; session_compact's resume debt owns
|
|
6907
|
+
// the next continuation. Older pi/test doubles without getContextUsage()
|
|
6908
|
+
// fail open to the legacy true-length path.
|
|
6909
|
+
const contextUsage = (() => {
|
|
6910
|
+
try { return typeof ctx.getContextUsage === "function" ? ctx.getContextUsage() : undefined; } catch { return undefined; }
|
|
6911
|
+
})();
|
|
6912
|
+
const contextStarvedLength = isContextStarvedLengthStop(rawLastA, contextUsage);
|
|
6913
|
+
const lc = tickLengthContinue(lastA?.stopReason === "length" && !contextStarvedLength);
|
|
6914
|
+
if (contextStarvedLength) {
|
|
6915
|
+
appendLedger(ctx.cwd, "length_continue_deferred_context_full", {
|
|
6916
|
+
outputTokens: rawLastA?.usage?.output,
|
|
6917
|
+
contextTokens: contextUsage?.tokens ?? null,
|
|
6918
|
+
contextWindow: contextUsage?.contextWindow ?? null,
|
|
6919
|
+
contextPercent: contextUsage?.percent ?? null,
|
|
6920
|
+
});
|
|
6921
|
+
ctx.ui.notify("glla: output-token stop was context starvation (tiny output at a nearly full context) — yielding to pi auto-compaction instead of re-sending.", "info");
|
|
6922
|
+
return;
|
|
6923
|
+
}
|
|
6629
6924
|
if (lc.giveUpNow) {
|
|
6630
6925
|
ctx.ui.notify(`glla: response hit the output-token cap ${LENGTH_CONTINUE_MAX}× in a row — stepping aside. Ask the model to split the work into smaller pieces.`, "warning");
|
|
6631
6926
|
notifyExternal(ctx, "Response truncated 3× in a row — giving up auto-continue.");
|
|
@@ -6640,9 +6935,9 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6640
6935
|
t.turns++;
|
|
6641
6936
|
state.goal.telemetry = t;
|
|
6642
6937
|
}
|
|
6643
|
-
if (!
|
|
6644
|
-
registerAgentTools(pi
|
|
6645
|
-
|
|
6938
|
+
if (!toolsRegistered) {
|
|
6939
|
+
registerAgentTools(pi);
|
|
6940
|
+
toolsRegistered = true;
|
|
6646
6941
|
}
|
|
6647
6942
|
ensureAgentToolsActive(pi, ctx);
|
|
6648
6943
|
// v0.27.3: nudge accounting — substantive analytical turns (long, novel
|
|
@@ -6830,16 +7125,16 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6830
7125
|
notifyExternal(ctx, `${goalNoun()} parked: provider erroring across 6 error-brake cycles — hourly top-of-hour probes scheduled.`);
|
|
6831
7126
|
appendLedger(ctx.cwd, "error_brake_capped", { streak: brakeStreak, reason });
|
|
6832
7127
|
const probeMs = msUntilNextHourBoundary(Date.now());
|
|
6833
|
-
|
|
7128
|
+
scheduleQuotaRetryForSession(ctx, probeMs / 1000, reason, (fresh) => {
|
|
6834
7129
|
// Re-check: only probe if STILL parked by the error-brake cap —
|
|
6835
7130
|
// a user pause/resume/cancel meanwhile is never stomped.
|
|
6836
7131
|
if (state.goal && state.goal.status === "paused" && state.goal.pauseKind === "error"
|
|
6837
7132
|
&& (state.goal.pauseReason ?? "").includes("error-brakes in a row")) {
|
|
6838
|
-
appendLedger(
|
|
6839
|
-
updateGoal({ status: "active" },
|
|
6840
|
-
appendLedger(
|
|
6841
|
-
|
|
6842
|
-
scheduleContinuation(
|
|
7133
|
+
appendLedger(fresh.cwd, "hourly_rate_probe", { goalId: state.goal.id, streak: state.goal.errorBrakeStreak ?? 0 });
|
|
7134
|
+
updateGoal({ status: "active" }, fresh);
|
|
7135
|
+
appendLedger(fresh.cwd, "goal_resumed", { via: "hourly-rate-probe" });
|
|
7136
|
+
fresh.ui.notify("Hourly probe: resuming (rate-limit windows typically expire at the top of the hour).", "info");
|
|
7137
|
+
scheduleContinuation(fresh, true);
|
|
6843
7138
|
}
|
|
6844
7139
|
}, "Hourly rate-limit probe");
|
|
6845
7140
|
return;
|
|
@@ -6861,14 +7156,14 @@ export default function (pi: ExtensionAPI): void {
|
|
|
6861
7156
|
ctx.ui.notify(`Goal paused: ${reason}.${quotaWall ? " Quota/rate-limit wall — resuming won't help until the window resets; switch /model to continue now." : ""}`, "warning");
|
|
6862
7157
|
notifyExternal(ctx, `Goal paused: ${reason}.`);
|
|
6863
7158
|
appendLedger(ctx.cwd, "goal_paused", { reason });
|
|
6864
|
-
|
|
7159
|
+
scheduleQuotaRetryForSession(ctx, cooldownMs / 1000, reason, (fresh) => {
|
|
6865
7160
|
// Re-check: only auto-resume if STILL paused for the error brake
|
|
6866
7161
|
// (a user /goal pause during the window is not stomped).
|
|
6867
7162
|
if (state.goal && state.goal.status === "paused" && (state.goal.pauseReason ?? "").startsWith("5 consecutive errors")) {
|
|
6868
|
-
updateGoal({ status: "active" },
|
|
6869
|
-
appendLedger(
|
|
6870
|
-
|
|
6871
|
-
scheduleContinuation(
|
|
7163
|
+
updateGoal({ status: "active" }, fresh);
|
|
7164
|
+
appendLedger(fresh.cwd, "goal_resumed", { via: "error-brake-retry" });
|
|
7165
|
+
fresh.ui.notify("Auto-resumed after the 5-error brake (cooldown elapsed).", "info");
|
|
7166
|
+
scheduleContinuation(fresh, true);
|
|
6872
7167
|
}
|
|
6873
7168
|
}, "5 consecutive errors — auto-retry");
|
|
6874
7169
|
return;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-goal-list-loop-audit",
|
|
3
|
-
"version": "0.34.
|
|
3
|
+
"version": "0.34.20",
|
|
4
4
|
"description": "Mission control for autonomous pi: interview-drafted goals, an audited task queue, and forever-loops (metric, spec, project-audit) that run for hours. An isolated extension-less auditor re-verifies every completion with raw evidence; confirmed drafts, decision pauses and consent gates keep you in charge.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "dracon",
|