pi-goal-list-loop-audit 0.35.4 → 0.35.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +7666 -0
- package/INSTALL.md +377 -0
- package/README.md +26 -13
- package/docs/DESIGN.md +31 -4
- package/docs/INDEX.md +32 -7
- package/examples/example-objective.md +119 -0
- package/extensions/faulty-objective-recovery.ts +13 -2
- package/extensions/goal-commands.ts +50 -6
- package/extensions/goal-heartbeat.ts +25 -21
- package/extensions/goal-loop-auditor-process.ts +184 -3
- package/extensions/goal-loop-core.ts +108 -15
- package/extensions/goal-loop-display.ts +82 -0
- package/extensions/goal-loop-shield.ts +55 -0
- package/extensions/goal-recovery.ts +317 -7
- package/extensions/goal-settings.ts +31 -1
- package/extensions/loops/goal-activation.ts +74 -7
- package/extensions/loops/goal-auditor-hooks.ts +30 -55
- package/extensions/loops/goal-list-queue.ts +18 -5
- package/extensions/loops/goal-orchestrator.ts +5 -0
- package/extensions/loops/goal-runtime-globals.ts +2 -0
- package/extensions/loops/goal-session.ts +101 -21
- package/extensions/loops/goal-settings-ui.ts +122 -63
- package/extensions/loops/goal-tools.ts +110 -36
- package/extensions/loops/goal-ui.ts +59 -2
- package/extensions/main-model-recovery.ts +24 -1
- package/extensions/model-picker.ts +3 -1
- package/extensions/model-selector.ts +2 -1
- package/extensions/multi-model-picker.ts +64 -7
- package/extensions/settings-menu.ts +16 -0
- package/package.json +4 -1
- package/prompts/goal-loop-continuation.md +11 -8
- package/prompts/goal-loop-draft.md +5 -5
- package/schemas/goal.schema.json +17 -0
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
# Examples: the three loops
|
|
2
|
+
|
|
3
|
+
Worked examples for `pi-goal-list-loop-audit`. State lives at `.pi-glla/`
|
|
4
|
+
(ledger `active.jsonl`, goal markdown in `goals/`, finished goals in `archive/`).
|
|
5
|
+
|
|
6
|
+
## Loop 1: `/goal` — single goal with isolated auditor
|
|
7
|
+
|
|
8
|
+
Direct (skip drafting):
|
|
9
|
+
|
|
10
|
+
```
|
|
11
|
+
/goal start Step 1. Add a /healthz endpoint to server.ts returning {status:'ok'}.
|
|
12
|
+
Step 2. Add a vitest test that hits it.
|
|
13
|
+
Done when:
|
|
14
|
+
- curl -fsS localhost:3000/healthz returns 200 with body {"status":"ok"}
|
|
15
|
+
- npm test exits 0 with 0 failures
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
Or drafting (recommended when the idea is still fuzzy):
|
|
19
|
+
|
|
20
|
+
```
|
|
21
|
+
/goal
|
|
22
|
+
# the agent grills one focused question at a time, then a Confirm dialog
|
|
23
|
+
# shows the objective + Done-when contract. Nothing starts before you confirm.
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
What happens next:
|
|
27
|
+
|
|
28
|
+
1. The agent works the objective turn by turn (`agent_end`-driven continuation).
|
|
29
|
+
Guards: a stall watchdog (3 consecutive turns with no tool calls), a
|
|
30
|
+
5-consecutive-errors pause, and the optional token guard — no wall-clock
|
|
31
|
+
cap; a goal ends via completion, pause/cancel, or a guard.
|
|
32
|
+
2. It calls `complete_goal` when it believes it is done.
|
|
33
|
+
3. The **isolated auditor** — a fresh pi session with no extensions and only
|
|
34
|
+
read tools — inspects the repo. Because this goal has a `Done when:`
|
|
35
|
+
contract, the auditor MUST quote raw command output per contract item in an
|
|
36
|
+
`<evidence>` block (regression_shield). An approval without complete
|
|
37
|
+
evidence is automatically converted to a disapproval.
|
|
38
|
+
4. Approved → goal archived to `.pi-glla/archive/<id>.md`. Disapproved → the
|
|
39
|
+
loop continues with the auditor's feedback.
|
|
40
|
+
|
|
41
|
+
Useful commands: `/goal status`, `/goal pause`, `/goal resume`, `/goal cancel`,
|
|
42
|
+
`/glla` (auditor model, thinking level, token limit, notify command, autoresume).
|
|
43
|
+
|
|
44
|
+
## Loop 2: `/list` — the shopping list of goals
|
|
45
|
+
|
|
46
|
+
```
|
|
47
|
+
/list Create one.txt containing one. Done when: grep -q one one.txt
|
|
48
|
+
/list Create two.txt containing two. Done when: grep -q two two.txt
|
|
49
|
+
/list # show active + waiting items
|
|
50
|
+
/list next # skip current item
|
|
51
|
+
/list remove 2 # drop item 2
|
|
52
|
+
/list clear # empty the list
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Each item is a full goal with its own contract and audit. When one completes
|
|
56
|
+
(or is aborted), the next activates automatically. Order is the default, not
|
|
57
|
+
the law — `/list next 3` activates item 3 directly. On session restore the
|
|
58
|
+
list HOLDS in a fresh session (nothing auto-starts); it auto-activates only
|
|
59
|
+
when Auto-resume is enabled in `/glla` settings (global-only since v0.29.5).
|
|
60
|
+
|
|
61
|
+
## Loop 3: `/loop` — metric-driven forever loop
|
|
62
|
+
|
|
63
|
+
```
|
|
64
|
+
/loop start "reduce TODO comments in src" measure="grep -rc TODO src | cut -d: -f2 | paste -sd+ | bc" direction=min window=5 max=50
|
|
65
|
+
/loop status # iteration, best, stall, recent values
|
|
66
|
+
/loop stop # halt with summary
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
The **orchestrator** runs your `measure` command after every agent turn — the
|
|
70
|
+
agent never self-reports a number. The loop stops on plateau (`window`
|
|
71
|
+
non-improving iterations), the `max` cap, a time/token bound, or `/loop stop`.
|
|
72
|
+
There is no completion check in loop 3 — "improve until X" is a GOAL; a loop
|
|
73
|
+
is a process. No auditor either: the metric is the verdict.
|
|
74
|
+
|
|
75
|
+
With `branch=1`, all work happens on a scratch branch
|
|
76
|
+
(`pi-glla-loop/<timestamp>-<slug>`): each improvement is committed, each
|
|
77
|
+
regression is hard-reset (scratch branch only), and on stop you return to your
|
|
78
|
+
original branch with merge instructions. Requires a clean working tree.
|
|
79
|
+
|
|
80
|
+
## Notifications (optional)
|
|
81
|
+
|
|
82
|
+
Run `/glla` and set the **Notify command** row (the event message is passed
|
|
83
|
+
as `$1`); `off` silences pushes, empty re-enables auto-detection. `/glla`
|
|
84
|
+
has an action namespace (`status`, `log`, `stats`, `audits`, `wipe`, …) —
|
|
85
|
+
it does NOT accept inline `key=value` assignments. Headless/scripted setups
|
|
86
|
+
can write the project settings file directly instead:
|
|
87
|
+
|
|
88
|
+
```
|
|
89
|
+
# .pi-glla/settings.json
|
|
90
|
+
{ "notifyCmd": "echo \"$1\" >> ~/goal-events.log" }
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Fires on goal complete, goal pause, and loop stop; message as `$1`.
|
|
94
|
+
|
|
95
|
+
## Token guard (opt-in)
|
|
96
|
+
|
|
97
|
+
Off by default. Set a per-goal budget and crossing it pauses the goal:
|
|
98
|
+
|
|
99
|
+
```
|
|
100
|
+
# /glla → Token limit row, or .pi-glla/settings.json
|
|
101
|
+
{ "tokenLimit": 2000000 }
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
## The auditor model rule
|
|
105
|
+
|
|
106
|
+
The detached auditor uses an explicit bounded cascade: an optional primary
|
|
107
|
+
pin, an optional fallback pin, then your pi session model. A model that fails
|
|
108
|
+
at runtime is retried once and the next detached candidate is tried; the
|
|
109
|
+
plugin never falls back into the parent in-process session.
|
|
110
|
+
|
|
111
|
+
```
|
|
112
|
+
# /glla → Auditor model row (and Auditor fallback), or .pi-glla/settings.json
|
|
113
|
+
{ "auditorModel": "provider/model-id" }
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
If every candidate errors with auth/provider failures, the stored completion
|
|
117
|
+
claim pauses and `/goal resume` retries it after you fix the model — the
|
|
118
|
+
auditor session has no extensions, so
|
|
119
|
+
providers that only exist via an extension may be unavailable to it.
|
|
@@ -35,10 +35,21 @@ export interface ObjectiveRepairProposal {
|
|
|
35
35
|
confidence: "best-effort";
|
|
36
36
|
}
|
|
37
37
|
|
|
38
|
-
|
|
38
|
+
// A verb that plausibly opens a real, actionable objective. The reviewer/
|
|
39
|
+
// verification momentum lives in the vocabulary and narrative regexes below;
|
|
40
|
+
// this list is the escape hatch that keeps genuine imperatives (including
|
|
41
|
+
// ones that mention the auditor/reviewer machinery legitimately, e.g.
|
|
42
|
+
// "Show the selected models … (main/auditor/drafter) on the goal card")
|
|
43
|
+
// from being misread as report text. Keep it broad: an objective that starts
|
|
44
|
+
// with any of these is actionable on its face.
|
|
45
|
+
const IMPERATIVE_START = /^(add|allow|audit|benchmark|build|cap|check|clarify|close|collapse|collect|compare|consolidate|create|describe|detect|diagnose|display|document|ensure|enforce|explain|expose|fix|harden|improve|implement|inspect|instrument|investigate|list|make|measure|migrate|monitor|open|optimize|overhaul|plan|preserve|prevent|print|profile|publish|read|recover|refactor|remove|render|repair|replace|research|resolve|restore|review|ship|show|simplify|strengthen|summarize|support|surface|test|trace|update|validate|verify|watch|wire|write)\b/i;
|
|
39
46
|
const COMMAND_ONLY = /^(?:bun|npm|pnpm|yarn|npx|node|deno|git)\s+(?:test|run|check|diff|status|show|log|exec)\b/i;
|
|
40
47
|
const REVIEWER_MARKER = /(?:^|[.!?;]\s+|[-*]\s+)(?:audit(?:\s+(?:report|result|findings?))?|review(?:er)?(?:\s+(?:report|result|feedback|findings?))?|verdict|evidence|output|item|required\s+fixes?|completion\s+claim)\s*:/i;
|
|
41
|
-
|
|
48
|
+
// Report vocabulary. "auditor"/"reviewer" count ONLY when tied to a verdict
|
|
49
|
+
// shape ("auditor approved", "reviewer finding") — a bare mention of the
|
|
50
|
+
// role ("(main/auditor/drafter)", "auditor workers") is legitimate task
|
|
51
|
+
// vocabulary and must not trip verification-fragment on its own.
|
|
52
|
+
const REVIEWER_VOCABULARY = /\b(?:passes\s+sequentially|zero\s+failures?|\d+\s+failures?|ran\s+\d+\s+tests?|verification\s+contract|regression\s+shield|auditor[- ]+(?:approved|report|disapproved)|reviewer[- ]+(?:approved|report|disapproved|feedback|finding)|completion\s+claim|<\/?(?:evidence|approved|disapproved|impossible)\b)\b/i;
|
|
42
53
|
// A report can look superficially task-like because it lists completed work
|
|
43
54
|
// with imperative-shaped nouns ("Added ...", "Focused tests ..."). Keep
|
|
44
55
|
// genuine phrases such as "Add focused tests for the audit path" valid, but
|
|
@@ -14,7 +14,7 @@ import {
|
|
|
14
14
|
DEFAULT_TOKEN_LIMIT, Goal, ListItem, Status, appendLedger, archiveDir, archivedGoalPath, bumpGoalRevision, sanitizeProviderDisplayText,
|
|
15
15
|
computeListDepth, clearQueueItemFiles, deleteQueueItemFile, extractVerificationContract, formatAuditLog, formatGoalAuditHistory, formatMainModelRecoveryStatus, queueItemSidecarCount,
|
|
16
16
|
formatListDepth, goalArgsNeedDrafting, ledgerPath, newGoalId, nowIso, parseListImport, parseListItemDeclaration,
|
|
17
|
-
readAuditLog, readQueueFromDisk, routeGoalArgs, routeListText, sanitizeDisplayText, sanitizeProviderAuditReport, statusLabel,
|
|
17
|
+
assignQueueOrder, compareQueueItems, readAuditLog, readQueueFromDisk, routeGoalArgs, routeListText, sanitizeDisplayText, sanitizeProviderAuditReport, statusLabel,
|
|
18
18
|
writeQueueItemFile, type ModeCommand, type State, LIST_MUTATING_SUBCOMMANDS, SETTINGS_MUTATING_ACTIONS,
|
|
19
19
|
} from "./goal-loop-core.js";
|
|
20
20
|
import { clearDispatchRecord, dispatchRecordExists } from "./goal-loop-dispatch.js";
|
|
@@ -341,6 +341,13 @@ async function cmdResume(ctx: ExtensionContext): Promise<void> {
|
|
|
341
341
|
void probeMainModelRecovery(ctx);
|
|
342
342
|
return;
|
|
343
343
|
}
|
|
344
|
+
if (state.mainModelRecovery?.primaryProbeAt || state.mainModelRecovery?.primaryProbeInFlight) {
|
|
345
|
+
clearMainModelRecoveryTimer();
|
|
346
|
+
flags.continuationDispatchStoodDown = false;
|
|
347
|
+
ctx.ui.notify("Probing the preferred primary now — the current fallback remains available if the primary is still unhealthy.", "info");
|
|
348
|
+
void probeMainModelRecovery(ctx);
|
|
349
|
+
return;
|
|
350
|
+
}
|
|
344
351
|
// v0.34.3: /goal resume on an ACTIVE-but-idle goal re-kicks its
|
|
345
352
|
// continuation (was: silent return — the user got NOTHING while the
|
|
346
353
|
// widget said "active"). One-active-thing still holds: an active loop
|
|
@@ -836,7 +843,23 @@ function recentlyCompletedObjectives(cwd: string): Set<string> {
|
|
|
836
843
|
return done;
|
|
837
844
|
}
|
|
838
845
|
|
|
846
|
+
export function hydrateListQueueFromDisk(ctx: ExtensionContext): number {
|
|
847
|
+
const memory = listQueue();
|
|
848
|
+
const exclude = new Set<string>();
|
|
849
|
+
if (state.goal?.id) exclude.add(state.goal.id);
|
|
850
|
+
const disk = readQueueFromDisk(ctx.cwd, exclude);
|
|
851
|
+
const known = new Set(memory.map((item) => item.id));
|
|
852
|
+
const recovered = disk.filter((item) => !known.has(item.id));
|
|
853
|
+
if (recovered.length === 0) return 0;
|
|
854
|
+
const merged = [...memory, ...recovered].sort(compareQueueItems);
|
|
855
|
+
replaceState({ ...state, list: merged });
|
|
856
|
+
persistState(ctx);
|
|
857
|
+
appendLedger(ctx.cwd, "list_recovered_from_disk", { count: recovered.length, hydrated: true });
|
|
858
|
+
return recovered.length;
|
|
859
|
+
}
|
|
860
|
+
|
|
839
861
|
function enqueueItems(ctx: ExtensionContext, texts: string[], source: string, opts?: { autoActivate?: boolean }): number {
|
|
862
|
+
hydrateListQueueFromDisk(ctx);
|
|
840
863
|
const recentlyDone = recentlyCompletedObjectives(ctx.cwd);
|
|
841
864
|
const fresh = texts.filter((t) => !recentlyDone.has(normalizeObjective(parseListItemDeclaration(t).objective)));
|
|
842
865
|
const skipped = texts.length - fresh.length;
|
|
@@ -899,13 +922,23 @@ function enqueueItems(ctx: ExtensionContext, texts: string[], source: string, op
|
|
|
899
922
|
ctx.ui.notify(`Refused ${refused.length} subtask item(s): ${refused.join(" | ")}.`, "warning");
|
|
900
923
|
}
|
|
901
924
|
if (resolved.length === 0) return 0;
|
|
902
|
-
const itemsToWrite = resolved;
|
|
925
|
+
const itemsToWrite = assignQueueOrder(resolved, listQueue());
|
|
903
926
|
// v0.34.60: disk-first write order. Each item lands on disk BEFORE any
|
|
904
927
|
// in-memory state mutation, so /list survives a stale extension handle
|
|
905
928
|
// (e.g. /reload, plugin re-init, RAM-only state loss). The
|
|
906
929
|
// .queue.json sidecar is atomic (temp + rename) and idempotent (skips
|
|
907
|
-
// existing files rather than overwriting).
|
|
930
|
+
// existing files rather than overwriting). A failed member aborts the
|
|
931
|
+
// batch and removes only sidecars written by this attempt.
|
|
908
932
|
const written = itemsToWrite.map((item) => writeQueueItemFile(ctx.cwd, item));
|
|
933
|
+
const failedWrite = written.find((result) => result.failed);
|
|
934
|
+
if (failedWrite) {
|
|
935
|
+
written.forEach((result, index) => {
|
|
936
|
+
if (result.wrote) deleteQueueItemFile(ctx.cwd, itemsToWrite[index]!.id);
|
|
937
|
+
});
|
|
938
|
+
appendLedger(ctx.cwd, "list_queue_disk_write_failed", { source, count: itemsToWrite.length, path: failedWrite.path });
|
|
939
|
+
ctx.ui.notify(`Could not persist the queued item(s) from ${source}; no in-memory queue mutation was applied. Fix disk access and retry.`, "warning");
|
|
940
|
+
return 0;
|
|
941
|
+
}
|
|
909
942
|
replaceState({ ...state, list: [...listQueue(), ...itemsToWrite] });
|
|
910
943
|
const diskFirst = written.filter((w) => w.wrote).length === itemsToWrite.length;
|
|
911
944
|
appendLedger(ctx.cwd, "list_queue_disk_first", { source, count: itemsToWrite.length, diskFirst });
|
|
@@ -991,6 +1024,10 @@ async function cmdList(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
991
1024
|
// resume/tweak/cancel gates branch on state.goal.policy — the gate used
|
|
992
1025
|
// to silently refuse the whole surface until a restart.
|
|
993
1026
|
healGoalPolicy(ctx);
|
|
1027
|
+
// v0.35.4 audit: sidecars are durable queue state, not a display-only
|
|
1028
|
+
// fallback. Hydrate orphaned disk items before any list surface can count,
|
|
1029
|
+
// remove, cancel, or activate them.
|
|
1030
|
+
if (!staleEntry) hydrateListQueueFromDisk(ctx);
|
|
994
1031
|
|
|
995
1032
|
if (sub === "audit") {
|
|
996
1033
|
// v0.31.0: /list audit [focus] — collect-then-drain (user design
|
|
@@ -1332,19 +1369,24 @@ async function cmdList(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
1332
1369
|
|
|
1333
1370
|
/** Append one objective to the list; activate immediately when idle. */
|
|
1334
1371
|
function addSingleItem(ctx: ExtensionContext, raw: string): void {
|
|
1372
|
+
hydrateListQueueFromDisk(ctx);
|
|
1335
1373
|
const extracted = parseListItemDeclaration(raw);
|
|
1336
|
-
const item = {
|
|
1374
|
+
const item = assignQueueOrder([{
|
|
1337
1375
|
id: newGoalId(),
|
|
1338
1376
|
objective: extracted.objective,
|
|
1339
1377
|
...(extracted.agentRole ? { agentRole: extracted.agentRole } : {}),
|
|
1340
1378
|
verificationContract: extracted.verificationContract || undefined,
|
|
1341
1379
|
...(extracted.parallelSafe === undefined ? {} : { parallelSafe: extracted.parallelSafe }),
|
|
1342
1380
|
addedAt: nowIso(),
|
|
1343
|
-
}
|
|
1381
|
+
}], listQueue())[0]!;
|
|
1344
1382
|
// v0.34.61: disk-first — write the sidecar BEFORE mutating state so the
|
|
1345
1383
|
// item survives an orchestrator-turn death between state mutation and
|
|
1346
1384
|
// persistState (the original bug for /list add "<direct text>").
|
|
1347
|
-
writeQueueItemFile(ctx.cwd, item);
|
|
1385
|
+
const written = writeQueueItemFile(ctx.cwd, item);
|
|
1386
|
+
if (written.failed) {
|
|
1387
|
+
ctx.ui.notify("Could not persist the list item; no in-memory queue mutation was applied. Fix disk access and retry.", "warning");
|
|
1388
|
+
return;
|
|
1389
|
+
}
|
|
1348
1390
|
replaceState({ ...state, list: [...listQueue(), item] });
|
|
1349
1391
|
persistState(ctx);
|
|
1350
1392
|
appendLedger(ctx.cwd, "list_added", { id: item.id, objective: item.objective });
|
|
@@ -2126,6 +2168,8 @@ async function cmdSettings(args: string, ctx: ExtensionContext): Promise<void> {
|
|
|
2126
2168
|
[
|
|
2127
2169
|
`mainAgentFallbackModels: ${formatMainModelFallbacks(effectiveSettings.mainModelFallbacks)} [${prov.mainModelFallbacks?.source ?? "default"}]`,
|
|
2128
2170
|
fmt("mainModelRetryMinutes", "mainModelRetryMinutes (base minutes; doubles per attempt)"),
|
|
2171
|
+
fmt("mainModelFailback", "mainModelFailback (auto/sticky)"),
|
|
2172
|
+
fmt("mainModelPrimaryProbeMinutes", "mainModelPrimaryProbeMinutes"),
|
|
2129
2173
|
fmt("drafterModel", "drafterAgent (drafting only)"),
|
|
2130
2174
|
fmt("drafterThinkingLevel", "drafterThinking (drafting only)"),
|
|
2131
2175
|
fmt("drafterModelFallbacks", "drafterFallbackAgents (drafting only)"),
|
|
@@ -378,16 +378,24 @@ function heartbeatTick(): void {
|
|
|
378
378
|
// that disposed handle as a host loss only creates the repeated
|
|
379
379
|
// "session invalidated" warning after the work is already safe on disk.
|
|
380
380
|
// Keep the probe for active goals/loops, detached completion audits,
|
|
381
|
-
// recoverable stale debt,
|
|
382
|
-
// can strand live work. A normal
|
|
383
|
-
// work; stale interruption debt
|
|
384
|
-
//
|
|
385
|
-
//
|
|
386
|
-
//
|
|
381
|
+
// recoverable stale debt, parked completion-audit recovery, and LIVE
|
|
382
|
+
// tracked subagents, where a dead handle can strand live work. A normal
|
|
383
|
+
// user pause does not count as host-bound work; stale interruption debt and
|
|
384
|
+
// parked completion claims do, because same-process self-heal still needs a
|
|
385
|
+
// heartbeat opportunity. Ended subagent probes remain in memory briefly for
|
|
386
|
+
// HUD/final-state reads, but they no longer own the host and must not keep
|
|
387
|
+
// this guard probing a disposed handle.
|
|
387
388
|
const terminalGoal = state.goal?.status === "complete" || state.goal?.status === "aborted";
|
|
388
389
|
const staleRecoveryDebt = (!terminalGoal && state.goal?.interruptedReason?.startsWith("extension api stale"))
|
|
389
390
|
|| state.loop?.stopReason?.startsWith("extension api stale");
|
|
390
|
-
|
|
391
|
+
const parkedCompletionAuditRecovery = state.goal?.status === "paused"
|
|
392
|
+
&& state.goal.pendingCompletion?.phase === "recovery-pending";
|
|
393
|
+
if (state.goal?.status !== "active"
|
|
394
|
+
&& state.goal?.status !== "auditing"
|
|
395
|
+
&& !isLoopActive()
|
|
396
|
+
&& !staleRecoveryDebt
|
|
397
|
+
&& !parkedCompletionAuditRecovery
|
|
398
|
+
&& !hasLiveSubagentHangProbes()) return;
|
|
391
399
|
// Probe the ExtensionAPI BEFORE probing the captured context. When pi
|
|
392
400
|
// invalidates both handles and emits no replacement session_start,
|
|
393
401
|
// freshCtx() deliberately returns null; probing it first used to make the
|
|
@@ -402,7 +410,9 @@ function heartbeatTick(): void {
|
|
|
402
410
|
// recovery block below is unreachable while the latch holds — the queue sat
|
|
403
411
|
// blocked ~30m+ while the worker's disapproval sat on disk. Park the stuck
|
|
404
412
|
// claim via the kept last context. A heartbeat must still NEVER launch
|
|
405
|
-
// another worker; the park is the
|
|
413
|
+
// another worker directly; the park is the durable recovery gate, and a
|
|
414
|
+
// later healthy same-session heartbeat may hand it to the one-shot recovery
|
|
415
|
+
// path. Explicit resume remains available when no healthy host returns.
|
|
406
416
|
if (
|
|
407
417
|
flags.extensionApiStale &&
|
|
408
418
|
knownCtx &&
|
|
@@ -450,6 +460,7 @@ function heartbeatTick(): void {
|
|
|
450
460
|
flags.heartbeatStaleStreak++;
|
|
451
461
|
if (flags.heartbeatStaleStreak < heartbeatStaleDebounce) return;
|
|
452
462
|
}
|
|
463
|
+
if (knownCtx && tryAbsorbHostSuccessor(knownCtx, "heartbeat-probe")) return;
|
|
453
464
|
if (knownCtx && !absorbStaleIfSuperseded(knownCtx)) goStaleTerminal(knownCtx, "heartbeat probe");
|
|
454
465
|
return;
|
|
455
466
|
}
|
|
@@ -473,19 +484,12 @@ function heartbeatTick(): void {
|
|
|
473
484
|
if (tryAbsorbHostSuccessor(knownCtx, "heartbeat-self-heal")) return;
|
|
474
485
|
rememberCtx(knownCtx);
|
|
475
486
|
// Reuse the same-session recovery gate as ordinary commands. It clears
|
|
476
|
-
// the durable interrupted marker and
|
|
477
|
-
//
|
|
478
|
-
//
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
flags.extensionApiStale
|
|
482
|
-
flags.zombieStoodDown = false;
|
|
483
|
-
flags.sessionHandoffPending = false;
|
|
484
|
-
try {
|
|
485
|
-
knownCtx.ui.notify("glla: pi recovered after a stale-handle terminal — self-healing in-memory state (no /reload needed).", "info");
|
|
486
|
-
} catch {
|
|
487
|
-
/* the ledger is the durable record; notify is best-effort */
|
|
488
|
-
}
|
|
487
|
+
// the durable interrupted marker and rebinds the owner only after BOTH
|
|
488
|
+
// the context and the captured ExtensionAPI are healthy. If that gate
|
|
489
|
+
// refuses the contact, stay parked: clearing these flags here used to
|
|
490
|
+
// announce a false recovery and immediately retry against the same stale
|
|
491
|
+
// API every heartbeat tick.
|
|
492
|
+
if (flags.staleTerminalDone || flags.extensionApiStale) return;
|
|
489
493
|
}
|
|
490
494
|
const ctx = freshCtx();
|
|
491
495
|
if (!ctx) return;
|
|
@@ -14,7 +14,23 @@ import { createHash, randomUUID } from "node:crypto";
|
|
|
14
14
|
import * as path from "node:path";
|
|
15
15
|
import { fileURLToPath } from "node:url";
|
|
16
16
|
|
|
17
|
-
import {
|
|
17
|
+
import {
|
|
18
|
+
stripThinkBlocks,
|
|
19
|
+
captureGoalRevision,
|
|
20
|
+
isRetriableInfraError,
|
|
21
|
+
isForbiddenModel,
|
|
22
|
+
type Goal,
|
|
23
|
+
type GoalRevisionToken,
|
|
24
|
+
} from "./goal-loop-core.js";
|
|
25
|
+
import {
|
|
26
|
+
classifyMainModelFailure,
|
|
27
|
+
isMainModelFallbackFailure,
|
|
28
|
+
mainModelFailureDelayMs,
|
|
29
|
+
modelRef,
|
|
30
|
+
nextUntriedModelRef,
|
|
31
|
+
normalizeMainModelFallbackRefs,
|
|
32
|
+
} from "./main-model-recovery.js";
|
|
33
|
+
import { ModelSelector, type ModelFallbackEvent } from "./model-selector.js";
|
|
18
34
|
import { buildGoalAuditorPrompt } from "./goal-loop-auditor.js";
|
|
19
35
|
import { checkRegressionShield, parseAuditorVerdict } from "./goal-loop-shield.js";
|
|
20
36
|
import { renameWithWindowsRetry } from "../scripts/goal-auditor-launch.mjs";
|
|
@@ -63,6 +79,159 @@ export interface AuditorProgress {
|
|
|
63
79
|
|
|
64
80
|
export type AuditorModel = string | { provider: string; id: string };
|
|
65
81
|
|
|
82
|
+
/** A resolved auditor candidate. `ref` is optional for compatibility with
|
|
83
|
+
* older callers; the shared fallback walker derives it from `model` when it
|
|
84
|
+
* can and otherwise treats the candidate as a unique, last-resort slot. */
|
|
85
|
+
export interface AuditorFallbackCandidate {
|
|
86
|
+
ref?: string;
|
|
87
|
+
model: any;
|
|
88
|
+
via: string;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
export interface AuditorFallbackPolicyOptions {
|
|
92
|
+
/** The user-configured forbidden refs. The selector skips these silently. */
|
|
93
|
+
forbiddenRefs?: readonly string[];
|
|
94
|
+
/** Lifecycle fence checked before and after each delayed attempt. */
|
|
95
|
+
shouldRetry?: () => boolean;
|
|
96
|
+
sleep?: (ms: number) => Promise<void>;
|
|
97
|
+
retryBaseMinutes?: number;
|
|
98
|
+
onRetry?: (candidate: AuditorFallbackCandidate, error: string, delayMs: number) => void;
|
|
99
|
+
onFallback?: (from: AuditorFallbackCandidate, to: AuditorFallbackCandidate, error: string, delayMs: number) => void;
|
|
100
|
+
onSelection?: (event: ModelFallbackEvent) => void;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* Run detached auditor candidates through the same policy as main-model
|
|
105
|
+
* recovery: normalize the ordered refs, gate forbidden/unregistered refs,
|
|
106
|
+
* select only an untried ref, classify provider failures, retry the current
|
|
107
|
+
* ref once, then use the bounded shared backoff before walking to the next
|
|
108
|
+
* ref. The worker transport remains unchanged; this function only owns the
|
|
109
|
+
* parent-side candidate cursor and timing.
|
|
110
|
+
*/
|
|
111
|
+
export async function runAuditorFallbackWithPolicy(
|
|
112
|
+
candidates: AuditorFallbackCandidate[],
|
|
113
|
+
run: (candidate: AuditorFallbackCandidate) => Promise<GoalAuditorResult>,
|
|
114
|
+
opts: AuditorFallbackPolicyOptions = {},
|
|
115
|
+
): Promise<{ result: GoalAuditorResult; retriedOnce: boolean; fallbackUsed: boolean; via: string }> {
|
|
116
|
+
const sequence = candidates.length > 0 ? candidates : [{ model: undefined, via: "unset" }];
|
|
117
|
+
if (candidates.length === 0) {
|
|
118
|
+
const result = await run(sequence[0]!);
|
|
119
|
+
return { result, retriedOnce: false, fallbackUsed: false, via: "unset" };
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
const normalized = sequence.map((candidate, index) => ({
|
|
123
|
+
candidate,
|
|
124
|
+
ref: (candidate.ref?.trim() || modelRef(candidate.model) || `auditor/candidate-${index}`),
|
|
125
|
+
}));
|
|
126
|
+
const refs = normalizeMainModelFallbackRefs(normalized.map((entry) => entry.ref));
|
|
127
|
+
const byRef = new Map<string, AuditorFallbackCandidate>();
|
|
128
|
+
for (const entry of normalized) {
|
|
129
|
+
const key = entry.ref.toLowerCase();
|
|
130
|
+
if (!byRef.has(key)) byRef.set(key, entry.candidate);
|
|
131
|
+
}
|
|
132
|
+
const scope = { kind: "auditor" } as const;
|
|
133
|
+
const selector = new ModelSelector({
|
|
134
|
+
getChain: () => refs,
|
|
135
|
+
resolve: (ref) => byRef.get(ref.toLowerCase())?.model,
|
|
136
|
+
isForbidden: (ref) => isForbiddenModel(ref, opts.forbiddenRefs ?? []),
|
|
137
|
+
record: opts.onSelection,
|
|
138
|
+
});
|
|
139
|
+
const attempted: string[] = [];
|
|
140
|
+
const addAttempted = (ref: string): void => {
|
|
141
|
+
if (!attempted.some((entry) => entry.toLowerCase() === ref.toLowerCase())) attempted.push(ref);
|
|
142
|
+
};
|
|
143
|
+
const isLive = (): boolean => {
|
|
144
|
+
if (!opts.shouldRetry) return true;
|
|
145
|
+
try { return opts.shouldRetry(); } catch { return false; }
|
|
146
|
+
};
|
|
147
|
+
const sleep = opts.sleep ?? ((ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms)));
|
|
148
|
+
let currentRef: string | undefined;
|
|
149
|
+
let retriedOnce = false;
|
|
150
|
+
let fallbackUsed = false;
|
|
151
|
+
let failureAttempt = 0;
|
|
152
|
+
let fallbackFrom: AuditorFallbackCandidate | undefined;
|
|
153
|
+
let fallbackError: string | undefined;
|
|
154
|
+
let fallbackDelayMs = 0;
|
|
155
|
+
let pendingResult: GoalAuditorResult | undefined;
|
|
156
|
+
const noCandidateResult = (): GoalAuditorResult => ({
|
|
157
|
+
approved: false,
|
|
158
|
+
disapproved: false,
|
|
159
|
+
output: "",
|
|
160
|
+
model: modelRef(sequence[0]!.model) ?? "",
|
|
161
|
+
error: ["no auditor", "model"].join(" "),
|
|
162
|
+
infrastructureClass: "no-verdict",
|
|
163
|
+
});
|
|
164
|
+
|
|
165
|
+
for (;;) {
|
|
166
|
+
// Keep the explicit pure cursor call here. ModelSelector.selectNextValid
|
|
167
|
+
// composes the same helper while adding the forbidden/unregistered walk.
|
|
168
|
+
if (nextUntriedModelRef(currentRef, refs, attempted) === undefined) {
|
|
169
|
+
const last = pendingResult ?? noCandidateResult();
|
|
170
|
+
return { result: last, retriedOnce, fallbackUsed, via: fallbackFrom?.via ?? sequence[0]!.via };
|
|
171
|
+
}
|
|
172
|
+
const selected = selector.selectNextValid(scope, currentRef, attempted);
|
|
173
|
+
for (const visited of selector.lastVisitedRefs) addAttempted(visited);
|
|
174
|
+
if (!("model" in selected) || typeof selected.ref !== "string") {
|
|
175
|
+
const result = pendingResult ?? noCandidateResult();
|
|
176
|
+
return { result, retriedOnce, fallbackUsed, via: fallbackFrom?.via ?? sequence[0]!.via };
|
|
177
|
+
}
|
|
178
|
+
const selectedRef = selected.ref;
|
|
179
|
+
const candidate = byRef.get(selectedRef.toLowerCase());
|
|
180
|
+
if (!candidate) {
|
|
181
|
+
addAttempted(selectedRef);
|
|
182
|
+
currentRef = selectedRef;
|
|
183
|
+
continue;
|
|
184
|
+
}
|
|
185
|
+
addAttempted(selectedRef);
|
|
186
|
+
if (fallbackFrom) {
|
|
187
|
+
fallbackUsed = true;
|
|
188
|
+
opts.onFallback?.(fallbackFrom, candidate, fallbackError ?? "auditor fallback", fallbackDelayMs);
|
|
189
|
+
fallbackFrom = undefined;
|
|
190
|
+
fallbackError = undefined;
|
|
191
|
+
fallbackDelayMs = 0;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
pendingResult = undefined;
|
|
195
|
+
const first = await run(candidate);
|
|
196
|
+
if (first.approved || first.disapproved || first.impossible || !first.error) {
|
|
197
|
+
return { result: first, retriedOnce, fallbackUsed, via: candidate.via };
|
|
198
|
+
}
|
|
199
|
+
let failure = classifyMainModelFailure(first.error);
|
|
200
|
+
if (!isRetriableInfraError(first.error) || !isMainModelFallbackFailure(failure)) {
|
|
201
|
+
return { result: first, retriedOnce, fallbackUsed, via: candidate.via };
|
|
202
|
+
}
|
|
203
|
+
failureAttempt += 1;
|
|
204
|
+
if (!isLive()) return { result: first, retriedOnce, fallbackUsed, via: candidate.via };
|
|
205
|
+
const retryDelayMs = mainModelFailureDelayMs(failure, failureAttempt, opts.retryBaseMinutes ?? 15);
|
|
206
|
+
opts.onRetry?.(candidate, first.error, retryDelayMs);
|
|
207
|
+
await sleep(retryDelayMs);
|
|
208
|
+
if (!isLive()) return { result: first, retriedOnce, fallbackUsed, via: candidate.via };
|
|
209
|
+
|
|
210
|
+
const second = await run(candidate);
|
|
211
|
+
pendingResult = second;
|
|
212
|
+
retriedOnce = true;
|
|
213
|
+
if (second.approved || second.disapproved || second.impossible || !second.error) {
|
|
214
|
+
return { result: second, retriedOnce, fallbackUsed, via: candidate.via };
|
|
215
|
+
}
|
|
216
|
+
failure = classifyMainModelFailure(second.error);
|
|
217
|
+
if (!isRetriableInfraError(second.error) || !isMainModelFallbackFailure(failure)) {
|
|
218
|
+
return { result: second, retriedOnce, fallbackUsed, via: candidate.via };
|
|
219
|
+
}
|
|
220
|
+
currentRef = selectedRef;
|
|
221
|
+
const nextRef = nextUntriedModelRef(currentRef, refs, attempted);
|
|
222
|
+
if (nextRef === undefined) {
|
|
223
|
+
return { result: second, retriedOnce, fallbackUsed, via: candidate.via };
|
|
224
|
+
}
|
|
225
|
+
failureAttempt += 1;
|
|
226
|
+
fallbackDelayMs = mainModelFailureDelayMs(failure, failureAttempt, opts.retryBaseMinutes ?? 15);
|
|
227
|
+
if (!isLive()) return { result: second, retriedOnce, fallbackUsed, via: candidate.via };
|
|
228
|
+
await sleep(fallbackDelayMs);
|
|
229
|
+
if (!isLive()) return { result: second, retriedOnce, fallbackUsed, via: candidate.via };
|
|
230
|
+
fallbackFrom = candidate;
|
|
231
|
+
fallbackError = second.error;
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
|
|
66
235
|
// The detached auditor intentionally exposes the full inspection/tooling
|
|
67
236
|
// surface, including bash, so it can run bounded tests and reproduce behavior.
|
|
68
237
|
// This is a power-oriented mode, not a read-only security boundary; callers
|
|
@@ -698,11 +867,23 @@ export async function runDetachedGoalCompletionAuditor(args: {
|
|
|
698
867
|
await writeAtomicJson(lockPath, { protocolVersion: PROTOCOL_VERSION, attemptId, pid: child.pid, role: "worker", workerPath: workerPathIdentity });
|
|
699
868
|
child.unref();
|
|
700
869
|
|
|
701
|
-
|
|
870
|
+
let abortTermination: Promise<void> | null = null;
|
|
871
|
+
const abort = () => {
|
|
872
|
+
if (child && childAlive(child)) {
|
|
873
|
+
abortTermination ??= terminateWorker(child).catch(() => {});
|
|
874
|
+
}
|
|
875
|
+
};
|
|
702
876
|
args.signal?.addEventListener("abort", abort, { once: true });
|
|
703
877
|
try {
|
|
704
878
|
while (true) {
|
|
705
|
-
if (args.signal?.aborted)
|
|
879
|
+
if (args.signal?.aborted) {
|
|
880
|
+
// Do not return the transport result until the detached worker's
|
|
881
|
+
// TERM→KILL teardown has settled. Returning first races the caller's
|
|
882
|
+
// cleanup with a TERM-ignoring worker and leaves its PID alive.
|
|
883
|
+
if (child && childAlive(child)) await terminateWorker(child).catch(() => {});
|
|
884
|
+
else if (abortTermination) await abortTermination;
|
|
885
|
+
return infra(model, thinkingLevel, "Auditor aborted.", "", capturedRevisionToken, "transport");
|
|
886
|
+
}
|
|
706
887
|
if (childSpawnError) return infra(model, thinkingLevel, `auditor worker launch failed: ${childSpawnError}`, "", capturedRevisionToken, "transport");
|
|
707
888
|
if (now() >= wallDeadlineAt) {
|
|
708
889
|
if (childAlive(child)) await terminateWorker(child);
|