pi-goal-list-loop-audit 0.35.4 → 0.35.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,119 @@
1
+ # Examples: the three loops
2
+
3
+ Worked examples for `pi-goal-list-loop-audit`. State lives at `.pi-glla/`
4
+ (ledger `active.jsonl`, goal markdown in `goals/`, finished goals in `archive/`).
5
+
6
+ ## Loop 1: `/goal` — single goal with isolated auditor
7
+
8
+ Direct (skip drafting):
9
+
10
+ ```
11
+ /goal start Step 1. Add a /healthz endpoint to server.ts returning {status:'ok'}.
12
+ Step 2. Add a vitest test that hits it.
13
+ Done when:
14
+ - curl -fsS localhost:3000/healthz returns 200 with body {"status":"ok"}
15
+ - npm test exits 0 with 0 failures
16
+ ```
17
+
18
+ Or drafting (recommended when the idea is still fuzzy):
19
+
20
+ ```
21
+ /goal
22
+ # the agent grills one focused question at a time, then a Confirm dialog
23
+ # shows the objective + Done-when contract. Nothing starts before you confirm.
24
+ ```
25
+
26
+ What happens next:
27
+
28
+ 1. The agent works the objective turn by turn (`agent_end`-driven continuation).
29
+ Guards: a stall watchdog (3 consecutive turns with no tool calls), a
30
+ 5-consecutive-errors pause, and the optional token guard — no wall-clock
31
+ cap; a goal ends via completion, pause/cancel, or a guard.
32
+ 2. It calls `complete_goal` when it believes it is done.
33
+ 3. The **isolated auditor** — a fresh pi session with no extensions and only
34
+ read tools — inspects the repo. Because this goal has a `Done when:`
35
+ contract, the auditor MUST quote raw command output per contract item in an
36
+ `<evidence>` block (regression_shield). An approval without complete
37
+ evidence is automatically converted to a disapproval.
38
+ 4. Approved → goal archived to `.pi-glla/archive/<id>.md`. Disapproved → the
39
+ loop continues with the auditor's feedback.
40
+
41
+ Useful commands: `/goal status`, `/goal pause`, `/goal resume`, `/goal cancel`,
42
+ `/glla` (auditor model, thinking level, token limit, notify command, autoresume).
43
+
44
+ ## Loop 2: `/list` — the shopping list of goals
45
+
46
+ ```
47
+ /list Create one.txt containing one. Done when: grep -q one one.txt
48
+ /list Create two.txt containing two. Done when: grep -q two two.txt
49
+ /list # show active + waiting items
50
+ /list next # skip current item
51
+ /list remove 2 # drop item 2
52
+ /list clear # empty the list
53
+ ```
54
+
55
+ Each item is a full goal with its own contract and audit. When one completes
56
+ (or is aborted), the next activates automatically. Order is the default, not
57
+ the law — `/list next 3` activates item 3 directly. On session restore the
58
+ list HOLDS in a fresh session (nothing auto-starts); it auto-activates only
59
+ when Auto-resume is enabled in `/glla` settings (global-only since v0.29.5).
60
+
61
+ ## Loop 3: `/loop` — metric-driven forever loop
62
+
63
+ ```
64
+ /loop start "reduce TODO comments in src" measure="grep -rc TODO src | cut -d: -f2 | paste -sd+ | bc" direction=min window=5 max=50
65
+ /loop status # iteration, best, stall, recent values
66
+ /loop stop # halt with summary
67
+ ```
68
+
69
+ The **orchestrator** runs your `measure` command after every agent turn — the
70
+ agent never self-reports a number. The loop stops on plateau (`window`
71
+ non-improving iterations), the `max` cap, a time/token bound, or `/loop stop`.
72
+ There is no completion check in loop 3 — "improve until X" is a GOAL; a loop
73
+ is a process. No auditor either: the metric is the verdict.
74
+
75
+ With `branch=1`, all work happens on a scratch branch
76
+ (`pi-glla-loop/<timestamp>-<slug>`): each improvement is committed, each
77
+ regression is hard-reset (scratch branch only), and on stop you return to your
78
+ original branch with merge instructions. Requires a clean working tree.
79
+
80
+ ## Notifications (optional)
81
+
82
+ Run `/glla` and set the **Notify command** row (the event message is passed
83
+ as `$1`); `off` silences pushes, empty re-enables auto-detection. `/glla`
84
+ has an action namespace (`status`, `log`, `stats`, `audits`, `wipe`, …) —
85
+ it does NOT accept inline `key=value` assignments. Headless/scripted setups
86
+ can write the project settings file directly instead:
87
+
88
+ ```
89
+ # .pi-glla/settings.json
90
+ { "notifyCmd": "echo \"$1\" >> ~/goal-events.log" }
91
+ ```
92
+
93
+ Fires on goal complete, goal pause, and loop stop; message as `$1`.
94
+
95
+ ## Token guard (opt-in)
96
+
97
+ Off by default. Set a per-goal budget and crossing it pauses the goal:
98
+
99
+ ```
100
+ # /glla → Token limit row, or .pi-glla/settings.json
101
+ { "tokenLimit": 2000000 }
102
+ ```
103
+
104
+ ## The auditor model rule
105
+
106
+ The detached auditor uses an explicit bounded cascade: an optional primary
107
+ pin, an optional fallback pin, then your pi session model. A model that fails
108
+ at runtime is retried once and the next detached candidate is tried; the
109
+ plugin never falls back into the parent in-process session.
110
+
111
+ ```
112
+ # /glla → Auditor model row (and Auditor fallback), or .pi-glla/settings.json
113
+ { "auditorModel": "provider/model-id" }
114
+ ```
115
+
116
+ If every candidate errors with auth/provider failures, the stored completion
117
+ claim pauses and `/goal resume` retries it after you fix the model — the
118
+ auditor session has no extensions, so
119
+ providers that only exist via an extension may be unavailable to it.
@@ -35,10 +35,21 @@ export interface ObjectiveRepairProposal {
35
35
  confidence: "best-effort";
36
36
  }
37
37
 
38
- const IMPERATIVE_START = /^(add|allow|audit|build|cap|clarify|close|collapse|consolidate|create|detect|document|ensure|enforce|fix|harden|improve|implement|instrument|investigate|make|migrate|overhaul|preserve|prevent|recover|refactor|remove|repair|replace|research|resolve|restore|review|ship|simplify|strengthen|support|test|update|validate|verify|wire|write)\b/i;
38
+ // A verb that plausibly opens a real, actionable objective. The reviewer/
39
+ // verification momentum lives in the vocabulary and narrative regexes below;
40
+ // this list is the escape hatch that keeps genuine imperatives (including
41
+ // ones that mention the auditor/reviewer machinery legitimately, e.g.
42
+ // "Show the selected models … (main/auditor/drafter) on the goal card")
43
+ // from being misread as report text. Keep it broad: an objective that starts
44
+ // with any of these is actionable on its face.
45
+ const IMPERATIVE_START = /^(add|allow|audit|benchmark|build|cap|check|clarify|close|collapse|collect|compare|consolidate|create|describe|detect|diagnose|display|document|ensure|enforce|explain|expose|fix|harden|improve|implement|inspect|instrument|investigate|list|make|measure|migrate|monitor|open|optimize|overhaul|plan|preserve|prevent|print|profile|publish|read|recover|refactor|remove|render|repair|replace|research|resolve|restore|review|ship|show|simplify|strengthen|summarize|support|surface|test|trace|update|validate|verify|watch|wire|write)\b/i;
39
46
  const COMMAND_ONLY = /^(?:bun|npm|pnpm|yarn|npx|node|deno|git)\s+(?:test|run|check|diff|status|show|log|exec)\b/i;
40
47
  const REVIEWER_MARKER = /(?:^|[.!?;]\s+|[-*]\s+)(?:audit(?:\s+(?:report|result|findings?))?|review(?:er)?(?:\s+(?:report|result|feedback|findings?))?|verdict|evidence|output|item|required\s+fixes?|completion\s+claim)\s*:/i;
41
- const REVIEWER_VOCABULARY = /\b(?:passes\s+sequentially|zero\s+failures?|\d+\s+failures?|ran\s+\d+\s+tests?|verification\s+contract|regression\s+shield|auditor(?:[- ](?:approved|report|disapproved))?|reviewer(?:[- ](?:approved|report|disapproved|feedback|finding))?|completion\s+claim|<\/?(?:evidence|approved|disapproved|impossible)\b)\b/i;
48
+ // Report vocabulary. "auditor"/"reviewer" count ONLY when tied to a verdict
49
+ // shape ("auditor approved", "reviewer finding") — a bare mention of the
50
+ // role ("(main/auditor/drafter)", "auditor workers") is legitimate task
51
+ // vocabulary and must not trip verification-fragment on its own.
52
+ const REVIEWER_VOCABULARY = /\b(?:passes\s+sequentially|zero\s+failures?|\d+\s+failures?|ran\s+\d+\s+tests?|verification\s+contract|regression\s+shield|auditor[- ]+(?:approved|report|disapproved)|reviewer[- ]+(?:approved|report|disapproved|feedback|finding)|completion\s+claim|<\/?(?:evidence|approved|disapproved|impossible)\b)\b/i;
42
53
  // A report can look superficially task-like because it lists completed work
43
54
  // with imperative-shaped nouns ("Added ...", "Focused tests ..."). Keep
44
55
  // genuine phrases such as "Add focused tests for the audit path" valid, but
@@ -14,7 +14,7 @@ import {
14
14
  DEFAULT_TOKEN_LIMIT, Goal, ListItem, Status, appendLedger, archiveDir, archivedGoalPath, bumpGoalRevision, sanitizeProviderDisplayText,
15
15
  computeListDepth, clearQueueItemFiles, deleteQueueItemFile, extractVerificationContract, formatAuditLog, formatGoalAuditHistory, formatMainModelRecoveryStatus, queueItemSidecarCount,
16
16
  formatListDepth, goalArgsNeedDrafting, ledgerPath, newGoalId, nowIso, parseListImport, parseListItemDeclaration,
17
- readAuditLog, readQueueFromDisk, routeGoalArgs, routeListText, sanitizeDisplayText, sanitizeProviderAuditReport, statusLabel,
17
+ assignQueueOrder, compareQueueItems, readAuditLog, readQueueFromDisk, routeGoalArgs, routeListText, sanitizeDisplayText, sanitizeProviderAuditReport, statusLabel,
18
18
  writeQueueItemFile, type ModeCommand, type State, LIST_MUTATING_SUBCOMMANDS, SETTINGS_MUTATING_ACTIONS,
19
19
  } from "./goal-loop-core.js";
20
20
  import { clearDispatchRecord, dispatchRecordExists } from "./goal-loop-dispatch.js";
@@ -341,6 +341,13 @@ async function cmdResume(ctx: ExtensionContext): Promise<void> {
341
341
  void probeMainModelRecovery(ctx);
342
342
  return;
343
343
  }
344
+ if (state.mainModelRecovery?.primaryProbeAt || state.mainModelRecovery?.primaryProbeInFlight) {
345
+ clearMainModelRecoveryTimer();
346
+ flags.continuationDispatchStoodDown = false;
347
+ ctx.ui.notify("Probing the preferred primary now — the current fallback remains available if the primary is still unhealthy.", "info");
348
+ void probeMainModelRecovery(ctx);
349
+ return;
350
+ }
344
351
  // v0.34.3: /goal resume on an ACTIVE-but-idle goal re-kicks its
345
352
  // continuation (was: silent return — the user got NOTHING while the
346
353
  // widget said "active"). One-active-thing still holds: an active loop
@@ -836,7 +843,23 @@ function recentlyCompletedObjectives(cwd: string): Set<string> {
836
843
  return done;
837
844
  }
838
845
 
846
+ export function hydrateListQueueFromDisk(ctx: ExtensionContext): number {
847
+ const memory = listQueue();
848
+ const exclude = new Set<string>();
849
+ if (state.goal?.id) exclude.add(state.goal.id);
850
+ const disk = readQueueFromDisk(ctx.cwd, exclude);
851
+ const known = new Set(memory.map((item) => item.id));
852
+ const recovered = disk.filter((item) => !known.has(item.id));
853
+ if (recovered.length === 0) return 0;
854
+ const merged = [...memory, ...recovered].sort(compareQueueItems);
855
+ replaceState({ ...state, list: merged });
856
+ persistState(ctx);
857
+ appendLedger(ctx.cwd, "list_recovered_from_disk", { count: recovered.length, hydrated: true });
858
+ return recovered.length;
859
+ }
860
+
839
861
  function enqueueItems(ctx: ExtensionContext, texts: string[], source: string, opts?: { autoActivate?: boolean }): number {
862
+ hydrateListQueueFromDisk(ctx);
840
863
  const recentlyDone = recentlyCompletedObjectives(ctx.cwd);
841
864
  const fresh = texts.filter((t) => !recentlyDone.has(normalizeObjective(parseListItemDeclaration(t).objective)));
842
865
  const skipped = texts.length - fresh.length;
@@ -899,13 +922,23 @@ function enqueueItems(ctx: ExtensionContext, texts: string[], source: string, op
899
922
  ctx.ui.notify(`Refused ${refused.length} subtask item(s): ${refused.join(" | ")}.`, "warning");
900
923
  }
901
924
  if (resolved.length === 0) return 0;
902
- const itemsToWrite = resolved;
925
+ const itemsToWrite = assignQueueOrder(resolved, listQueue());
903
926
  // v0.34.60: disk-first write order. Each item lands on disk BEFORE any
904
927
  // in-memory state mutation, so /list survives a stale extension handle
905
928
  // (e.g. /reload, plugin re-init, RAM-only state loss). The
906
929
  // .queue.json sidecar is atomic (temp + rename) and idempotent (skips
907
- // existing files rather than overwriting).
930
+ // existing files rather than overwriting). A failed member aborts the
931
+ // batch and removes only sidecars written by this attempt.
908
932
  const written = itemsToWrite.map((item) => writeQueueItemFile(ctx.cwd, item));
933
+ const failedWrite = written.find((result) => result.failed);
934
+ if (failedWrite) {
935
+ written.forEach((result, index) => {
936
+ if (result.wrote) deleteQueueItemFile(ctx.cwd, itemsToWrite[index]!.id);
937
+ });
938
+ appendLedger(ctx.cwd, "list_queue_disk_write_failed", { source, count: itemsToWrite.length, path: failedWrite.path });
939
+ ctx.ui.notify(`Could not persist the queued item(s) from ${source}; no in-memory queue mutation was applied. Fix disk access and retry.`, "warning");
940
+ return 0;
941
+ }
909
942
  replaceState({ ...state, list: [...listQueue(), ...itemsToWrite] });
910
943
  const diskFirst = written.filter((w) => w.wrote).length === itemsToWrite.length;
911
944
  appendLedger(ctx.cwd, "list_queue_disk_first", { source, count: itemsToWrite.length, diskFirst });
@@ -991,6 +1024,10 @@ async function cmdList(args: string, ctx: ExtensionContext): Promise<void> {
991
1024
  // resume/tweak/cancel gates branch on state.goal.policy — the gate used
992
1025
  // to silently refuse the whole surface until a restart.
993
1026
  healGoalPolicy(ctx);
1027
+ // v0.35.4 audit: sidecars are durable queue state, not a display-only
1028
+ // fallback. Hydrate orphaned disk items before any list surface can count,
1029
+ // remove, cancel, or activate them.
1030
+ if (!staleEntry) hydrateListQueueFromDisk(ctx);
994
1031
 
995
1032
  if (sub === "audit") {
996
1033
  // v0.31.0: /list audit [focus] — collect-then-drain (user design
@@ -1332,19 +1369,24 @@ async function cmdList(args: string, ctx: ExtensionContext): Promise<void> {
1332
1369
 
1333
1370
  /** Append one objective to the list; activate immediately when idle. */
1334
1371
  function addSingleItem(ctx: ExtensionContext, raw: string): void {
1372
+ hydrateListQueueFromDisk(ctx);
1335
1373
  const extracted = parseListItemDeclaration(raw);
1336
- const item = {
1374
+ const item = assignQueueOrder([{
1337
1375
  id: newGoalId(),
1338
1376
  objective: extracted.objective,
1339
1377
  ...(extracted.agentRole ? { agentRole: extracted.agentRole } : {}),
1340
1378
  verificationContract: extracted.verificationContract || undefined,
1341
1379
  ...(extracted.parallelSafe === undefined ? {} : { parallelSafe: extracted.parallelSafe }),
1342
1380
  addedAt: nowIso(),
1343
- };
1381
+ }], listQueue())[0]!;
1344
1382
  // v0.34.61: disk-first — write the sidecar BEFORE mutating state so the
1345
1383
  // item survives an orchestrator-turn death between state mutation and
1346
1384
  // persistState (the original bug for /list add "<direct text>").
1347
- writeQueueItemFile(ctx.cwd, item);
1385
+ const written = writeQueueItemFile(ctx.cwd, item);
1386
+ if (written.failed) {
1387
+ ctx.ui.notify("Could not persist the list item; no in-memory queue mutation was applied. Fix disk access and retry.", "warning");
1388
+ return;
1389
+ }
1348
1390
  replaceState({ ...state, list: [...listQueue(), item] });
1349
1391
  persistState(ctx);
1350
1392
  appendLedger(ctx.cwd, "list_added", { id: item.id, objective: item.objective });
@@ -2126,6 +2168,8 @@ async function cmdSettings(args: string, ctx: ExtensionContext): Promise<void> {
2126
2168
  [
2127
2169
  `mainAgentFallbackModels: ${formatMainModelFallbacks(effectiveSettings.mainModelFallbacks)} [${prov.mainModelFallbacks?.source ?? "default"}]`,
2128
2170
  fmt("mainModelRetryMinutes", "mainModelRetryMinutes (base minutes; doubles per attempt)"),
2171
+ fmt("mainModelFailback", "mainModelFailback (auto/sticky)"),
2172
+ fmt("mainModelPrimaryProbeMinutes", "mainModelPrimaryProbeMinutes"),
2129
2173
  fmt("drafterModel", "drafterAgent (drafting only)"),
2130
2174
  fmt("drafterThinkingLevel", "drafterThinking (drafting only)"),
2131
2175
  fmt("drafterModelFallbacks", "drafterFallbackAgents (drafting only)"),
@@ -378,16 +378,24 @@ function heartbeatTick(): void {
378
378
  // that disposed handle as a host loss only creates the repeated
379
379
  // "session invalidated" warning after the work is already safe on disk.
380
380
  // Keep the probe for active goals/loops, detached completion audits,
381
- // recoverable stale debt, and LIVE tracked subagents, where a dead handle
382
- // can strand live work. A normal user pause does not count as host-bound
383
- // work; stale interruption debt does, because same-process self-heal still
384
- // needs a heartbeat opportunity. Ended subagent probes remain in memory
385
- // briefly for HUD/final-state reads, but they no longer own the host and
386
- // must not keep this guard probing a disposed handle.
381
+ // recoverable stale debt, parked completion-audit recovery, and LIVE
382
+ // tracked subagents, where a dead handle can strand live work. A normal
383
+ // user pause does not count as host-bound work; stale interruption debt and
384
+ // parked completion claims do, because same-process self-heal still needs a
385
+ // heartbeat opportunity. Ended subagent probes remain in memory briefly for
386
+ // HUD/final-state reads, but they no longer own the host and must not keep
387
+ // this guard probing a disposed handle.
387
388
  const terminalGoal = state.goal?.status === "complete" || state.goal?.status === "aborted";
388
389
  const staleRecoveryDebt = (!terminalGoal && state.goal?.interruptedReason?.startsWith("extension api stale"))
389
390
  || state.loop?.stopReason?.startsWith("extension api stale");
390
- if (state.goal?.status !== "active" && state.goal?.status !== "auditing" && !isLoopActive() && !staleRecoveryDebt && !hasLiveSubagentHangProbes()) return;
391
+ const parkedCompletionAuditRecovery = state.goal?.status === "paused"
392
+ && state.goal.pendingCompletion?.phase === "recovery-pending";
393
+ if (state.goal?.status !== "active"
394
+ && state.goal?.status !== "auditing"
395
+ && !isLoopActive()
396
+ && !staleRecoveryDebt
397
+ && !parkedCompletionAuditRecovery
398
+ && !hasLiveSubagentHangProbes()) return;
391
399
  // Probe the ExtensionAPI BEFORE probing the captured context. When pi
392
400
  // invalidates both handles and emits no replacement session_start,
393
401
  // freshCtx() deliberately returns null; probing it first used to make the
@@ -402,7 +410,9 @@ function heartbeatTick(): void {
402
410
  // recovery block below is unreachable while the latch holds — the queue sat
403
411
  // blocked ~30m+ while the worker's disapproval sat on disk. Park the stuck
404
412
  // claim via the kept last context. A heartbeat must still NEVER launch
405
- // another worker; the park is the explicit-resume gate.
413
+ // another worker directly; the park is the durable recovery gate, and a
414
+ // later healthy same-session heartbeat may hand it to the one-shot recovery
415
+ // path. Explicit resume remains available when no healthy host returns.
406
416
  if (
407
417
  flags.extensionApiStale &&
408
418
  knownCtx &&
@@ -450,6 +460,7 @@ function heartbeatTick(): void {
450
460
  flags.heartbeatStaleStreak++;
451
461
  if (flags.heartbeatStaleStreak < heartbeatStaleDebounce) return;
452
462
  }
463
+ if (knownCtx && tryAbsorbHostSuccessor(knownCtx, "heartbeat-probe")) return;
453
464
  if (knownCtx && !absorbStaleIfSuperseded(knownCtx)) goStaleTerminal(knownCtx, "heartbeat probe");
454
465
  return;
455
466
  }
@@ -473,19 +484,12 @@ function heartbeatTick(): void {
473
484
  if (tryAbsorbHostSuccessor(knownCtx, "heartbeat-self-heal")) return;
474
485
  rememberCtx(knownCtx);
475
486
  // Reuse the same-session recovery gate as ordinary commands. It clears
476
- // the durable interrupted marker and can resume unattended work; the
477
- // fallback below preserves the old probe-only behavior if ownership is
478
- // still ambiguous.
479
- if (!flags.staleTerminalDone && !flags.extensionApiStale) return;
480
- flags.staleTerminalDone = false;
481
- flags.extensionApiStale = false;
482
- flags.zombieStoodDown = false;
483
- flags.sessionHandoffPending = false;
484
- try {
485
- knownCtx.ui.notify("glla: pi recovered after a stale-handle terminal — self-healing in-memory state (no /reload needed).", "info");
486
- } catch {
487
- /* the ledger is the durable record; notify is best-effort */
488
- }
487
+ // the durable interrupted marker and rebinds the owner only after BOTH
488
+ // the context and the captured ExtensionAPI are healthy. If that gate
489
+ // refuses the contact, stay parked: clearing these flags here used to
490
+ // announce a false recovery and immediately retry against the same stale
491
+ // API every heartbeat tick.
492
+ if (flags.staleTerminalDone || flags.extensionApiStale) return;
489
493
  }
490
494
  const ctx = freshCtx();
491
495
  if (!ctx) return;
@@ -14,7 +14,23 @@ import { createHash, randomUUID } from "node:crypto";
14
14
  import * as path from "node:path";
15
15
  import { fileURLToPath } from "node:url";
16
16
 
17
- import { stripThinkBlocks, captureGoalRevision, type Goal, type GoalRevisionToken } from "./goal-loop-core.js";
17
+ import {
18
+ stripThinkBlocks,
19
+ captureGoalRevision,
20
+ isRetriableInfraError,
21
+ isForbiddenModel,
22
+ type Goal,
23
+ type GoalRevisionToken,
24
+ } from "./goal-loop-core.js";
25
+ import {
26
+ classifyMainModelFailure,
27
+ isMainModelFallbackFailure,
28
+ mainModelFailureDelayMs,
29
+ modelRef,
30
+ nextUntriedModelRef,
31
+ normalizeMainModelFallbackRefs,
32
+ } from "./main-model-recovery.js";
33
+ import { ModelSelector, type ModelFallbackEvent } from "./model-selector.js";
18
34
  import { buildGoalAuditorPrompt } from "./goal-loop-auditor.js";
19
35
  import { checkRegressionShield, parseAuditorVerdict } from "./goal-loop-shield.js";
20
36
  import { renameWithWindowsRetry } from "../scripts/goal-auditor-launch.mjs";
@@ -63,6 +79,159 @@ export interface AuditorProgress {
63
79
 
64
80
  export type AuditorModel = string | { provider: string; id: string };
65
81
 
82
+ /** A resolved auditor candidate. `ref` is optional for compatibility with
83
+ * older callers; the shared fallback walker derives it from `model` when it
84
+ * can and otherwise treats the candidate as a unique, last-resort slot. */
85
+ export interface AuditorFallbackCandidate {
86
+ ref?: string;
87
+ model: any;
88
+ via: string;
89
+ }
90
+
91
+ export interface AuditorFallbackPolicyOptions {
92
+ /** The user-configured forbidden refs. The selector skips these silently. */
93
+ forbiddenRefs?: readonly string[];
94
+ /** Lifecycle fence checked before and after each delayed attempt. */
95
+ shouldRetry?: () => boolean;
96
+ sleep?: (ms: number) => Promise<void>;
97
+ retryBaseMinutes?: number;
98
+ onRetry?: (candidate: AuditorFallbackCandidate, error: string, delayMs: number) => void;
99
+ onFallback?: (from: AuditorFallbackCandidate, to: AuditorFallbackCandidate, error: string, delayMs: number) => void;
100
+ onSelection?: (event: ModelFallbackEvent) => void;
101
+ }
102
+
103
+ /**
104
+ * Run detached auditor candidates through the same policy as main-model
105
+ * recovery: normalize the ordered refs, gate forbidden/unregistered refs,
106
+ * select only an untried ref, classify provider failures, retry the current
107
+ * ref once, then use the bounded shared backoff before walking to the next
108
+ * ref. The worker transport remains unchanged; this function only owns the
109
+ * parent-side candidate cursor and timing.
110
+ */
111
+ export async function runAuditorFallbackWithPolicy(
112
+ candidates: AuditorFallbackCandidate[],
113
+ run: (candidate: AuditorFallbackCandidate) => Promise<GoalAuditorResult>,
114
+ opts: AuditorFallbackPolicyOptions = {},
115
+ ): Promise<{ result: GoalAuditorResult; retriedOnce: boolean; fallbackUsed: boolean; via: string }> {
116
+ const sequence = candidates.length > 0 ? candidates : [{ model: undefined, via: "unset" }];
117
+ if (candidates.length === 0) {
118
+ const result = await run(sequence[0]!);
119
+ return { result, retriedOnce: false, fallbackUsed: false, via: "unset" };
120
+ }
121
+
122
+ const normalized = sequence.map((candidate, index) => ({
123
+ candidate,
124
+ ref: (candidate.ref?.trim() || modelRef(candidate.model) || `auditor/candidate-${index}`),
125
+ }));
126
+ const refs = normalizeMainModelFallbackRefs(normalized.map((entry) => entry.ref));
127
+ const byRef = new Map<string, AuditorFallbackCandidate>();
128
+ for (const entry of normalized) {
129
+ const key = entry.ref.toLowerCase();
130
+ if (!byRef.has(key)) byRef.set(key, entry.candidate);
131
+ }
132
+ const scope = { kind: "auditor" } as const;
133
+ const selector = new ModelSelector({
134
+ getChain: () => refs,
135
+ resolve: (ref) => byRef.get(ref.toLowerCase())?.model,
136
+ isForbidden: (ref) => isForbiddenModel(ref, opts.forbiddenRefs ?? []),
137
+ record: opts.onSelection,
138
+ });
139
+ const attempted: string[] = [];
140
+ const addAttempted = (ref: string): void => {
141
+ if (!attempted.some((entry) => entry.toLowerCase() === ref.toLowerCase())) attempted.push(ref);
142
+ };
143
+ const isLive = (): boolean => {
144
+ if (!opts.shouldRetry) return true;
145
+ try { return opts.shouldRetry(); } catch { return false; }
146
+ };
147
+ const sleep = opts.sleep ?? ((ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms)));
148
+ let currentRef: string | undefined;
149
+ let retriedOnce = false;
150
+ let fallbackUsed = false;
151
+ let failureAttempt = 0;
152
+ let fallbackFrom: AuditorFallbackCandidate | undefined;
153
+ let fallbackError: string | undefined;
154
+ let fallbackDelayMs = 0;
155
+ let pendingResult: GoalAuditorResult | undefined;
156
+ const noCandidateResult = (): GoalAuditorResult => ({
157
+ approved: false,
158
+ disapproved: false,
159
+ output: "",
160
+ model: modelRef(sequence[0]!.model) ?? "",
161
+ error: ["no auditor", "model"].join(" "),
162
+ infrastructureClass: "no-verdict",
163
+ });
164
+
165
+ for (;;) {
166
+ // Keep the explicit pure cursor call here. ModelSelector.selectNextValid
167
+ // composes the same helper while adding the forbidden/unregistered walk.
168
+ if (nextUntriedModelRef(currentRef, refs, attempted) === undefined) {
169
+ const last = pendingResult ?? noCandidateResult();
170
+ return { result: last, retriedOnce, fallbackUsed, via: fallbackFrom?.via ?? sequence[0]!.via };
171
+ }
172
+ const selected = selector.selectNextValid(scope, currentRef, attempted);
173
+ for (const visited of selector.lastVisitedRefs) addAttempted(visited);
174
+ if (!("model" in selected) || typeof selected.ref !== "string") {
175
+ const result = pendingResult ?? noCandidateResult();
176
+ return { result, retriedOnce, fallbackUsed, via: fallbackFrom?.via ?? sequence[0]!.via };
177
+ }
178
+ const selectedRef = selected.ref;
179
+ const candidate = byRef.get(selectedRef.toLowerCase());
180
+ if (!candidate) {
181
+ addAttempted(selectedRef);
182
+ currentRef = selectedRef;
183
+ continue;
184
+ }
185
+ addAttempted(selectedRef);
186
+ if (fallbackFrom) {
187
+ fallbackUsed = true;
188
+ opts.onFallback?.(fallbackFrom, candidate, fallbackError ?? "auditor fallback", fallbackDelayMs);
189
+ fallbackFrom = undefined;
190
+ fallbackError = undefined;
191
+ fallbackDelayMs = 0;
192
+ }
193
+
194
+ pendingResult = undefined;
195
+ const first = await run(candidate);
196
+ if (first.approved || first.disapproved || first.impossible || !first.error) {
197
+ return { result: first, retriedOnce, fallbackUsed, via: candidate.via };
198
+ }
199
+ let failure = classifyMainModelFailure(first.error);
200
+ if (!isRetriableInfraError(first.error) || !isMainModelFallbackFailure(failure)) {
201
+ return { result: first, retriedOnce, fallbackUsed, via: candidate.via };
202
+ }
203
+ failureAttempt += 1;
204
+ if (!isLive()) return { result: first, retriedOnce, fallbackUsed, via: candidate.via };
205
+ const retryDelayMs = mainModelFailureDelayMs(failure, failureAttempt, opts.retryBaseMinutes ?? 15);
206
+ opts.onRetry?.(candidate, first.error, retryDelayMs);
207
+ await sleep(retryDelayMs);
208
+ if (!isLive()) return { result: first, retriedOnce, fallbackUsed, via: candidate.via };
209
+
210
+ const second = await run(candidate);
211
+ pendingResult = second;
212
+ retriedOnce = true;
213
+ if (second.approved || second.disapproved || second.impossible || !second.error) {
214
+ return { result: second, retriedOnce, fallbackUsed, via: candidate.via };
215
+ }
216
+ failure = classifyMainModelFailure(second.error);
217
+ if (!isRetriableInfraError(second.error) || !isMainModelFallbackFailure(failure)) {
218
+ return { result: second, retriedOnce, fallbackUsed, via: candidate.via };
219
+ }
220
+ currentRef = selectedRef;
221
+ const nextRef = nextUntriedModelRef(currentRef, refs, attempted);
222
+ if (nextRef === undefined) {
223
+ return { result: second, retriedOnce, fallbackUsed, via: candidate.via };
224
+ }
225
+ failureAttempt += 1;
226
+ fallbackDelayMs = mainModelFailureDelayMs(failure, failureAttempt, opts.retryBaseMinutes ?? 15);
227
+ if (!isLive()) return { result: second, retriedOnce, fallbackUsed, via: candidate.via };
228
+ await sleep(fallbackDelayMs);
229
+ if (!isLive()) return { result: second, retriedOnce, fallbackUsed, via: candidate.via };
230
+ fallbackFrom = candidate;
231
+ fallbackError = second.error;
232
+ }
233
+ }
234
+
66
235
  // The detached auditor intentionally exposes the full inspection/tooling
67
236
  // surface, including bash, so it can run bounded tests and reproduce behavior.
68
237
  // This is a power-oriented mode, not a read-only security boundary; callers
@@ -698,11 +867,23 @@ export async function runDetachedGoalCompletionAuditor(args: {
698
867
  await writeAtomicJson(lockPath, { protocolVersion: PROTOCOL_VERSION, attemptId, pid: child.pid, role: "worker", workerPath: workerPathIdentity });
699
868
  child.unref();
700
869
 
701
- const abort = () => { if (child && childAlive(child)) void terminateWorker(child).catch(() => {}); };
870
+ let abortTermination: Promise<void> | null = null;
871
+ const abort = () => {
872
+ if (child && childAlive(child)) {
873
+ abortTermination ??= terminateWorker(child).catch(() => {});
874
+ }
875
+ };
702
876
  args.signal?.addEventListener("abort", abort, { once: true });
703
877
  try {
704
878
  while (true) {
705
- if (args.signal?.aborted) return infra(model, thinkingLevel, "Auditor aborted.", "", capturedRevisionToken, "transport");
879
+ if (args.signal?.aborted) {
880
+ // Do not return the transport result until the detached worker's
881
+ // TERM→KILL teardown has settled. Returning first races the caller's
882
+ // cleanup with a TERM-ignoring worker and leaves its PID alive.
883
+ if (child && childAlive(child)) await terminateWorker(child).catch(() => {});
884
+ else if (abortTermination) await abortTermination;
885
+ return infra(model, thinkingLevel, "Auditor aborted.", "", capturedRevisionToken, "transport");
886
+ }
706
887
  if (childSpawnError) return infra(model, thinkingLevel, `auditor worker launch failed: ${childSpawnError}`, "", capturedRevisionToken, "transport");
707
888
  if (now() >= wallDeadlineAt) {
708
889
  if (childAlive(child)) await terminateWorker(child);