@prestyj/cli 5.16.2 → 5.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/dist/app-sidecar.js +27 -7
  2. package/dist/app-sidecar.js.map +1 -1
  3. package/dist/core/agent-session.d.ts +34 -0
  4. package/dist/core/agent-session.d.ts.map +1 -1
  5. package/dist/core/agent-session.js +356 -29
  6. package/dist/core/agent-session.js.map +1 -1
  7. package/dist/core/autopilot-cycle.d.ts +7 -3
  8. package/dist/core/autopilot-cycle.d.ts.map +1 -1
  9. package/dist/core/autopilot-cycle.js +43 -6
  10. package/dist/core/autopilot-cycle.js.map +1 -1
  11. package/dist/core/autopilot-verdict.d.ts +2 -0
  12. package/dist/core/autopilot-verdict.d.ts.map +1 -1
  13. package/dist/core/autopilot-verdict.js +25 -0
  14. package/dist/core/autopilot-verdict.js.map +1 -1
  15. package/dist/core/event-bus.d.ts +1 -0
  16. package/dist/core/event-bus.d.ts.map +1 -1
  17. package/dist/core/event-bus.js.map +1 -1
  18. package/dist/core/ideal-review-subagent.d.ts +43 -0
  19. package/dist/core/ideal-review-subagent.d.ts.map +1 -0
  20. package/dist/core/ideal-review-subagent.js +95 -0
  21. package/dist/core/ideal-review-subagent.js.map +1 -0
  22. package/dist/core/ideal-review.d.ts.map +1 -1
  23. package/dist/core/ideal-review.js +18 -5
  24. package/dist/core/ideal-review.js.map +1 -1
  25. package/dist/core/nolan-prompt.js +32 -14
  26. package/dist/core/nolan-prompt.js.map +1 -1
  27. package/dist/core/persistent-shell.d.ts.map +1 -1
  28. package/dist/core/persistent-shell.js +3 -1
  29. package/dist/core/persistent-shell.js.map +1 -1
  30. package/dist/core/semantic-loop-check.d.ts +59 -0
  31. package/dist/core/semantic-loop-check.d.ts.map +1 -0
  32. package/dist/core/semantic-loop-check.js +103 -0
  33. package/dist/core/semantic-loop-check.js.map +1 -0
  34. package/dist/core/session-manager.d.ts +1 -1
  35. package/dist/core/session-manager.d.ts.map +1 -1
  36. package/dist/core/shell.d.ts.map +1 -1
  37. package/dist/core/shell.js +6 -3
  38. package/dist/core/shell.js.map +1 -1
  39. package/dist/core/subagent-manager.d.ts +7 -1
  40. package/dist/core/subagent-manager.d.ts.map +1 -1
  41. package/dist/core/subagent-manager.js +15 -5
  42. package/dist/core/subagent-manager.js.map +1 -1
  43. package/dist/core/verification-evidence.d.ts +7 -1
  44. package/dist/core/verification-evidence.d.ts.map +1 -1
  45. package/dist/core/verification-evidence.js +70 -20
  46. package/dist/core/verification-evidence.js.map +1 -1
  47. package/dist/core/verification-gate.d.ts +58 -18
  48. package/dist/core/verification-gate.d.ts.map +1 -1
  49. package/dist/core/verification-gate.js +256 -23
  50. package/dist/core/verification-gate.js.map +1 -1
  51. package/dist/system-prompt.d.ts.map +1 -1
  52. package/dist/system-prompt.js +4 -6
  53. package/dist/system-prompt.js.map +1 -1
  54. package/dist/tools/bash.d.ts.map +1 -1
  55. package/dist/tools/bash.js +3 -1
  56. package/dist/tools/bash.js.map +1 -1
  57. package/dist/utils/github-ci.d.ts +19 -0
  58. package/dist/utils/github-ci.d.ts.map +1 -0
  59. package/dist/utils/github-ci.js +137 -0
  60. package/dist/utils/github-ci.js.map +1 -0
  61. package/package.json +5 -5
@@ -1,5 +1,5 @@
1
1
  import { agentLoop, isAbortError, isUsageLimitError, } from "@prestyj/agent";
2
- import { ProviderError, } from "@prestyj/ai";
2
+ import { ProviderError, stream, } from "@prestyj/ai";
3
3
  import { EventBus } from "./event-bus.js";
4
4
  import { SlashCommandRegistry, createBuiltinCommands, } from "./slash-commands.js";
5
5
  import { PROMPT_COMMANDS, getPromptCommand } from "./prompt-commands.js";
@@ -22,7 +22,7 @@ import { buildSubAgentSystemPrompt, buildSystemPrompt, } from "../system-prompt.
22
22
  import { createTools, createWebSearchTool, } from "../tools/index.js";
23
23
  import { partitionToolsByTier } from "../tools/tool-tiers.js";
24
24
  import { buildProcessCompletionFollowUp } from "./process-gate.js";
25
- import { buildSubAgentCompletionFollowUp } from "./subagent-manager.js";
25
+ import { buildSubAgentCompletionFollowUp, } from "./subagent-manager.js";
26
26
  import { applyAsyncSubagentPolicy } from "./subagent-policy.js";
27
27
  import { z } from "zod";
28
28
  import { MCPClientManager, getAllMcpServers } from "./mcp/index.js";
@@ -42,10 +42,13 @@ import { detectProjectStack } from "./language-detector.js";
42
42
  import { evaluateIdealReview, buildIdealReviewMessage, buildReviewCoverageEscalationMessage, buildReviewCoverageMessage, MAX_REVIEW_COVERAGE_INJECTIONS, withReviewCoverageRequirements, detectTestDrift, ReviewCoverageTracker, } from "./ideal-review.js";
43
43
  import { evaluateLoopBreak, buildLoopBreakMessage, CycleDetector, ToolCallProgressTracker, detectTextRepetition, } from "./loop-breaker.js";
44
44
  import { buildRegroundingMessage } from "./regrounding.js";
45
+ import { buildSemanticLoopJudgePrompt, buildSemanticLoopMessage, MAX_SEMANTIC_LOOP_CALLS, parseSemanticLoopVerdict, shouldRunSemanticLoopCheck, SEMANTIC_LOOP_JUDGE_TIMEOUT_MS, withJudgeTimeout, } from "./semantic-loop-check.js";
46
+ import { buildIndependentReviewMessage, buildReviewerTask, INDEPENDENT_REVIEW_SCORE_THRESHOLD, parseReviewerFindings, REVIEWER_TOOLS, REVIEWER_WAIT_MS, } from "./ideal-review-subagent.js";
45
47
  import { buildEnvDeltaMessage } from "./env-delta.js";
46
48
  import { wrapSteeringText, buildNotificationSteeringText, STEERING_PREFIX } from "./steering.js";
47
49
  import { AgentNotificationQueue } from "./agent-notifications.js";
48
- import { VerificationGate, extractAddedLines, isCheckOwnFile, isCodeFilePath, isVerificationCommand, } from "./verification-gate.js";
50
+ import { VerificationGate, extractAddedLines, isCheckOwnFile, isCodeFilePath, VERIFICATION_STATE_KIND, isVerificationCommand, } from "./verification-gate.js";
51
+ import { classifyVerificationCommand } from "./verification-evidence.js";
49
52
  import { findUserSessionPrompt, getUserSessionPrompt } from "./session-preview.js";
50
53
  import { normalizeMessageImages } from "./message-images.js";
51
54
  import crypto from "node:crypto";
@@ -57,6 +60,14 @@ import path from "node:path";
57
60
  * progressing — refuse to extend its turn budget.
58
61
  */
59
62
  const TURN_EXTENSION_MAX_FAILURE_RATIO = 0.5;
63
+ /** Terminal subagent states — mirrors SubAgentManager's private isTerminal. */
64
+ function isTerminalSubAgentState(state) {
65
+ return (state === "completed" ||
66
+ state === "failed" ||
67
+ state === "interrupted" ||
68
+ state === "closed" ||
69
+ state === "reaped");
70
+ }
60
71
  // ── Tool-result policy ─────────────────────────────────────
61
72
  /** Resolve the per-result cap passed to the agent loop for the active transport. */
62
73
  export function resolveSessionToolResultCharLimit(model, provider, accountId) {
@@ -140,6 +151,7 @@ export class AgentSession {
140
151
  hookCyclicPattern = null;
141
152
  hookFileEditCounts = new Map();
142
153
  hookToolCalls = new Map();
154
+ backgroundVerification = new Map();
143
155
  idealReviewPhase = "idle";
144
156
  /** Runtime-only suppression while Nolan owns verification in autopilot mode. */
145
157
  idealReviewSuppressed = false;
@@ -157,6 +169,14 @@ export class AgentSession {
157
169
  /** 0 = none; 1 = first nudge sent; 2 = final stop-and-report injected. */
158
170
  loopBreakInjected = 0;
159
171
  regroundingInjected = false;
172
+ /** Recent tool-call digests for the semantic loop judge — bounded ring. */
173
+ hookRecentCalls = [];
174
+ /** LLM-judged loop detection state. `verdict` holds a LOOP verdict awaiting
175
+ * injection at the next steering poll; judge failures fail open (no
176
+ * injection) and still consume budget + cooldown. */
177
+ semanticLoop = { checksUsed: 0, lastCheckTurn: 0, pending: false, verdict: null, injected: false };
178
+ /** Independent Ideal reviewer spawned once per run (score-gated). */
179
+ independentReviewStarted = false;
160
180
  /**
161
181
  * The environment as the cached system prompt currently describes it.
162
182
  * Re-recorded on every prompt build, so a rebuild (e.g. `/add-dir`) needs no
@@ -997,9 +1017,23 @@ export class AgentSession {
997
1017
  this.idealDriftProbe = null;
998
1018
  this.loopBreakInjected = 0;
999
1019
  this.regroundingInjected = false;
1020
+ this.hookRecentCalls = [];
1021
+ this.semanticLoop = {
1022
+ checksUsed: 0,
1023
+ lastCheckTurn: 0,
1024
+ pending: false,
1025
+ verdict: null,
1026
+ injected: false,
1027
+ };
1028
+ this.independentReviewStarted = false;
1000
1029
  this.runStartedAt = Date.now();
1001
1030
  this.processGateInjected = 0;
1002
- this.verificationGate.reset();
1031
+ this.verificationGate.beginRun();
1032
+ const processes = new Set(this.processManager?.list().map((p) => p.id) ?? []);
1033
+ for (const id of this.backgroundVerification.keys()) {
1034
+ if (!processes.has(id))
1035
+ this.backgroundVerification.delete(id);
1036
+ }
1003
1037
  this.compactionOccurred = false;
1004
1038
  this.originalRequest = originalRequest;
1005
1039
  }
@@ -1014,7 +1048,25 @@ export class AgentSession {
1014
1048
  this.hookText += event.text;
1015
1049
  break;
1016
1050
  case "tool_call_start":
1017
- this.hookToolCalls.set(event.toolCallId, { name: event.name, args: event.args ?? {} });
1051
+ this.hookToolCalls.set(event.toolCallId, {
1052
+ name: event.name,
1053
+ args: event.args ?? {},
1054
+ revision: this.verificationGate.revision,
1055
+ });
1056
+ if (event.name === "bash" &&
1057
+ typeof event.args?.command === "string" &&
1058
+ (isVerificationCommand(event.args.command) ||
1059
+ classifyVerificationCommand(event.args.command).accepted)) {
1060
+ // A check that can rewrite files (--fix, build scripts, emitters)
1061
+ // invalidates earlier in-flight evidence AND marks the run as
1062
+ // touched. A check that is merely UNRECOGNIZED (`make test`, `deno
1063
+ // test`) rewrites nothing we can point to: bumping the revision for
1064
+ // it poisoned the gate on green output and re-armed the hook into
1065
+ // every later question turn.
1066
+ const classification = classifyVerificationCommand(event.args.command);
1067
+ this.verificationGate.requireFreshVerification(!classification.accepted && classification.mayMutate, event.args.command);
1068
+ await this.persistVerificationState();
1069
+ }
1018
1070
  break;
1019
1071
  case "tool_call_end": {
1020
1072
  const call = this.hookToolCalls.get(event.toolCallId);
@@ -1032,14 +1084,28 @@ export class AgentSession {
1032
1084
  this.hookConsecutiveFailures = event.isError ? this.hookConsecutiveFailures + 1 : 0;
1033
1085
  this.hookRepeatedNoProgressCalls = this.hookProgressTracker.record(name, args, event.result, event.isError);
1034
1086
  this.hookCyclicPattern = this.hookCycleDetector.record(name, args, event.result, event.isError);
1087
+ // Semantic-loop judge input: a bounded digest of WHAT was attempted and
1088
+ // HOW it came out. Args/results are sliced AT RECORD TIME — a write with
1089
+ // a 50 KiB payload or a bash dump must never inflate the ring, and the
1090
+ // judge needs shapes, not payloads.
1091
+ this.hookRecentCalls.push({
1092
+ tool: name,
1093
+ args: args === undefined ? "" : JSON.stringify(args).slice(0, 300),
1094
+ ok: !event.isError,
1095
+ result: event.result.slice(0, 400),
1096
+ });
1097
+ if (this.hookRecentCalls.length > MAX_SEMANTIC_LOOP_CALLS) {
1098
+ this.hookRecentCalls.splice(0, this.hookRecentCalls.length - MAX_SEMANTIC_LOOP_CALLS);
1099
+ }
1035
1100
  if (name === "edit" && !event.isError) {
1036
1101
  const diff = event.details?.diff ?? event.result;
1037
1102
  const added = (diff.match(/^\+[^+]/gm) ?? []).length;
1038
1103
  const removed = (diff.match(/^-[^-]/gm) ?? []).length;
1039
1104
  this.hookStats.changedLines += added + removed;
1040
1105
  }
1041
- // Verification-gate bookkeeping: successful code mutations and completed
1042
- // foreground verification commands, in occurrence order.
1106
+ // Only host-observed successful mutations and trustworthy check results
1107
+ // affect approval. The model's text is never evidence.
1108
+ let verificationChanged = false;
1043
1109
  if (!event.isError && args) {
1044
1110
  if (name === "edit" || name === "write") {
1045
1111
  const filePath = String(args.file_path ?? "");
@@ -1051,26 +1117,59 @@ export class AgentSession {
1051
1117
  ? String(args.content ?? "")
1052
1118
  : extractAddedLines(event.details?.diff ?? event.result);
1053
1119
  this.verificationGate.recordMutation(filePath, addedText);
1120
+ verificationChanged = true;
1121
+ }
1122
+ }
1123
+ }
1124
+ if (args && call && name === "bash") {
1125
+ const command = typeof args.command === "string" ? args.command : "";
1126
+ const classification = classifyVerificationCommand(command);
1127
+ if (classification.accepted) {
1128
+ if (args.run_in_background === true && !event.isError && args.persist !== true) {
1129
+ const id = /^ID:\s*(\S+)/m.exec(event.result)?.[1];
1130
+ // No parseable ID means the check cannot be tracked to a real exit
1131
+ // code — no evidence either way. Recording a FAILURE here made
1132
+ // every later green run of a different spelling look owed.
1133
+ if (id)
1134
+ this.backgroundVerification.set(id, { revision: call.revision, command });
1135
+ }
1136
+ else if (args.persist === true) {
1137
+ // Persistent-shell checks are not bounded evidence (steering can
1138
+ // interleave): neither a pass nor a failure. A recorded failure
1139
+ // here blocked approval for sessions that prefer the shell.
1140
+ }
1141
+ else {
1142
+ if (!event.isError && /^Exit code:\s*0(?:\s|$)/i.test(event.result.trim())) {
1143
+ this.verificationGate.recordVerification(call.revision, command);
1144
+ }
1145
+ else {
1146
+ this.verificationGate.recordFailedVerification(command, call.revision);
1147
+ }
1148
+ verificationChanged = true;
1054
1149
  }
1055
1150
  }
1056
- if (name === "bash" &&
1057
- !args.run_in_background &&
1058
- isVerificationCommand(String(args.command ?? ""))) {
1059
- this.verificationGate.recordVerification();
1151
+ else if (classification.candidate) {
1152
+ // Green but untrusted: remember WHY so the demand can tell the
1153
+ // agent which command shape actually clears the gate.
1154
+ this.verificationGate.recordRejectedCheck(command, classification.reason);
1060
1155
  }
1061
- // Reading the final output of an EXITED background verification run
1062
- // counts: the process gate forces this read anyway, so without it the
1063
- // gate would demand a redundant foreground re-run of tests the agent
1064
- // already watched finish.
1065
- if (name === "task_output") {
1066
- const proc = this.processManager
1067
- ?.list()
1068
- .find((p) => p.id === args.id);
1069
- if (proc && proc.exitCode !== null && isVerificationCommand(proc.command)) {
1070
- this.verificationGate.recordVerification();
1156
+ }
1157
+ if (!event.isError && args && name === "task_output" && typeof args.id === "string") {
1158
+ const started = this.backgroundVerification.get(args.id);
1159
+ const proc = this.processManager?.list().find((p) => p.id === args.id);
1160
+ if (started && proc && proc.exitCode !== null) {
1161
+ if (proc.exitCode === 0) {
1162
+ this.verificationGate.recordVerification(started.revision, started.command);
1163
+ }
1164
+ else {
1165
+ this.verificationGate.recordFailedVerification(started.command, started.revision);
1071
1166
  }
1167
+ this.backgroundVerification.delete(args.id);
1168
+ verificationChanged = true;
1072
1169
  }
1073
1170
  }
1171
+ if (verificationChanged)
1172
+ await this.persistVerificationState();
1074
1173
  // Tool results are what push the run over the review gate, and they all
1075
1174
  // land before the model writes its candidate final answer — so this is
1076
1175
  // the point where arming still beats the draft's first token.
@@ -1219,16 +1318,21 @@ export class AgentSession {
1219
1318
  return null;
1220
1319
  if (!this.settingsManager.get("idealReviewEnabled"))
1221
1320
  return null;
1321
+ // Deterministic stuck verdict, computed once and shared: the semantic
1322
+ // judge must not spend tokens on a burst the deterministic breaker is
1323
+ // about to correct itself.
1324
+ const deterministicDecision = evaluateLoopBreak({
1325
+ consecutiveFailures: this.hookConsecutiveFailures,
1326
+ repeatedNoProgressCalls: this.hookRepeatedNoProgressCalls,
1327
+ textRepetitionDetected: detectTextRepetition(this.hookText),
1328
+ ...(this.hookCyclicPattern ? { cyclicPattern: this.hookCyclicPattern } : {}),
1329
+ });
1330
+ this.maybeStartSemanticLoopCheck(deterministicDecision.shouldBreak);
1222
1331
  // Two-stage loop-breaker: stage 1 nudges; a FRESH detection after that
1223
1332
  // injects the harsher final stop-and-report prompt. Signals reset after
1224
1333
  // each injection so stage 2 only fires on new evidence.
1225
1334
  if (this.loopBreakInjected < 2) {
1226
- const decision = evaluateLoopBreak({
1227
- consecutiveFailures: this.hookConsecutiveFailures,
1228
- repeatedNoProgressCalls: this.hookRepeatedNoProgressCalls,
1229
- textRepetitionDetected: detectTextRepetition(this.hookText),
1230
- ...(this.hookCyclicPattern ? { cyclicPattern: this.hookCyclicPattern } : {}),
1231
- });
1335
+ const decision = deterministicDecision;
1232
1336
  if (decision.shouldBreak) {
1233
1337
  const stage = this.loopBreakInjected === 0 ? 1 : 2;
1234
1338
  this.loopBreakInjected = stage;
@@ -1237,6 +1341,9 @@ export class AgentSession {
1237
1341
  this.hookCyclicPattern = null;
1238
1342
  this.hookConsecutiveFailures = 0;
1239
1343
  this.hookRepeatedNoProgressCalls = 0;
1344
+ // The deterministic breaker owns this burst — a semantic verdict from
1345
+ // the same burst must not double-correct on the next poll.
1346
+ this.semanticLoop.verdict = null;
1240
1347
  // Clear the text buffer too — otherwise a stage-1 text-repetition
1241
1348
  // trigger still sees the same repeated tail on the next check and
1242
1349
  // escalates to stage 2 on stale evidence.
@@ -1249,6 +1356,22 @@ export class AgentSession {
1249
1356
  return [buildLoopBreakMessage(decision.reasons, stage === 2)];
1250
1357
  }
1251
1358
  }
1359
+ // Semantic loop-break: an LLM verdict (started by maybeStartSemanticLoopCheck
1360
+ // on a suspicious-but-syntactically-quiet burst) is consumed here, exactly
1361
+ // once, with the deterministic breaker getting priority above. Fail-open:
1362
+ // no verdict or no-loop verdict injects nothing.
1363
+ if (this.semanticLoop.verdict && !this.semanticLoop.injected) {
1364
+ const verdict = this.semanticLoop.verdict;
1365
+ this.semanticLoop.injected = true;
1366
+ this.semanticLoop.verdict = null;
1367
+ // The judged burst has been addressed; a fresh burst must re-accumulate.
1368
+ this.hookConsecutiveFailures = 0;
1369
+ log("INFO", "loop-break", "Injecting semantic loop-break steering", {
1370
+ reason: verdict.reason,
1371
+ });
1372
+ this.eventBus.emit("hook", { kind: "loop_break" });
1373
+ return [buildSemanticLoopMessage(verdict)];
1374
+ }
1252
1375
  if (!this.regroundingInjected && this.compactionOccurred) {
1253
1376
  this.regroundingInjected = true;
1254
1377
  this.eventBus.emit("hook", { kind: "regrounding" });
@@ -1256,6 +1379,159 @@ export class AgentSession {
1256
1379
  }
1257
1380
  return null;
1258
1381
  }
1382
+ /** Fire the semantic loop judge when deterministic evidence is suspicious but
1383
+ * the deterministic breaker stayed quiet — the syntactic blind spot where
1384
+ * every retry differs slightly and the run still makes no progress.
1385
+ * Fire-and-forget: the call runs while the next turn streams, and a finished
1386
+ * verdict is consumed by the NEXT steering poll — never blocking a turn. */
1387
+ maybeStartSemanticLoopCheck(deterministicBreak) {
1388
+ if (!shouldRunSemanticLoopCheck({
1389
+ consecutiveFailures: this.hookConsecutiveFailures,
1390
+ totalFailures: this.hookStats.toolFailures,
1391
+ turns: this.hookStats.turns,
1392
+ lastCheckTurn: this.semanticLoop.lastCheckTurn,
1393
+ checksUsed: this.semanticLoop.checksUsed,
1394
+ checkPending: this.semanticLoop.pending,
1395
+ deterministicBreak,
1396
+ })) {
1397
+ return;
1398
+ }
1399
+ // resetHookState replaces this object on every prompt. A late judge must
1400
+ // not publish a verdict or consume the next run's budget/cooldown.
1401
+ const runState = this.semanticLoop;
1402
+ runState.pending = true;
1403
+ log("INFO", "loop-break", "Starting semantic loop judge", {
1404
+ turn: String(this.hookStats.turns),
1405
+ consecutiveFailures: String(this.hookConsecutiveFailures),
1406
+ recentCalls: String(this.hookRecentCalls.length),
1407
+ });
1408
+ void (async () => {
1409
+ try {
1410
+ const prompt = buildSemanticLoopJudgePrompt(this.hookRecentCalls, this.originalRequest);
1411
+ const raw = await (this.opts.semanticLoopJudge?.(prompt) ??
1412
+ this.callSemanticLoopJudge(prompt));
1413
+ const verdict = parseSemanticLoopVerdict(raw);
1414
+ if (this.semanticLoop === runState && verdict?.loop)
1415
+ runState.verdict = verdict;
1416
+ }
1417
+ catch (error) {
1418
+ // Fail open: judge errors never stop a run. Budget and cooldown are
1419
+ // still consumed in `finally` so a flaky judge cannot retry-loop.
1420
+ log("WARN", "loop-break", "Semantic loop judge failed", {
1421
+ error: error instanceof Error ? error.message : String(error),
1422
+ });
1423
+ }
1424
+ finally {
1425
+ if (this.semanticLoop === runState) {
1426
+ runState.pending = false;
1427
+ runState.checksUsed += 1;
1428
+ runState.lastCheckTurn = this.hookStats.turns;
1429
+ }
1430
+ }
1431
+ })();
1432
+ }
1433
+ /** One-shot judge call on the session's ACTIVE model — deliberately not a
1434
+ * cheaper routing: judging a model's own failure patterns with a weaker
1435
+ * model swaps false negatives for false positives. */
1436
+ async callSemanticLoopJudge(prompt) {
1437
+ const creds = await this.authStorage.resolveCredentials(this.provider, {
1438
+ storageKeys: this.currentAuthStorageKeys(),
1439
+ });
1440
+ const result = stream({
1441
+ provider: this.provider,
1442
+ model: this.model,
1443
+ messages: [{ role: "user", content: prompt }],
1444
+ maxTokens: 500,
1445
+ apiKey: creds.accessToken,
1446
+ accountId: creds.accountId,
1447
+ projectId: creds.projectId,
1448
+ baseUrl: this.baseUrl ?? creds.baseUrl,
1449
+ signal: this.opts.signal,
1450
+ });
1451
+ const response = await withJudgeTimeout(result.response, SEMANTIC_LOOP_JUDGE_TIMEOUT_MS);
1452
+ // Providers differ in reply shape: some return a bare string, others an
1453
+ // array of parts (glm-5.3 among them) — joining text parts covers both,
1454
+ // where the string-only branch silently dropped the whole verdict.
1455
+ const content = response.message.content;
1456
+ if (typeof content === "string")
1457
+ return content;
1458
+ return Array.isArray(content)
1459
+ ? content
1460
+ .filter((part) => part.type === "text")
1461
+ .map((part) => part.text)
1462
+ .join("\n")
1463
+ : "";
1464
+ }
1465
+ /** Independent fresh-context review of the finished work (Codex Guardian
1466
+ * pattern). Spawns a READ-ONLY child on the ACTIVE model, waits bounded,
1467
+ * and returns findings for the acting agent to address — or nothing when
1468
+ * the review passes, is unavailable, or fails (in-thread review remains the
1469
+ * fallback; the feature degrades, never blocks).
1470
+ *
1471
+ * Runs inside the pre-stop poll, so the candidate final answer is already
1472
+ * held by arming and this wait cannot race a streamed answer. */
1473
+ async runIndependentReview(decision) {
1474
+ if (!this.subAgentManager)
1475
+ return [];
1476
+ if (this.independentReviewStarted)
1477
+ return [];
1478
+ // An allow-listed session (a subagent worker itself) must not spawn
1479
+ // harness-owned grandchildren the tool policy never granted.
1480
+ if (this.opts.allowedTools && !this.opts.allowedTools.includes("spawn_agent"))
1481
+ return [];
1482
+ if (decision.score < INDEPENDENT_REVIEW_SCORE_THRESHOLD)
1483
+ return [];
1484
+ this.independentReviewStarted = true;
1485
+ const taskName = `ideal-reviewer-${Math.random().toString(36).slice(2, 8)}`;
1486
+ let agentId;
1487
+ try {
1488
+ const task = buildReviewerTask({
1489
+ originalRequest: this.originalRequest,
1490
+ changedFiles: [...this.hookFileEditCounts.keys()],
1491
+ stats: this.hookStats,
1492
+ triggerReasons: decision.reasons,
1493
+ });
1494
+ // Active model forced at spawn time — never routed to a fast/review model.
1495
+ const snapshot = await this.subAgentManager.spawn(taskName, task, undefined, {
1496
+ model: this.model,
1497
+ tools: REVIEWER_TOOLS,
1498
+ });
1499
+ agentId = snapshot.agent_id;
1500
+ const waited = await this.subAgentManager.wait([agentId], "all", REVIEWER_WAIT_MS);
1501
+ const agent = waited.agents[0];
1502
+ if (!agent || !isTerminalSubAgentState(agent.state)) {
1503
+ // Timeout: collect the straggler so the completion gate cannot fire on
1504
+ // it later, then fall back to the in-thread review.
1505
+ await this.subAgentManager.interrupt(agentId, true).catch(() => { });
1506
+ log("WARN", "ideal", "Independent reviewer timed out; falling back to in-thread review", {
1507
+ agentId,
1508
+ });
1509
+ return [];
1510
+ }
1511
+ const findings = parseReviewerFindings(agent.output ?? "");
1512
+ if (!findings) {
1513
+ log("WARN", "ideal", "Independent reviewer output unparseable; falling back", { agentId });
1514
+ return [];
1515
+ }
1516
+ if (findings.clean) {
1517
+ log("INFO", "ideal", "Independent reviewer verdict: clean", { agentId });
1518
+ return [];
1519
+ }
1520
+ log("INFO", "ideal", "Independent reviewer flagged findings", {
1521
+ agentId,
1522
+ count: String(findings.findings.length),
1523
+ });
1524
+ return [buildIndependentReviewMessage(findings.findings)];
1525
+ }
1526
+ catch (error) {
1527
+ if (agentId)
1528
+ await this.subAgentManager.interrupt(agentId, true).catch(() => { });
1529
+ log("WARN", "ideal", "Independent reviewer failed; falling back to in-thread review", {
1530
+ error: error instanceof Error ? error.message : String(error),
1531
+ });
1532
+ return [];
1533
+ }
1534
+ }
1259
1535
  /**
1260
1536
  * Turn-budget extension gate. The loop consults this instead of stopping
1261
1537
  * mid-task when it exhausts `maxTurns`. Grant ONLY on evidence of progress —
@@ -1373,7 +1649,7 @@ export class AgentSession {
1373
1649
  * Pre-stop Ideal review phase machine. Once review starts, completion is
1374
1650
  * blocked until harness-owned post-injection reads cover every changed file.
1375
1651
  */
1376
- getHookFollowUpMessages() {
1652
+ async getHookFollowUpMessages() {
1377
1653
  const childCompletionFollowUp = buildSubAgentCompletionFollowUp(this.subAgentManager);
1378
1654
  if (childCompletionFollowUp)
1379
1655
  return childCompletionFollowUp;
@@ -1394,13 +1670,17 @@ export class AgentSession {
1394
1670
  if (this.opts.selfCorrectionHooks !== false &&
1395
1671
  this.settingsManager.get("verificationGateEnabled") &&
1396
1672
  (!this.opts.allowedTools || this.opts.allowedTools.includes("bash"))) {
1673
+ const verificationReason = this.verificationGate.pendingReason();
1397
1674
  const verificationFollowUp = this.verificationGate.followUp();
1398
1675
  if (verificationFollowUp) {
1399
1676
  log("INFO", "verification-gate", "Injecting verification follow-up", {});
1400
1677
  // Announce, THEN disarm: clients release held text on disarm, so the
1401
1678
  // reverse order paints the draft and immediately deletes it — the exact
1402
1679
  // flash arming exists to prevent.
1403
- this.eventBus.emit("hook", { kind: "verification" });
1680
+ this.eventBus.emit("hook", {
1681
+ kind: "verification",
1682
+ ...(verificationReason === "recheck" ? { verificationReason } : {}),
1683
+ });
1404
1684
  this.refreshHookArming();
1405
1685
  return verificationFollowUp;
1406
1686
  }
@@ -1460,6 +1740,10 @@ export class AgentSession {
1460
1740
  const driftedFiles = detectTestDrift(this.hookFileEditCounts.keys(), this.cwd).slice(0, 5);
1461
1741
  if (!decision.shouldReview && driftedFiles.length === 0)
1462
1742
  return null;
1743
+ // Independent reviewer first (async, bounded): its findings ride in the
1744
+ // SAME follow-up batch as the in-thread review + coverage requirements, so
1745
+ // addressing everything still costs one extra turn.
1746
+ const independentMessages = await this.runIndependentReview(decision);
1463
1747
  this.reviewCoverage.start(this.hookFileEditCounts.keys());
1464
1748
  this.idealReviewPhase = "reviewing";
1465
1749
  const coverage = this.reviewCoverage.evidence();
@@ -1483,6 +1767,7 @@ export class AgentSession {
1483
1767
  lspMissing: lspEvidence.missing,
1484
1768
  });
1485
1769
  return [
1770
+ ...independentMessages,
1486
1771
  this.withReviewLspEvidence(withReviewCoverageRequirements(buildIdealReviewMessage(decision.reasons, driftedFiles), coverage.missing), lspEvidence),
1487
1772
  ];
1488
1773
  }
@@ -2083,6 +2368,7 @@ export class AgentSession {
2083
2368
  await this.rePersistNolanTurns();
2084
2369
  await this.rePersistAutopilotMarkers();
2085
2370
  await this.rePersistAppMarkers();
2371
+ await this.persistVerificationState();
2086
2372
  await this.persistAppMarker("compaction", {
2087
2373
  originalCount: result.originalCount,
2088
2374
  newCount: result.newCount,
@@ -2239,6 +2525,8 @@ export class AgentSession {
2239
2525
  // Approved-plan execution is a clean checkpoint of the same conversation;
2240
2526
  // explicit new sessions reset the conversation identity.
2241
2527
  if (!preserveConversation) {
2528
+ this.verificationGate.reset();
2529
+ this.backgroundVerification.clear();
2242
2530
  this.conversationId = "";
2243
2531
  this.checkpointGeneration = 0;
2244
2532
  this.sessionPreview = "";
@@ -2778,6 +3066,28 @@ export class AgentSession {
2778
3066
  * instead of dropping the marker or falling back to a raw verdict string.
2779
3067
  * No-op persistence for transient sessions (kept in memory only).
2780
3068
  */
3069
+ getVerificationProblem() {
3070
+ return (this.verificationGate.verificationProblem() ??
3071
+ (this.backgroundVerification.size > 0
3072
+ ? "Unverified: a background check is still running or its result has not been confirmed."
3073
+ : null));
3074
+ }
3075
+ async persistVerificationState() {
3076
+ if (!this.sessionPath)
3077
+ return;
3078
+ const entry = {
3079
+ type: "custom",
3080
+ kind: VERIFICATION_STATE_KIND,
3081
+ id: crypto.randomUUID(),
3082
+ parentId: null,
3083
+ timestamp: new Date().toISOString(),
3084
+ data: {
3085
+ ...this.verificationGate.snapshot(),
3086
+ unknown: this.getVerificationProblem() !== null,
3087
+ },
3088
+ };
3089
+ await this.sessionManager.appendEntry(this.sessionPath, entry);
3090
+ }
2781
3091
  async persistAutopilotMarker(phase, extra) {
2782
3092
  const afterMessageCount = this.persistedTranscriptCount();
2783
3093
  const payload = {
@@ -3019,6 +3329,23 @@ export class AgentSession {
3019
3329
  const loaded = await this.sessionManager.load(canonicalPath);
3020
3330
  // Use the leaf from the header to walk the correct branch
3021
3331
  const loadedMessages = this.sessionManager.getMessages(loaded.entries, loaded.header.leafId);
3332
+ this.backgroundVerification.clear();
3333
+ const savedVerification = [...loaded.entries]
3334
+ .reverse()
3335
+ .find((entry) => entry.type === "custom" && entry.kind === VERIFICATION_STATE_KIND);
3336
+ if (savedVerification?.type === "custom") {
3337
+ this.verificationGate.restore(savedVerification.data);
3338
+ }
3339
+ else {
3340
+ this.verificationGate.reset();
3341
+ // Legacy sessions have no host checkpoint. Tool-authored code history is
3342
+ // not proof of today's files; require fresh evidence before approval.
3343
+ if (loadedMessages.some((message) => message.role === "assistant" &&
3344
+ Array.isArray(message.content) &&
3345
+ message.content.some((part) => part.type === "tool_call" && (part.name === "edit" || part.name === "write")))) {
3346
+ this.verificationGate.requireFreshVerification();
3347
+ }
3348
+ }
3022
3349
  this.checkpointGeneration = loaded.header.generation ?? 0;
3023
3350
  this.conversationId = loaded.header.conversationId ?? loaded.header.id;
3024
3351
  const legacyLabel = [...loaded.entries]