@prestyj/cli 5.16.2 → 5.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/app-sidecar.js +27 -7
- package/dist/app-sidecar.js.map +1 -1
- package/dist/core/agent-session.d.ts +34 -0
- package/dist/core/agent-session.d.ts.map +1 -1
- package/dist/core/agent-session.js +356 -29
- package/dist/core/agent-session.js.map +1 -1
- package/dist/core/autopilot-cycle.d.ts +7 -3
- package/dist/core/autopilot-cycle.d.ts.map +1 -1
- package/dist/core/autopilot-cycle.js +43 -6
- package/dist/core/autopilot-cycle.js.map +1 -1
- package/dist/core/autopilot-verdict.d.ts +2 -0
- package/dist/core/autopilot-verdict.d.ts.map +1 -1
- package/dist/core/autopilot-verdict.js +25 -0
- package/dist/core/autopilot-verdict.js.map +1 -1
- package/dist/core/event-bus.d.ts +1 -0
- package/dist/core/event-bus.d.ts.map +1 -1
- package/dist/core/event-bus.js.map +1 -1
- package/dist/core/ideal-review-subagent.d.ts +43 -0
- package/dist/core/ideal-review-subagent.d.ts.map +1 -0
- package/dist/core/ideal-review-subagent.js +95 -0
- package/dist/core/ideal-review-subagent.js.map +1 -0
- package/dist/core/ideal-review.d.ts.map +1 -1
- package/dist/core/ideal-review.js +18 -5
- package/dist/core/ideal-review.js.map +1 -1
- package/dist/core/nolan-prompt.js +32 -14
- package/dist/core/nolan-prompt.js.map +1 -1
- package/dist/core/persistent-shell.d.ts.map +1 -1
- package/dist/core/persistent-shell.js +3 -1
- package/dist/core/persistent-shell.js.map +1 -1
- package/dist/core/semantic-loop-check.d.ts +59 -0
- package/dist/core/semantic-loop-check.d.ts.map +1 -0
- package/dist/core/semantic-loop-check.js +103 -0
- package/dist/core/semantic-loop-check.js.map +1 -0
- package/dist/core/session-manager.d.ts +1 -1
- package/dist/core/session-manager.d.ts.map +1 -1
- package/dist/core/shell.d.ts.map +1 -1
- package/dist/core/shell.js +6 -3
- package/dist/core/shell.js.map +1 -1
- package/dist/core/subagent-manager.d.ts +7 -1
- package/dist/core/subagent-manager.d.ts.map +1 -1
- package/dist/core/subagent-manager.js +15 -5
- package/dist/core/subagent-manager.js.map +1 -1
- package/dist/core/verification-evidence.d.ts +7 -1
- package/dist/core/verification-evidence.d.ts.map +1 -1
- package/dist/core/verification-evidence.js +70 -20
- package/dist/core/verification-evidence.js.map +1 -1
- package/dist/core/verification-gate.d.ts +58 -18
- package/dist/core/verification-gate.d.ts.map +1 -1
- package/dist/core/verification-gate.js +256 -23
- package/dist/core/verification-gate.js.map +1 -1
- package/dist/system-prompt.d.ts.map +1 -1
- package/dist/system-prompt.js +4 -6
- package/dist/system-prompt.js.map +1 -1
- package/dist/tools/bash.d.ts.map +1 -1
- package/dist/tools/bash.js +3 -1
- package/dist/tools/bash.js.map +1 -1
- package/dist/utils/github-ci.d.ts +19 -0
- package/dist/utils/github-ci.d.ts.map +1 -0
- package/dist/utils/github-ci.js +137 -0
- package/dist/utils/github-ci.js.map +1 -0
- package/package.json +5 -5
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { agentLoop, isAbortError, isUsageLimitError, } from "@prestyj/agent";
|
|
2
|
-
import { ProviderError, } from "@prestyj/ai";
|
|
2
|
+
import { ProviderError, stream, } from "@prestyj/ai";
|
|
3
3
|
import { EventBus } from "./event-bus.js";
|
|
4
4
|
import { SlashCommandRegistry, createBuiltinCommands, } from "./slash-commands.js";
|
|
5
5
|
import { PROMPT_COMMANDS, getPromptCommand } from "./prompt-commands.js";
|
|
@@ -22,7 +22,7 @@ import { buildSubAgentSystemPrompt, buildSystemPrompt, } from "../system-prompt.
|
|
|
22
22
|
import { createTools, createWebSearchTool, } from "../tools/index.js";
|
|
23
23
|
import { partitionToolsByTier } from "../tools/tool-tiers.js";
|
|
24
24
|
import { buildProcessCompletionFollowUp } from "./process-gate.js";
|
|
25
|
-
import { buildSubAgentCompletionFollowUp } from "./subagent-manager.js";
|
|
25
|
+
import { buildSubAgentCompletionFollowUp, } from "./subagent-manager.js";
|
|
26
26
|
import { applyAsyncSubagentPolicy } from "./subagent-policy.js";
|
|
27
27
|
import { z } from "zod";
|
|
28
28
|
import { MCPClientManager, getAllMcpServers } from "./mcp/index.js";
|
|
@@ -42,10 +42,13 @@ import { detectProjectStack } from "./language-detector.js";
|
|
|
42
42
|
import { evaluateIdealReview, buildIdealReviewMessage, buildReviewCoverageEscalationMessage, buildReviewCoverageMessage, MAX_REVIEW_COVERAGE_INJECTIONS, withReviewCoverageRequirements, detectTestDrift, ReviewCoverageTracker, } from "./ideal-review.js";
|
|
43
43
|
import { evaluateLoopBreak, buildLoopBreakMessage, CycleDetector, ToolCallProgressTracker, detectTextRepetition, } from "./loop-breaker.js";
|
|
44
44
|
import { buildRegroundingMessage } from "./regrounding.js";
|
|
45
|
+
import { buildSemanticLoopJudgePrompt, buildSemanticLoopMessage, MAX_SEMANTIC_LOOP_CALLS, parseSemanticLoopVerdict, shouldRunSemanticLoopCheck, SEMANTIC_LOOP_JUDGE_TIMEOUT_MS, withJudgeTimeout, } from "./semantic-loop-check.js";
|
|
46
|
+
import { buildIndependentReviewMessage, buildReviewerTask, INDEPENDENT_REVIEW_SCORE_THRESHOLD, parseReviewerFindings, REVIEWER_TOOLS, REVIEWER_WAIT_MS, } from "./ideal-review-subagent.js";
|
|
45
47
|
import { buildEnvDeltaMessage } from "./env-delta.js";
|
|
46
48
|
import { wrapSteeringText, buildNotificationSteeringText, STEERING_PREFIX } from "./steering.js";
|
|
47
49
|
import { AgentNotificationQueue } from "./agent-notifications.js";
|
|
48
|
-
import { VerificationGate, extractAddedLines, isCheckOwnFile, isCodeFilePath, isVerificationCommand, } from "./verification-gate.js";
|
|
50
|
+
import { VerificationGate, extractAddedLines, isCheckOwnFile, isCodeFilePath, VERIFICATION_STATE_KIND, isVerificationCommand, } from "./verification-gate.js";
|
|
51
|
+
import { classifyVerificationCommand } from "./verification-evidence.js";
|
|
49
52
|
import { findUserSessionPrompt, getUserSessionPrompt } from "./session-preview.js";
|
|
50
53
|
import { normalizeMessageImages } from "./message-images.js";
|
|
51
54
|
import crypto from "node:crypto";
|
|
@@ -57,6 +60,14 @@ import path from "node:path";
|
|
|
57
60
|
* progressing — refuse to extend its turn budget.
|
|
58
61
|
*/
|
|
59
62
|
const TURN_EXTENSION_MAX_FAILURE_RATIO = 0.5;
|
|
63
|
+
/** Terminal subagent states — mirrors SubAgentManager's private isTerminal. */
|
|
64
|
+
function isTerminalSubAgentState(state) {
|
|
65
|
+
return (state === "completed" ||
|
|
66
|
+
state === "failed" ||
|
|
67
|
+
state === "interrupted" ||
|
|
68
|
+
state === "closed" ||
|
|
69
|
+
state === "reaped");
|
|
70
|
+
}
|
|
60
71
|
// ── Tool-result policy ─────────────────────────────────────
|
|
61
72
|
/** Resolve the per-result cap passed to the agent loop for the active transport. */
|
|
62
73
|
export function resolveSessionToolResultCharLimit(model, provider, accountId) {
|
|
@@ -140,6 +151,7 @@ export class AgentSession {
|
|
|
140
151
|
hookCyclicPattern = null;
|
|
141
152
|
hookFileEditCounts = new Map();
|
|
142
153
|
hookToolCalls = new Map();
|
|
154
|
+
backgroundVerification = new Map();
|
|
143
155
|
idealReviewPhase = "idle";
|
|
144
156
|
/** Runtime-only suppression while Nolan owns verification in autopilot mode. */
|
|
145
157
|
idealReviewSuppressed = false;
|
|
@@ -157,6 +169,14 @@ export class AgentSession {
|
|
|
157
169
|
/** 0 = none; 1 = first nudge sent; 2 = final stop-and-report injected. */
|
|
158
170
|
loopBreakInjected = 0;
|
|
159
171
|
regroundingInjected = false;
|
|
172
|
+
/** Recent tool-call digests for the semantic loop judge — bounded ring. */
|
|
173
|
+
hookRecentCalls = [];
|
|
174
|
+
/** LLM-judged loop detection state. `verdict` holds a LOOP verdict awaiting
|
|
175
|
+
* injection at the next steering poll; judge failures fail open (no
|
|
176
|
+
* injection) and still consume budget + cooldown. */
|
|
177
|
+
semanticLoop = { checksUsed: 0, lastCheckTurn: 0, pending: false, verdict: null, injected: false };
|
|
178
|
+
/** Independent Ideal reviewer spawned once per run (score-gated). */
|
|
179
|
+
independentReviewStarted = false;
|
|
160
180
|
/**
|
|
161
181
|
* The environment as the cached system prompt currently describes it.
|
|
162
182
|
* Re-recorded on every prompt build, so a rebuild (e.g. `/add-dir`) needs no
|
|
@@ -997,9 +1017,23 @@ export class AgentSession {
|
|
|
997
1017
|
this.idealDriftProbe = null;
|
|
998
1018
|
this.loopBreakInjected = 0;
|
|
999
1019
|
this.regroundingInjected = false;
|
|
1020
|
+
this.hookRecentCalls = [];
|
|
1021
|
+
this.semanticLoop = {
|
|
1022
|
+
checksUsed: 0,
|
|
1023
|
+
lastCheckTurn: 0,
|
|
1024
|
+
pending: false,
|
|
1025
|
+
verdict: null,
|
|
1026
|
+
injected: false,
|
|
1027
|
+
};
|
|
1028
|
+
this.independentReviewStarted = false;
|
|
1000
1029
|
this.runStartedAt = Date.now();
|
|
1001
1030
|
this.processGateInjected = 0;
|
|
1002
|
-
this.verificationGate.
|
|
1031
|
+
this.verificationGate.beginRun();
|
|
1032
|
+
const processes = new Set(this.processManager?.list().map((p) => p.id) ?? []);
|
|
1033
|
+
for (const id of this.backgroundVerification.keys()) {
|
|
1034
|
+
if (!processes.has(id))
|
|
1035
|
+
this.backgroundVerification.delete(id);
|
|
1036
|
+
}
|
|
1003
1037
|
this.compactionOccurred = false;
|
|
1004
1038
|
this.originalRequest = originalRequest;
|
|
1005
1039
|
}
|
|
@@ -1014,7 +1048,25 @@ export class AgentSession {
|
|
|
1014
1048
|
this.hookText += event.text;
|
|
1015
1049
|
break;
|
|
1016
1050
|
case "tool_call_start":
|
|
1017
|
-
this.hookToolCalls.set(event.toolCallId, {
|
|
1051
|
+
this.hookToolCalls.set(event.toolCallId, {
|
|
1052
|
+
name: event.name,
|
|
1053
|
+
args: event.args ?? {},
|
|
1054
|
+
revision: this.verificationGate.revision,
|
|
1055
|
+
});
|
|
1056
|
+
if (event.name === "bash" &&
|
|
1057
|
+
typeof event.args?.command === "string" &&
|
|
1058
|
+
(isVerificationCommand(event.args.command) ||
|
|
1059
|
+
classifyVerificationCommand(event.args.command).accepted)) {
|
|
1060
|
+
// A check that can rewrite files (--fix, build scripts, emitters)
|
|
1061
|
+
// invalidates earlier in-flight evidence AND marks the run as
|
|
1062
|
+
// touched. A check that is merely UNRECOGNIZED (`make test`, `deno
|
|
1063
|
+
// test`) rewrites nothing we can point to: bumping the revision for
|
|
1064
|
+
// it poisoned the gate on green output and re-armed the hook into
|
|
1065
|
+
// every later question turn.
|
|
1066
|
+
const classification = classifyVerificationCommand(event.args.command);
|
|
1067
|
+
this.verificationGate.requireFreshVerification(!classification.accepted && classification.mayMutate, event.args.command);
|
|
1068
|
+
await this.persistVerificationState();
|
|
1069
|
+
}
|
|
1018
1070
|
break;
|
|
1019
1071
|
case "tool_call_end": {
|
|
1020
1072
|
const call = this.hookToolCalls.get(event.toolCallId);
|
|
@@ -1032,14 +1084,28 @@ export class AgentSession {
|
|
|
1032
1084
|
this.hookConsecutiveFailures = event.isError ? this.hookConsecutiveFailures + 1 : 0;
|
|
1033
1085
|
this.hookRepeatedNoProgressCalls = this.hookProgressTracker.record(name, args, event.result, event.isError);
|
|
1034
1086
|
this.hookCyclicPattern = this.hookCycleDetector.record(name, args, event.result, event.isError);
|
|
1087
|
+
// Semantic-loop judge input: a bounded digest of WHAT was attempted and
|
|
1088
|
+
// HOW it came out. Args/results are sliced AT RECORD TIME — a write with
|
|
1089
|
+
// a 50 KiB payload or a bash dump must never inflate the ring, and the
|
|
1090
|
+
// judge needs shapes, not payloads.
|
|
1091
|
+
this.hookRecentCalls.push({
|
|
1092
|
+
tool: name,
|
|
1093
|
+
args: args === undefined ? "" : JSON.stringify(args).slice(0, 300),
|
|
1094
|
+
ok: !event.isError,
|
|
1095
|
+
result: event.result.slice(0, 400),
|
|
1096
|
+
});
|
|
1097
|
+
if (this.hookRecentCalls.length > MAX_SEMANTIC_LOOP_CALLS) {
|
|
1098
|
+
this.hookRecentCalls.splice(0, this.hookRecentCalls.length - MAX_SEMANTIC_LOOP_CALLS);
|
|
1099
|
+
}
|
|
1035
1100
|
if (name === "edit" && !event.isError) {
|
|
1036
1101
|
const diff = event.details?.diff ?? event.result;
|
|
1037
1102
|
const added = (diff.match(/^\+[^+]/gm) ?? []).length;
|
|
1038
1103
|
const removed = (diff.match(/^-[^-]/gm) ?? []).length;
|
|
1039
1104
|
this.hookStats.changedLines += added + removed;
|
|
1040
1105
|
}
|
|
1041
|
-
//
|
|
1042
|
-
//
|
|
1106
|
+
// Only host-observed successful mutations and trustworthy check results
|
|
1107
|
+
// affect approval. The model's text is never evidence.
|
|
1108
|
+
let verificationChanged = false;
|
|
1043
1109
|
if (!event.isError && args) {
|
|
1044
1110
|
if (name === "edit" || name === "write") {
|
|
1045
1111
|
const filePath = String(args.file_path ?? "");
|
|
@@ -1051,26 +1117,59 @@ export class AgentSession {
|
|
|
1051
1117
|
? String(args.content ?? "")
|
|
1052
1118
|
: extractAddedLines(event.details?.diff ?? event.result);
|
|
1053
1119
|
this.verificationGate.recordMutation(filePath, addedText);
|
|
1120
|
+
verificationChanged = true;
|
|
1121
|
+
}
|
|
1122
|
+
}
|
|
1123
|
+
}
|
|
1124
|
+
if (args && call && name === "bash") {
|
|
1125
|
+
const command = typeof args.command === "string" ? args.command : "";
|
|
1126
|
+
const classification = classifyVerificationCommand(command);
|
|
1127
|
+
if (classification.accepted) {
|
|
1128
|
+
if (args.run_in_background === true && !event.isError && args.persist !== true) {
|
|
1129
|
+
const id = /^ID:\s*(\S+)/m.exec(event.result)?.[1];
|
|
1130
|
+
// No parseable ID means the check cannot be tracked to a real exit
|
|
1131
|
+
// code — no evidence either way. Recording a FAILURE here made
|
|
1132
|
+
// every later green run of a different spelling look owed.
|
|
1133
|
+
if (id)
|
|
1134
|
+
this.backgroundVerification.set(id, { revision: call.revision, command });
|
|
1135
|
+
}
|
|
1136
|
+
else if (args.persist === true) {
|
|
1137
|
+
// Persistent-shell checks are not bounded evidence (steering can
|
|
1138
|
+
// interleave): neither a pass nor a failure. A recorded failure
|
|
1139
|
+
// here blocked approval for sessions that prefer the shell.
|
|
1140
|
+
}
|
|
1141
|
+
else {
|
|
1142
|
+
if (!event.isError && /^Exit code:\s*0(?:\s|$)/i.test(event.result.trim())) {
|
|
1143
|
+
this.verificationGate.recordVerification(call.revision, command);
|
|
1144
|
+
}
|
|
1145
|
+
else {
|
|
1146
|
+
this.verificationGate.recordFailedVerification(command, call.revision);
|
|
1147
|
+
}
|
|
1148
|
+
verificationChanged = true;
|
|
1054
1149
|
}
|
|
1055
1150
|
}
|
|
1056
|
-
if (
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
this.verificationGate.
|
|
1151
|
+
else if (classification.candidate) {
|
|
1152
|
+
// Green but untrusted: remember WHY so the demand can tell the
|
|
1153
|
+
// agent which command shape actually clears the gate.
|
|
1154
|
+
this.verificationGate.recordRejectedCheck(command, classification.reason);
|
|
1060
1155
|
}
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
|
|
1064
|
-
|
|
1065
|
-
if (
|
|
1066
|
-
|
|
1067
|
-
|
|
1068
|
-
|
|
1069
|
-
|
|
1070
|
-
this.verificationGate.
|
|
1156
|
+
}
|
|
1157
|
+
if (!event.isError && args && name === "task_output" && typeof args.id === "string") {
|
|
1158
|
+
const started = this.backgroundVerification.get(args.id);
|
|
1159
|
+
const proc = this.processManager?.list().find((p) => p.id === args.id);
|
|
1160
|
+
if (started && proc && proc.exitCode !== null) {
|
|
1161
|
+
if (proc.exitCode === 0) {
|
|
1162
|
+
this.verificationGate.recordVerification(started.revision, started.command);
|
|
1163
|
+
}
|
|
1164
|
+
else {
|
|
1165
|
+
this.verificationGate.recordFailedVerification(started.command, started.revision);
|
|
1071
1166
|
}
|
|
1167
|
+
this.backgroundVerification.delete(args.id);
|
|
1168
|
+
verificationChanged = true;
|
|
1072
1169
|
}
|
|
1073
1170
|
}
|
|
1171
|
+
if (verificationChanged)
|
|
1172
|
+
await this.persistVerificationState();
|
|
1074
1173
|
// Tool results are what push the run over the review gate, and they all
|
|
1075
1174
|
// land before the model writes its candidate final answer — so this is
|
|
1076
1175
|
// the point where arming still beats the draft's first token.
|
|
@@ -1219,16 +1318,21 @@ export class AgentSession {
|
|
|
1219
1318
|
return null;
|
|
1220
1319
|
if (!this.settingsManager.get("idealReviewEnabled"))
|
|
1221
1320
|
return null;
|
|
1321
|
+
// Deterministic stuck verdict, computed once and shared: the semantic
|
|
1322
|
+
// judge must not spend tokens on a burst the deterministic breaker is
|
|
1323
|
+
// about to correct itself.
|
|
1324
|
+
const deterministicDecision = evaluateLoopBreak({
|
|
1325
|
+
consecutiveFailures: this.hookConsecutiveFailures,
|
|
1326
|
+
repeatedNoProgressCalls: this.hookRepeatedNoProgressCalls,
|
|
1327
|
+
textRepetitionDetected: detectTextRepetition(this.hookText),
|
|
1328
|
+
...(this.hookCyclicPattern ? { cyclicPattern: this.hookCyclicPattern } : {}),
|
|
1329
|
+
});
|
|
1330
|
+
this.maybeStartSemanticLoopCheck(deterministicDecision.shouldBreak);
|
|
1222
1331
|
// Two-stage loop-breaker: stage 1 nudges; a FRESH detection after that
|
|
1223
1332
|
// injects the harsher final stop-and-report prompt. Signals reset after
|
|
1224
1333
|
// each injection so stage 2 only fires on new evidence.
|
|
1225
1334
|
if (this.loopBreakInjected < 2) {
|
|
1226
|
-
const decision =
|
|
1227
|
-
consecutiveFailures: this.hookConsecutiveFailures,
|
|
1228
|
-
repeatedNoProgressCalls: this.hookRepeatedNoProgressCalls,
|
|
1229
|
-
textRepetitionDetected: detectTextRepetition(this.hookText),
|
|
1230
|
-
...(this.hookCyclicPattern ? { cyclicPattern: this.hookCyclicPattern } : {}),
|
|
1231
|
-
});
|
|
1335
|
+
const decision = deterministicDecision;
|
|
1232
1336
|
if (decision.shouldBreak) {
|
|
1233
1337
|
const stage = this.loopBreakInjected === 0 ? 1 : 2;
|
|
1234
1338
|
this.loopBreakInjected = stage;
|
|
@@ -1237,6 +1341,9 @@ export class AgentSession {
|
|
|
1237
1341
|
this.hookCyclicPattern = null;
|
|
1238
1342
|
this.hookConsecutiveFailures = 0;
|
|
1239
1343
|
this.hookRepeatedNoProgressCalls = 0;
|
|
1344
|
+
// The deterministic breaker owns this burst — a semantic verdict from
|
|
1345
|
+
// the same burst must not double-correct on the next poll.
|
|
1346
|
+
this.semanticLoop.verdict = null;
|
|
1240
1347
|
// Clear the text buffer too — otherwise a stage-1 text-repetition
|
|
1241
1348
|
// trigger still sees the same repeated tail on the next check and
|
|
1242
1349
|
// escalates to stage 2 on stale evidence.
|
|
@@ -1249,6 +1356,22 @@ export class AgentSession {
|
|
|
1249
1356
|
return [buildLoopBreakMessage(decision.reasons, stage === 2)];
|
|
1250
1357
|
}
|
|
1251
1358
|
}
|
|
1359
|
+
// Semantic loop-break: an LLM verdict (started by maybeStartSemanticLoopCheck
|
|
1360
|
+
// on a suspicious-but-syntactically-quiet burst) is consumed here, exactly
|
|
1361
|
+
// once, with the deterministic breaker getting priority above. Fail-open:
|
|
1362
|
+
// no verdict or no-loop verdict injects nothing.
|
|
1363
|
+
if (this.semanticLoop.verdict && !this.semanticLoop.injected) {
|
|
1364
|
+
const verdict = this.semanticLoop.verdict;
|
|
1365
|
+
this.semanticLoop.injected = true;
|
|
1366
|
+
this.semanticLoop.verdict = null;
|
|
1367
|
+
// The judged burst has been addressed; a fresh burst must re-accumulate.
|
|
1368
|
+
this.hookConsecutiveFailures = 0;
|
|
1369
|
+
log("INFO", "loop-break", "Injecting semantic loop-break steering", {
|
|
1370
|
+
reason: verdict.reason,
|
|
1371
|
+
});
|
|
1372
|
+
this.eventBus.emit("hook", { kind: "loop_break" });
|
|
1373
|
+
return [buildSemanticLoopMessage(verdict)];
|
|
1374
|
+
}
|
|
1252
1375
|
if (!this.regroundingInjected && this.compactionOccurred) {
|
|
1253
1376
|
this.regroundingInjected = true;
|
|
1254
1377
|
this.eventBus.emit("hook", { kind: "regrounding" });
|
|
@@ -1256,6 +1379,159 @@ export class AgentSession {
|
|
|
1256
1379
|
}
|
|
1257
1380
|
return null;
|
|
1258
1381
|
}
|
|
1382
|
+
/** Fire the semantic loop judge when deterministic evidence is suspicious but
|
|
1383
|
+
* the deterministic breaker stayed quiet — the syntactic blind spot where
|
|
1384
|
+
* every retry differs slightly and the run still makes no progress.
|
|
1385
|
+
* Fire-and-forget: the call runs while the next turn streams, and a finished
|
|
1386
|
+
* verdict is consumed by the NEXT steering poll — never blocking a turn. */
|
|
1387
|
+
maybeStartSemanticLoopCheck(deterministicBreak) {
|
|
1388
|
+
if (!shouldRunSemanticLoopCheck({
|
|
1389
|
+
consecutiveFailures: this.hookConsecutiveFailures,
|
|
1390
|
+
totalFailures: this.hookStats.toolFailures,
|
|
1391
|
+
turns: this.hookStats.turns,
|
|
1392
|
+
lastCheckTurn: this.semanticLoop.lastCheckTurn,
|
|
1393
|
+
checksUsed: this.semanticLoop.checksUsed,
|
|
1394
|
+
checkPending: this.semanticLoop.pending,
|
|
1395
|
+
deterministicBreak,
|
|
1396
|
+
})) {
|
|
1397
|
+
return;
|
|
1398
|
+
}
|
|
1399
|
+
// resetHookState replaces this object on every prompt. A late judge must
|
|
1400
|
+
// not publish a verdict or consume the next run's budget/cooldown.
|
|
1401
|
+
const runState = this.semanticLoop;
|
|
1402
|
+
runState.pending = true;
|
|
1403
|
+
log("INFO", "loop-break", "Starting semantic loop judge", {
|
|
1404
|
+
turn: String(this.hookStats.turns),
|
|
1405
|
+
consecutiveFailures: String(this.hookConsecutiveFailures),
|
|
1406
|
+
recentCalls: String(this.hookRecentCalls.length),
|
|
1407
|
+
});
|
|
1408
|
+
void (async () => {
|
|
1409
|
+
try {
|
|
1410
|
+
const prompt = buildSemanticLoopJudgePrompt(this.hookRecentCalls, this.originalRequest);
|
|
1411
|
+
const raw = await (this.opts.semanticLoopJudge?.(prompt) ??
|
|
1412
|
+
this.callSemanticLoopJudge(prompt));
|
|
1413
|
+
const verdict = parseSemanticLoopVerdict(raw);
|
|
1414
|
+
if (this.semanticLoop === runState && verdict?.loop)
|
|
1415
|
+
runState.verdict = verdict;
|
|
1416
|
+
}
|
|
1417
|
+
catch (error) {
|
|
1418
|
+
// Fail open: judge errors never stop a run. Budget and cooldown are
|
|
1419
|
+
// still consumed in `finally` so a flaky judge cannot retry-loop.
|
|
1420
|
+
log("WARN", "loop-break", "Semantic loop judge failed", {
|
|
1421
|
+
error: error instanceof Error ? error.message : String(error),
|
|
1422
|
+
});
|
|
1423
|
+
}
|
|
1424
|
+
finally {
|
|
1425
|
+
if (this.semanticLoop === runState) {
|
|
1426
|
+
runState.pending = false;
|
|
1427
|
+
runState.checksUsed += 1;
|
|
1428
|
+
runState.lastCheckTurn = this.hookStats.turns;
|
|
1429
|
+
}
|
|
1430
|
+
}
|
|
1431
|
+
})();
|
|
1432
|
+
}
|
|
1433
|
+
/** One-shot judge call on the session's ACTIVE model — deliberately not a
|
|
1434
|
+
* cheaper routing: judging a model's own failure patterns with a weaker
|
|
1435
|
+
* model swaps false negatives for false positives. */
|
|
1436
|
+
async callSemanticLoopJudge(prompt) {
|
|
1437
|
+
const creds = await this.authStorage.resolveCredentials(this.provider, {
|
|
1438
|
+
storageKeys: this.currentAuthStorageKeys(),
|
|
1439
|
+
});
|
|
1440
|
+
const result = stream({
|
|
1441
|
+
provider: this.provider,
|
|
1442
|
+
model: this.model,
|
|
1443
|
+
messages: [{ role: "user", content: prompt }],
|
|
1444
|
+
maxTokens: 500,
|
|
1445
|
+
apiKey: creds.accessToken,
|
|
1446
|
+
accountId: creds.accountId,
|
|
1447
|
+
projectId: creds.projectId,
|
|
1448
|
+
baseUrl: this.baseUrl ?? creds.baseUrl,
|
|
1449
|
+
signal: this.opts.signal,
|
|
1450
|
+
});
|
|
1451
|
+
const response = await withJudgeTimeout(result.response, SEMANTIC_LOOP_JUDGE_TIMEOUT_MS);
|
|
1452
|
+
// Providers differ in reply shape: some return a bare string, others an
|
|
1453
|
+
// array of parts (glm-5.3 among them) — joining text parts covers both,
|
|
1454
|
+
// where the string-only branch silently dropped the whole verdict.
|
|
1455
|
+
const content = response.message.content;
|
|
1456
|
+
if (typeof content === "string")
|
|
1457
|
+
return content;
|
|
1458
|
+
return Array.isArray(content)
|
|
1459
|
+
? content
|
|
1460
|
+
.filter((part) => part.type === "text")
|
|
1461
|
+
.map((part) => part.text)
|
|
1462
|
+
.join("\n")
|
|
1463
|
+
: "";
|
|
1464
|
+
}
|
|
1465
|
+
/** Independent fresh-context review of the finished work (Codex Guardian
|
|
1466
|
+
* pattern). Spawns a READ-ONLY child on the ACTIVE model, waits bounded,
|
|
1467
|
+
* and returns findings for the acting agent to address — or nothing when
|
|
1468
|
+
* the review passes, is unavailable, or fails (in-thread review remains the
|
|
1469
|
+
* fallback; the feature degrades, never blocks).
|
|
1470
|
+
*
|
|
1471
|
+
* Runs inside the pre-stop poll, so the candidate final answer is already
|
|
1472
|
+
* held by arming and this wait cannot race a streamed answer. */
|
|
1473
|
+
async runIndependentReview(decision) {
|
|
1474
|
+
if (!this.subAgentManager)
|
|
1475
|
+
return [];
|
|
1476
|
+
if (this.independentReviewStarted)
|
|
1477
|
+
return [];
|
|
1478
|
+
// An allow-listed session (a subagent worker itself) must not spawn
|
|
1479
|
+
// harness-owned grandchildren the tool policy never granted.
|
|
1480
|
+
if (this.opts.allowedTools && !this.opts.allowedTools.includes("spawn_agent"))
|
|
1481
|
+
return [];
|
|
1482
|
+
if (decision.score < INDEPENDENT_REVIEW_SCORE_THRESHOLD)
|
|
1483
|
+
return [];
|
|
1484
|
+
this.independentReviewStarted = true;
|
|
1485
|
+
const taskName = `ideal-reviewer-${Math.random().toString(36).slice(2, 8)}`;
|
|
1486
|
+
let agentId;
|
|
1487
|
+
try {
|
|
1488
|
+
const task = buildReviewerTask({
|
|
1489
|
+
originalRequest: this.originalRequest,
|
|
1490
|
+
changedFiles: [...this.hookFileEditCounts.keys()],
|
|
1491
|
+
stats: this.hookStats,
|
|
1492
|
+
triggerReasons: decision.reasons,
|
|
1493
|
+
});
|
|
1494
|
+
// Active model forced at spawn time — never routed to a fast/review model.
|
|
1495
|
+
const snapshot = await this.subAgentManager.spawn(taskName, task, undefined, {
|
|
1496
|
+
model: this.model,
|
|
1497
|
+
tools: REVIEWER_TOOLS,
|
|
1498
|
+
});
|
|
1499
|
+
agentId = snapshot.agent_id;
|
|
1500
|
+
const waited = await this.subAgentManager.wait([agentId], "all", REVIEWER_WAIT_MS);
|
|
1501
|
+
const agent = waited.agents[0];
|
|
1502
|
+
if (!agent || !isTerminalSubAgentState(agent.state)) {
|
|
1503
|
+
// Timeout: collect the straggler so the completion gate cannot fire on
|
|
1504
|
+
// it later, then fall back to the in-thread review.
|
|
1505
|
+
await this.subAgentManager.interrupt(agentId, true).catch(() => { });
|
|
1506
|
+
log("WARN", "ideal", "Independent reviewer timed out; falling back to in-thread review", {
|
|
1507
|
+
agentId,
|
|
1508
|
+
});
|
|
1509
|
+
return [];
|
|
1510
|
+
}
|
|
1511
|
+
const findings = parseReviewerFindings(agent.output ?? "");
|
|
1512
|
+
if (!findings) {
|
|
1513
|
+
log("WARN", "ideal", "Independent reviewer output unparseable; falling back", { agentId });
|
|
1514
|
+
return [];
|
|
1515
|
+
}
|
|
1516
|
+
if (findings.clean) {
|
|
1517
|
+
log("INFO", "ideal", "Independent reviewer verdict: clean", { agentId });
|
|
1518
|
+
return [];
|
|
1519
|
+
}
|
|
1520
|
+
log("INFO", "ideal", "Independent reviewer flagged findings", {
|
|
1521
|
+
agentId,
|
|
1522
|
+
count: String(findings.findings.length),
|
|
1523
|
+
});
|
|
1524
|
+
return [buildIndependentReviewMessage(findings.findings)];
|
|
1525
|
+
}
|
|
1526
|
+
catch (error) {
|
|
1527
|
+
if (agentId)
|
|
1528
|
+
await this.subAgentManager.interrupt(agentId, true).catch(() => { });
|
|
1529
|
+
log("WARN", "ideal", "Independent reviewer failed; falling back to in-thread review", {
|
|
1530
|
+
error: error instanceof Error ? error.message : String(error),
|
|
1531
|
+
});
|
|
1532
|
+
return [];
|
|
1533
|
+
}
|
|
1534
|
+
}
|
|
1259
1535
|
/**
|
|
1260
1536
|
* Turn-budget extension gate. The loop consults this instead of stopping
|
|
1261
1537
|
* mid-task when it exhausts `maxTurns`. Grant ONLY on evidence of progress —
|
|
@@ -1373,7 +1649,7 @@ export class AgentSession {
|
|
|
1373
1649
|
* Pre-stop Ideal review phase machine. Once review starts, completion is
|
|
1374
1650
|
* blocked until harness-owned post-injection reads cover every changed file.
|
|
1375
1651
|
*/
|
|
1376
|
-
getHookFollowUpMessages() {
|
|
1652
|
+
async getHookFollowUpMessages() {
|
|
1377
1653
|
const childCompletionFollowUp = buildSubAgentCompletionFollowUp(this.subAgentManager);
|
|
1378
1654
|
if (childCompletionFollowUp)
|
|
1379
1655
|
return childCompletionFollowUp;
|
|
@@ -1394,13 +1670,17 @@ export class AgentSession {
|
|
|
1394
1670
|
if (this.opts.selfCorrectionHooks !== false &&
|
|
1395
1671
|
this.settingsManager.get("verificationGateEnabled") &&
|
|
1396
1672
|
(!this.opts.allowedTools || this.opts.allowedTools.includes("bash"))) {
|
|
1673
|
+
const verificationReason = this.verificationGate.pendingReason();
|
|
1397
1674
|
const verificationFollowUp = this.verificationGate.followUp();
|
|
1398
1675
|
if (verificationFollowUp) {
|
|
1399
1676
|
log("INFO", "verification-gate", "Injecting verification follow-up", {});
|
|
1400
1677
|
// Announce, THEN disarm: clients release held text on disarm, so the
|
|
1401
1678
|
// reverse order paints the draft and immediately deletes it — the exact
|
|
1402
1679
|
// flash arming exists to prevent.
|
|
1403
|
-
this.eventBus.emit("hook", {
|
|
1680
|
+
this.eventBus.emit("hook", {
|
|
1681
|
+
kind: "verification",
|
|
1682
|
+
...(verificationReason === "recheck" ? { verificationReason } : {}),
|
|
1683
|
+
});
|
|
1404
1684
|
this.refreshHookArming();
|
|
1405
1685
|
return verificationFollowUp;
|
|
1406
1686
|
}
|
|
@@ -1460,6 +1740,10 @@ export class AgentSession {
|
|
|
1460
1740
|
const driftedFiles = detectTestDrift(this.hookFileEditCounts.keys(), this.cwd).slice(0, 5);
|
|
1461
1741
|
if (!decision.shouldReview && driftedFiles.length === 0)
|
|
1462
1742
|
return null;
|
|
1743
|
+
// Independent reviewer first (async, bounded): its findings ride in the
|
|
1744
|
+
// SAME follow-up batch as the in-thread review + coverage requirements, so
|
|
1745
|
+
// addressing everything still costs one extra turn.
|
|
1746
|
+
const independentMessages = await this.runIndependentReview(decision);
|
|
1463
1747
|
this.reviewCoverage.start(this.hookFileEditCounts.keys());
|
|
1464
1748
|
this.idealReviewPhase = "reviewing";
|
|
1465
1749
|
const coverage = this.reviewCoverage.evidence();
|
|
@@ -1483,6 +1767,7 @@ export class AgentSession {
|
|
|
1483
1767
|
lspMissing: lspEvidence.missing,
|
|
1484
1768
|
});
|
|
1485
1769
|
return [
|
|
1770
|
+
...independentMessages,
|
|
1486
1771
|
this.withReviewLspEvidence(withReviewCoverageRequirements(buildIdealReviewMessage(decision.reasons, driftedFiles), coverage.missing), lspEvidence),
|
|
1487
1772
|
];
|
|
1488
1773
|
}
|
|
@@ -2083,6 +2368,7 @@ export class AgentSession {
|
|
|
2083
2368
|
await this.rePersistNolanTurns();
|
|
2084
2369
|
await this.rePersistAutopilotMarkers();
|
|
2085
2370
|
await this.rePersistAppMarkers();
|
|
2371
|
+
await this.persistVerificationState();
|
|
2086
2372
|
await this.persistAppMarker("compaction", {
|
|
2087
2373
|
originalCount: result.originalCount,
|
|
2088
2374
|
newCount: result.newCount,
|
|
@@ -2239,6 +2525,8 @@ export class AgentSession {
|
|
|
2239
2525
|
// Approved-plan execution is a clean checkpoint of the same conversation;
|
|
2240
2526
|
// explicit new sessions reset the conversation identity.
|
|
2241
2527
|
if (!preserveConversation) {
|
|
2528
|
+
this.verificationGate.reset();
|
|
2529
|
+
this.backgroundVerification.clear();
|
|
2242
2530
|
this.conversationId = "";
|
|
2243
2531
|
this.checkpointGeneration = 0;
|
|
2244
2532
|
this.sessionPreview = "";
|
|
@@ -2778,6 +3066,28 @@ export class AgentSession {
|
|
|
2778
3066
|
* instead of dropping the marker or falling back to a raw verdict string.
|
|
2779
3067
|
* No-op persistence for transient sessions (kept in memory only).
|
|
2780
3068
|
*/
|
|
3069
|
+
getVerificationProblem() {
|
|
3070
|
+
return (this.verificationGate.verificationProblem() ??
|
|
3071
|
+
(this.backgroundVerification.size > 0
|
|
3072
|
+
? "Unverified: a background check is still running or its result has not been confirmed."
|
|
3073
|
+
: null));
|
|
3074
|
+
}
|
|
3075
|
+
async persistVerificationState() {
|
|
3076
|
+
if (!this.sessionPath)
|
|
3077
|
+
return;
|
|
3078
|
+
const entry = {
|
|
3079
|
+
type: "custom",
|
|
3080
|
+
kind: VERIFICATION_STATE_KIND,
|
|
3081
|
+
id: crypto.randomUUID(),
|
|
3082
|
+
parentId: null,
|
|
3083
|
+
timestamp: new Date().toISOString(),
|
|
3084
|
+
data: {
|
|
3085
|
+
...this.verificationGate.snapshot(),
|
|
3086
|
+
unknown: this.getVerificationProblem() !== null,
|
|
3087
|
+
},
|
|
3088
|
+
};
|
|
3089
|
+
await this.sessionManager.appendEntry(this.sessionPath, entry);
|
|
3090
|
+
}
|
|
2781
3091
|
async persistAutopilotMarker(phase, extra) {
|
|
2782
3092
|
const afterMessageCount = this.persistedTranscriptCount();
|
|
2783
3093
|
const payload = {
|
|
@@ -3019,6 +3329,23 @@ export class AgentSession {
|
|
|
3019
3329
|
const loaded = await this.sessionManager.load(canonicalPath);
|
|
3020
3330
|
// Use the leaf from the header to walk the correct branch
|
|
3021
3331
|
const loadedMessages = this.sessionManager.getMessages(loaded.entries, loaded.header.leafId);
|
|
3332
|
+
this.backgroundVerification.clear();
|
|
3333
|
+
const savedVerification = [...loaded.entries]
|
|
3334
|
+
.reverse()
|
|
3335
|
+
.find((entry) => entry.type === "custom" && entry.kind === VERIFICATION_STATE_KIND);
|
|
3336
|
+
if (savedVerification?.type === "custom") {
|
|
3337
|
+
this.verificationGate.restore(savedVerification.data);
|
|
3338
|
+
}
|
|
3339
|
+
else {
|
|
3340
|
+
this.verificationGate.reset();
|
|
3341
|
+
// Legacy sessions have no host checkpoint. Tool-authored code history is
|
|
3342
|
+
// not proof of today's files; require fresh evidence before approval.
|
|
3343
|
+
if (loadedMessages.some((message) => message.role === "assistant" &&
|
|
3344
|
+
Array.isArray(message.content) &&
|
|
3345
|
+
message.content.some((part) => part.type === "tool_call" && (part.name === "edit" || part.name === "write")))) {
|
|
3346
|
+
this.verificationGate.requireFreshVerification();
|
|
3347
|
+
}
|
|
3348
|
+
}
|
|
3022
3349
|
this.checkpointGeneration = loaded.header.generation ?? 0;
|
|
3023
3350
|
this.conversationId = loaded.header.conversationId ?? loaded.header.id;
|
|
3024
3351
|
const legacyLabel = [...loaded.entries]
|