@prestyj/cli 5.28.1 → 5.29.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (316) hide show
  1. package/assets/motion/bin/contact-sheet.mjs +9 -3
  2. package/assets/motion/bin/cues.mjs +337 -0
  3. package/assets/motion/bin/library.mjs +53 -10
  4. package/assets/motion/bin/motion-blur.mjs +943 -0
  5. package/assets/motion/bin/motion-check.mjs +354 -3
  6. package/assets/motion/bin/music-fit.mjs +436 -0
  7. package/assets/motion/bin/pdf-extract.mjs +22 -10
  8. package/assets/motion/bin/reference-study.mjs +346 -0
  9. package/assets/motion/bin/score-synth.mjs +1052 -93
  10. package/assets/motion/library/README.md +50 -14
  11. package/assets/motion/library/kit/moves.js +1981 -0
  12. package/assets/motion/library/library.json +232 -0
  13. package/assets/motion/library/pieces/camera-rig/meta.json +13 -0
  14. package/assets/motion/library/pieces/camera-rig/piece.html +153 -0
  15. package/assets/motion/library/pieces/camera-rig/preview.jpg +0 -0
  16. package/assets/motion/library/pieces/chain-knock/meta.json +13 -0
  17. package/assets/motion/library/pieces/chain-knock/piece.html +195 -0
  18. package/assets/motion/library/pieces/chain-knock/preview.jpg +0 -0
  19. package/assets/motion/library/pieces/gather-to-logo/meta.json +13 -0
  20. package/assets/motion/library/pieces/gather-to-logo/piece.html +159 -0
  21. package/assets/motion/library/pieces/gather-to-logo/preview.jpg +0 -0
  22. package/assets/motion/library/pieces/morph-carry/meta.json +13 -0
  23. package/assets/motion/library/pieces/morph-carry/piece.html +173 -0
  24. package/assets/motion/library/pieces/morph-carry/preview.jpg +0 -0
  25. package/assets/motion/library/pieces/one-shape-journey/meta.json +13 -0
  26. package/assets/motion/library/pieces/one-shape-journey/piece.html +195 -0
  27. package/assets/motion/library/pieces/one-shape-journey/preview.jpg +0 -0
  28. package/assets/motion/library/pieces/open-from-subject/meta.json +13 -0
  29. package/assets/motion/library/pieces/open-from-subject/piece.html +168 -0
  30. package/assets/motion/library/pieces/open-from-subject/preview.jpg +0 -0
  31. package/assets/motion/library/pieces/request-to-result/meta.json +13 -0
  32. package/assets/motion/library/pieces/request-to-result/piece.html +212 -0
  33. package/assets/motion/library/pieces/request-to-result/preview.jpg +0 -0
  34. package/assets/motion/library/pieces/scale-dive/meta.json +13 -0
  35. package/assets/motion/library/pieces/scale-dive/piece.html +321 -0
  36. package/assets/motion/library/pieces/scale-dive/preview.jpg +0 -0
  37. package/assets/motion/library/pieces/screen-replica-steps/meta.json +13 -0
  38. package/assets/motion/library/pieces/screen-replica-steps/piece.html +366 -0
  39. package/assets/motion/library/pieces/screen-replica-steps/preview.jpg +0 -0
  40. package/assets/motion/library/pieces/zoom-into-card/meta.json +13 -0
  41. package/assets/motion/library/pieces/zoom-into-card/piece.html +179 -0
  42. package/assets/motion/library/pieces/zoom-into-card/preview.jpg +0 -0
  43. package/assets/motion/library/sheets/diagram.jpg +0 -0
  44. package/assets/motion/library/sheets/frame.jpg +0 -0
  45. package/assets/motion/library/sheets/transition.jpg +0 -0
  46. package/assets/motion/library/sheets/ui.jpg +0 -0
  47. package/assets/motion/references/build-sheet.md +206 -0
  48. package/assets/motion/references/runtime/determinism-rules.md +1 -1
  49. package/assets/motion/references/runtime/gsap-easing-and-stagger.md +29 -29
  50. package/assets/motion/references/runtime/inputs-and-assets.md +7 -12
  51. package/assets/motion/references/runtime/lint-validate-inspect.md +3 -3
  52. package/assets/motion/references/runtime/minimal-composition.md +1 -1
  53. package/assets/motion/references/runtime/preview-render.md +3 -3
  54. package/assets/motion/skills/app-walkthrough/SKILL.md +66 -0
  55. package/assets/motion/skills/before-after/SKILL.md +53 -0
  56. package/assets/motion/skills/brand-kit/SKILL.md +3 -3
  57. package/assets/motion/skills/dev-tool-video/SKILL.md +57 -0
  58. package/assets/motion/skills/launch-video/SKILL.md +62 -0
  59. package/assets/motion/skills/match-reference/SKILL.md +58 -0
  60. package/assets/motion/skills/motion/SKILL.md +78 -85
  61. package/assets/motion/skills/source-ingest/SKILL.md +17 -6
  62. package/assets/motion/skills/website-video/SKILL.md +59 -0
  63. package/assets/skills/bulletproof/SKILL.md +36 -11
  64. package/assets/skills/bulletproof/references/agent-surface.md +19 -9
  65. package/assets/skills/bulletproof/references/audit-protocol.md +20 -5
  66. package/assets/skills/bulletproof/references/platform-playbooks.md +5 -4
  67. package/assets/skills/bulletproof/references/provenance.md +26 -1
  68. package/assets/skills/bulletproof/references/secure-defaults.md +6 -5
  69. package/assets/skills/bulletproof/references/supply-chain.md +21 -17
  70. package/assets/skills/bulletproof/references/threat-landscape.md +28 -26
  71. package/assets/skills/bulletproof/references/verification.md +2 -0
  72. package/assets/skills/clarify/SKILL.md +25 -16
  73. package/assets/skills/code-review/SKILL.md +71 -13
  74. package/assets/skills/code-review/references/agent-diffs.md +27 -0
  75. package/assets/skills/code-review/references/tests.md +19 -0
  76. package/assets/skills/compliance-guard/SKILL.md +20 -5
  77. package/assets/skills/compliance-guard/references/artifacts.md +1 -1
  78. package/assets/skills/compliance-guard/references/eu-uk.md +16 -16
  79. package/assets/skills/compliance-guard/references/lawsuit-vectors.md +5 -5
  80. package/assets/skills/compliance-guard/references/provenance.md +41 -2
  81. package/assets/skills/compliance-guard/references/sector-gates.md +3 -3
  82. package/assets/skills/compliance-guard/references/security-baseline.md +2 -2
  83. package/assets/skills/compliance-guard/references/trigger-map.md +5 -5
  84. package/assets/skills/compliance-guard/references/us.md +27 -21
  85. package/assets/skills/durable/SKILL.md +87 -79
  86. package/assets/skills/durable/references/agent-db-safety.md +69 -0
  87. package/assets/skills/durable/references/backups-and-runtime.md +19 -12
  88. package/assets/skills/durable/references/migrations-and-schema.md +13 -6
  89. package/assets/skills/evidence-led-ui/SKILL.md +69 -127
  90. package/assets/skills/evidence-led-ui/references/anti-defaults.md +107 -208
  91. package/assets/skills/evidence-led-ui/references/direction.md +124 -0
  92. package/assets/skills/evidence-led-ui/references/production-contract.md +8 -0
  93. package/assets/skills/evidence-led-ui/references/provenance.md +24 -1
  94. package/assets/skills/lean/SKILL.md +90 -71
  95. package/assets/skills/lean/references/memory-and-processes.md +3 -2
  96. package/assets/skills/lean/references/playbooks.md +37 -12
  97. package/assets/skills/refactoring/SKILL.md +24 -3
  98. package/assets/skills/refactoring/references/agent-pitfalls.md +4 -1
  99. package/assets/skills/refactoring/references/legacy.md +21 -0
  100. package/assets/skills/root-cause/SKILL.md +20 -10
  101. package/assets/skills/shared-language/SKILL.md +16 -14
  102. package/assets/skills/tdd/SKILL.md +27 -15
  103. package/dist/app-sidecar.js +203 -47
  104. package/dist/app-sidecar.js.map +1 -1
  105. package/dist/cli.js +17 -26
  106. package/dist/cli.js.map +1 -1
  107. package/dist/core/acceptance-checks.d.ts +48 -0
  108. package/dist/core/acceptance-checks.js +144 -0
  109. package/dist/core/acceptance-checks.js.map +1 -0
  110. package/dist/core/agent-session.d.ts +106 -72
  111. package/dist/core/agent-session.js +539 -399
  112. package/dist/core/agent-session.js.map +1 -1
  113. package/dist/core/agents.d.ts +6 -5
  114. package/dist/core/agents.js.map +1 -1
  115. package/dist/core/ask-user.d.ts +90 -8
  116. package/dist/core/ask-user.js +124 -13
  117. package/dist/core/ask-user.js.map +1 -1
  118. package/dist/core/bundled-agents.js +1 -3
  119. package/dist/core/bundled-agents.js.map +1 -1
  120. package/dist/core/cache-diagnostics.d.ts +68 -0
  121. package/dist/core/cache-diagnostics.js +196 -0
  122. package/dist/core/cache-diagnostics.js.map +1 -0
  123. package/dist/core/cache-expiry.d.ts +87 -0
  124. package/dist/core/cache-expiry.js +111 -0
  125. package/dist/core/cache-expiry.js.map +1 -0
  126. package/dist/core/compaction/compactor.js +78 -48
  127. package/dist/core/compaction/compactor.js.map +1 -1
  128. package/dist/core/compaction/plan-step-policy.d.ts +46 -0
  129. package/dist/core/compaction/plan-step-policy.js +57 -0
  130. package/dist/core/compaction/plan-step-policy.js.map +1 -0
  131. package/dist/core/destructive-git-guard.d.ts +90 -0
  132. package/dist/core/destructive-git-guard.js +871 -0
  133. package/dist/core/destructive-git-guard.js.map +1 -0
  134. package/dist/core/event-bus.d.ts +3 -0
  135. package/dist/core/event-bus.js +5 -0
  136. package/dist/core/event-bus.js.map +1 -1
  137. package/dist/core/injection-detect.d.ts +38 -0
  138. package/dist/core/injection-detect.js +232 -0
  139. package/dist/core/injection-detect.js.map +1 -0
  140. package/dist/core/keep-awake.d.ts +88 -0
  141. package/dist/core/keep-awake.js +251 -0
  142. package/dist/core/keep-awake.js.map +1 -0
  143. package/dist/core/mcp/client.d.ts +72 -0
  144. package/dist/core/mcp/client.js +264 -41
  145. package/dist/core/mcp/client.js.map +1 -1
  146. package/dist/core/mcp/content.js +6 -2
  147. package/dist/core/mcp/content.js.map +1 -1
  148. package/dist/core/mcp/store.d.ts +6 -1
  149. package/dist/core/mcp/store.js +12 -1
  150. package/dist/core/mcp/store.js.map +1 -1
  151. package/dist/core/mcp/types.d.ts +18 -0
  152. package/dist/core/model-unavailable.d.ts +14 -0
  153. package/dist/core/model-unavailable.js +23 -0
  154. package/dist/core/model-unavailable.js.map +1 -0
  155. package/dist/core/node-debugger.d.ts +148 -0
  156. package/dist/core/node-debugger.js +642 -0
  157. package/dist/core/node-debugger.js.map +1 -0
  158. package/dist/core/package-threats.d.ts +18 -0
  159. package/dist/core/package-threats.js +168 -0
  160. package/dist/core/package-threats.js.map +1 -0
  161. package/dist/core/persistent-shell.d.ts +58 -6
  162. package/dist/core/persistent-shell.js +331 -49
  163. package/dist/core/persistent-shell.js.map +1 -1
  164. package/dist/core/process-manager.d.ts +14 -0
  165. package/dist/core/process-manager.js +61 -0
  166. package/dist/core/process-manager.js.map +1 -1
  167. package/dist/core/progress/git-xp.js +8 -14
  168. package/dist/core/progress/git-xp.js.map +1 -1
  169. package/dist/core/session-history.d.ts +12 -0
  170. package/dist/core/session-history.js +27 -0
  171. package/dist/core/session-history.js.map +1 -1
  172. package/dist/core/session-manager.d.ts +13 -1
  173. package/dist/core/session-manager.js +38 -18
  174. package/dist/core/session-manager.js.map +1 -1
  175. package/dist/core/session-summary-index.d.ts +37 -0
  176. package/dist/core/session-summary-index.js +172 -0
  177. package/dist/core/session-summary-index.js.map +1 -0
  178. package/dist/core/settings-manager.d.ts +2 -0
  179. package/dist/core/settings-manager.js +10 -0
  180. package/dist/core/settings-manager.js.map +1 -1
  181. package/dist/core/shell-threats-popular-packages.d.ts +11 -0
  182. package/dist/core/shell-threats-popular-packages.js +675 -0
  183. package/dist/core/shell-threats-popular-packages.js.map +1 -0
  184. package/dist/core/shell-threats.d.ts +8 -0
  185. package/dist/core/shell-threats.js +186 -0
  186. package/dist/core/shell-threats.js.map +1 -0
  187. package/dist/core/skills.js +3 -1
  188. package/dist/core/skills.js.map +1 -1
  189. package/dist/core/stream-rules.d.ts +30 -0
  190. package/dist/core/stream-rules.js +151 -0
  191. package/dist/core/stream-rules.js.map +1 -0
  192. package/dist/core/subagent-manager.d.ts +20 -5
  193. package/dist/core/subagent-manager.js +22 -8
  194. package/dist/core/subagent-manager.js.map +1 -1
  195. package/dist/core/subagent-receipt.d.ts +54 -0
  196. package/dist/core/subagent-receipt.js +276 -0
  197. package/dist/core/subagent-receipt.js.map +1 -0
  198. package/dist/core/subagent-turn-record.d.ts +2 -0
  199. package/dist/core/subagent-turn-record.js.map +1 -1
  200. package/dist/core/test-impact.d.ts +73 -0
  201. package/dist/core/test-impact.js +467 -0
  202. package/dist/core/test-impact.js.map +1 -0
  203. package/dist/core/thinking-level.d.ts +1 -1
  204. package/dist/core/thinking-level.js +1 -1
  205. package/dist/core/thinking-level.js.map +1 -1
  206. package/dist/core/verification-gate.d.ts +2 -0
  207. package/dist/core/verification-gate.js +4 -0
  208. package/dist/core/verification-gate.js.map +1 -1
  209. package/dist/core/verification-snapshot.js +3 -5
  210. package/dist/core/verification-snapshot.js.map +1 -1
  211. package/dist/core/workspace-guard.d.ts +18 -7
  212. package/dist/core/workspace-guard.js +227 -60
  213. package/dist/core/workspace-guard.js.map +1 -1
  214. package/dist/interactive.js +2 -1
  215. package/dist/interactive.js.map +1 -1
  216. package/dist/modes/subagent-worker-mode.js +34 -6
  217. package/dist/modes/subagent-worker-mode.js.map +1 -1
  218. package/dist/motion-agent/motion-agent.d.ts +6 -2
  219. package/dist/motion-agent/motion-agent.js +5 -7
  220. package/dist/motion-agent/motion-agent.js.map +1 -1
  221. package/dist/motion-agent/motion-prompt.d.ts +1 -1
  222. package/dist/motion-agent/motion-prompt.js +14 -19
  223. package/dist/motion-agent/motion-prompt.js.map +1 -1
  224. package/dist/motion-agent/motion-review.d.ts +10 -3
  225. package/dist/motion-agent/motion-review.js +15 -7
  226. package/dist/motion-agent/motion-review.js.map +1 -1
  227. package/dist/motion-agent/motion-studio-context.js +1 -1
  228. package/dist/motion-agent/motion-studio-context.js.map +1 -1
  229. package/dist/system-prompt.js +3 -1
  230. package/dist/system-prompt.js.map +1 -1
  231. package/dist/test-support/keep-alive.d.ts +14 -0
  232. package/dist/test-support/keep-alive.js +17 -0
  233. package/dist/test-support/keep-alive.js.map +1 -0
  234. package/dist/tools/ask-user.js +3 -3
  235. package/dist/tools/ask-user.js.map +1 -1
  236. package/dist/tools/bash-read-evidence.d.ts +10 -0
  237. package/dist/tools/bash-read-evidence.js +133 -0
  238. package/dist/tools/bash-read-evidence.js.map +1 -0
  239. package/dist/tools/bash.d.ts +10 -1
  240. package/dist/tools/bash.js +115 -7
  241. package/dist/tools/bash.js.map +1 -1
  242. package/dist/tools/debug.d.ts +54 -0
  243. package/dist/tools/debug.js +233 -0
  244. package/dist/tools/debug.js.map +1 -0
  245. package/dist/tools/edit.js +12 -4
  246. package/dist/tools/edit.js.map +1 -1
  247. package/dist/tools/goals.d.ts +1 -1
  248. package/dist/tools/index.d.ts +19 -2
  249. package/dist/tools/index.js +47 -7
  250. package/dist/tools/index.js.map +1 -1
  251. package/dist/tools/prompt-hints.js +2 -0
  252. package/dist/tools/prompt-hints.js.map +1 -1
  253. package/dist/tools/read-tracker.d.ts +5 -0
  254. package/dist/tools/read-tracker.js +19 -9
  255. package/dist/tools/read-tracker.js.map +1 -1
  256. package/dist/tools/read.js +3 -2
  257. package/dist/tools/read.js.map +1 -1
  258. package/dist/tools/skill.js +5 -0
  259. package/dist/tools/skill.js.map +1 -1
  260. package/dist/tools/subagent-control.js +44 -8
  261. package/dist/tools/subagent-control.js.map +1 -1
  262. package/dist/tools/subagent-shared.d.ts +29 -8
  263. package/dist/tools/subagent-shared.js +45 -14
  264. package/dist/tools/subagent-shared.js.map +1 -1
  265. package/dist/tools/subagent.d.ts +8 -2
  266. package/dist/tools/subagent.js +28 -10
  267. package/dist/tools/subagent.js.map +1 -1
  268. package/dist/tools/task-output.js +3 -2
  269. package/dist/tools/task-output.js.map +1 -1
  270. package/dist/tools/task-send.d.ts +1 -1
  271. package/dist/tools/task-send.js +15 -1
  272. package/dist/tools/task-send.js.map +1 -1
  273. package/dist/tools/tool-tiers.d.ts +2 -2
  274. package/dist/tools/tool-tiers.js +3 -2
  275. package/dist/tools/tool-tiers.js.map +1 -1
  276. package/dist/tools/truncate.d.ts +21 -0
  277. package/dist/tools/truncate.js +187 -0
  278. package/dist/tools/truncate.js.map +1 -1
  279. package/dist/tools/ui-adopt.js +2 -0
  280. package/dist/tools/ui-adopt.js.map +1 -1
  281. package/dist/ui/App.d.ts +0 -4
  282. package/dist/ui/App.js +5 -28
  283. package/dist/ui/App.js.map +1 -1
  284. package/dist/ui/components/ActivityIndicator.js +1 -0
  285. package/dist/ui/components/ActivityIndicator.js.map +1 -1
  286. package/dist/ui/hooks/useAgentLoop.d.ts +1 -8
  287. package/dist/ui/hooks/useAgentLoop.js +1 -119
  288. package/dist/ui/hooks/useAgentLoop.js.map +1 -1
  289. package/dist/ui/render.d.ts +0 -4
  290. package/dist/ui/render.js +0 -2
  291. package/dist/ui/render.js.map +1 -1
  292. package/dist/utils/git.d.ts +77 -0
  293. package/dist/utils/git.js +285 -21
  294. package/dist/utils/git.js.map +1 -1
  295. package/dist/utils/github-ci.js +2 -1
  296. package/dist/utils/github-ci.js.map +1 -1
  297. package/dist/utils/github.js +11 -9
  298. package/dist/utils/github.js.map +1 -1
  299. package/dist/utils/image.d.ts +14 -0
  300. package/dist/utils/image.js +16 -0
  301. package/dist/utils/image.js.map +1 -1
  302. package/dist/utils/process.d.ts +20 -0
  303. package/dist/utils/process.js +98 -0
  304. package/dist/utils/process.js.map +1 -1
  305. package/package.json +5 -5
  306. package/assets/motion/references/motion-language.md +0 -128
  307. package/assets/motion/skills/video-qa/SKILL.md +0 -89
  308. package/dist/core/ideal-review-subagent.d.ts +0 -56
  309. package/dist/core/ideal-review-subagent.js +0 -112
  310. package/dist/core/ideal-review-subagent.js.map +0 -1
  311. package/dist/core/ideal-review.d.ts +0 -82
  312. package/dist/core/ideal-review.js +0 -242
  313. package/dist/core/ideal-review.js.map +0 -1
  314. package/dist/motion-agent/motion-check-tool.d.ts +0 -35
  315. package/dist/motion-agent/motion-check-tool.js +0 -514
  316. package/dist/motion-agent/motion-check-tool.js.map +0 -1
@@ -1,6 +1,7 @@
1
- import { agentLoop, isAbortError, isUsageLimitError, } from "@prestyj/agent";
1
+ import { agentLoop, isAbortError, isUsageLimitError, repairToolPairingAdjacent, } from "@prestyj/agent";
2
2
  import { ProviderError, stream, } from "@prestyj/ai";
3
3
  import { EventBus } from "./event-bus.js";
4
+ import { flagUntrustedToolResult } from "./injection-detect.js";
4
5
  import { COMPLETION_REVIEW_STATE_KIND, } from "./completion-review.js";
5
6
  import { SlashCommandRegistry, createBuiltinCommands, } from "./slash-commands.js";
6
7
  import { PROMPT_COMMANDS, getPromptCommand } from "./prompt-commands.js";
@@ -23,8 +24,10 @@ import { ensureAppDirs } from "../config.js";
23
24
  import { buildSubAgentSystemPrompt, buildSystemPrompt, } from "../system-prompt.js";
24
25
  import { createTools, createWebSearchTool, } from "../tools/index.js";
25
26
  import { partitionToolsByTier } from "../tools/tool-tiers.js";
27
+ import { formatImpactForVerification } from "./test-impact.js";
28
+ import { autoBackgroundedId } from "../tools/bash.js";
26
29
  import { buildProcessCompletionFollowUp } from "./process-gate.js";
27
- import { buildSubAgentCompletionFollowUp, } from "./subagent-manager.js";
30
+ import { buildSubAgentCompletionFollowUp } from "./subagent-manager.js";
28
31
  import { applyAsyncSubagentPolicy } from "./subagent-policy.js";
29
32
  import { z } from "zod";
30
33
  import { MCPClientManager, getAllMcpServers } from "./mcp/index.js";
@@ -36,27 +39,31 @@ import { createToolSearchTool } from "../tools/tool-search.js";
36
39
  import { createSessionStatsTool } from "../tools/session-stats.js";
37
40
  import { createDiagnoseCommand, isInternalDiagnosticsEnabled, SessionDiagnosticsRecorder, } from "./internal-diagnostics.js";
38
41
  import { log } from "./logger.js";
39
- import { setEstimatorModel, calibrateEstimatorFromUsage } from "./compaction/token-estimator.js";
42
+ import { CacheDiagnostics } from "./cache-diagnostics.js";
43
+ import { assessCacheExpiry, resolveCacheTtl } from "./cache-expiry.js";
44
+ import { setEstimatorModel, calibrateEstimatorFromUsage, estimateConversationTokens, } from "./compaction/token-estimator.js";
40
45
  import { calculateActiveContextTokens } from "./compaction/active-context.js";
41
46
  import { resolveCompactionPolicy } from "./compaction/policy.js";
47
+ import { decidePlanStepCompaction, DEFAULT_CACHE_WRITE_READ_RATIO, PLAN_STEP_KEEP_TOKENS, } from "./compaction/plan-step-policy.js";
48
+ import { extractPlanSteps, findCompletedMarkers } from "../utils/plan-steps.js";
49
+ import { readFileSync } from "node:fs";
42
50
  import { clampThinkingForPlanMode } from "./thinking-level.js";
43
51
  import { pruneStaleToolResults } from "./compaction/tool-result-pruner.js";
44
52
  import { discoverAgents } from "./agents.js";
45
53
  import { enhancePrompt } from "../utils/prompt-enhancer.js";
46
54
  import { detectLanguages, detectProjectStack } from "./language-detector.js";
47
- import { evaluateIdealReview, buildIdealReviewMessage, buildReviewCoverageEscalationMessage, buildReviewCoverageMessage, MAX_REVIEW_COVERAGE_INJECTIONS, withReviewCoverageRequirements, detectTestDrift, ReviewCoverageTracker, } from "./ideal-review.js";
48
55
  import { evaluateLoopBreak, buildLoopBreakMessage, CycleDetector, ToolCallProgressTracker, detectTextRepetition, } from "./loop-breaker.js";
49
56
  import { buildRegroundingMessage, requestTextForRegrounding } from "./regrounding.js";
50
57
  import { buildSemanticLoopJudgePrompt, buildSemanticLoopMessage, MAX_SEMANTIC_LOOP_CALLS, parseSemanticLoopVerdict, shouldRunSemanticLoopCheck, SEMANTIC_LOOP_JUDGE_TIMEOUT_MS, withJudgeTimeout, } from "./semantic-loop-check.js";
51
- import { buildIndependentReviewMessage, buildReviewerTask, INDEPENDENT_REVIEW_SCORE_THRESHOLD, parseReviewerFindings, REVIEWER_TOOLS, REVIEWER_TURN_TIMEOUT_MS, REVIEWER_WAIT_MS, } from "./ideal-review-subagent.js";
52
58
  import { buildEnvDeltaMessage } from "./env-delta.js";
53
59
  import { wrapSteeringText, buildNotificationSteeringText, STEERING_PREFIX } from "./steering.js";
54
60
  import { AgentNotificationQueue } from "./agent-notifications.js";
55
- import { VerificationGate, extractAddedLines, isCheckOwnFile, isCodeFilePath, VERIFICATION_STATE_KIND, isVerificationCommand, } from "./verification-gate.js";
61
+ import { VerificationGate, isCheckOwnFile, extractAddedLines, isCodeFilePath, VERIFICATION_STATE_KIND, isVerificationCommand, } from "./verification-gate.js";
56
62
  import { classifyVerificationCommand } from "./verification-evidence.js";
57
63
  import { captureVerificationSnapshot } from "./verification-snapshot.js";
58
64
  import { findUserSessionPrompt, getUserSessionPrompt } from "./session-preview.js";
59
65
  import { normalizeMessageImages } from "./message-images.js";
66
+ import { loadStreamRules } from "./stream-rules.js";
60
67
  import crypto from "node:crypto";
61
68
  import fs from "node:fs/promises";
62
69
  import os from "node:os";
@@ -66,14 +73,6 @@ import path from "node:path";
66
73
  * progressing — refuse to extend its turn budget.
67
74
  */
68
75
  const TURN_EXTENSION_MAX_FAILURE_RATIO = 0.5;
69
- /** Terminal subagent states — mirrors SubAgentManager's private isTerminal. */
70
- function isTerminalSubAgentState(state) {
71
- return (state === "completed" ||
72
- state === "failed" ||
73
- state === "interrupted" ||
74
- state === "closed" ||
75
- state === "reaped");
76
- }
77
76
  // ── Tool-result policy ─────────────────────────────────────
78
77
  /** Resolve the per-result cap passed to the agent loop for the active transport. */
79
78
  export function resolveSessionToolResultCharLimit(model, provider, accountId) {
@@ -130,6 +129,7 @@ export class AgentSession {
130
129
  // transcript rows the live run showed.
131
130
  appMarkers = [];
132
131
  turnMetrics = [];
132
+ cacheDiagnostics = new CacheDiagnostics();
133
133
  /** Internal-only (EZ_INTERNAL): live per-session cost/reliability recorder.
134
134
  * Absent entirely in public builds — see core/internal-diagnostics.ts. */
135
135
  diagnosticsRecorder;
@@ -141,20 +141,13 @@ export class AgentSession {
141
141
  /** Forgets every file read; called whenever the conversation is replaced or
142
142
  * rewound, so the model must re-read a file before changing it. */
143
143
  clearReadTracker;
144
+ recordBashReads;
144
145
  skills = [];
145
146
  cacheKeyLogged = false;
146
147
  // ── Self-correction hook state (mirrors the TUI's useAgentLoop refs) ──
147
148
  // Reset at the start of every run; observed from the event stream; read by
148
- // the loop-break (mid-loop) and ideal-review (pre-stop) callbacks.
149
- hookStats = {
150
- changedLines: 0,
151
- toolCalls: 0,
152
- toolFailures: 0,
153
- turns: 0,
154
- writeCalls: 0,
155
- editCalls: 0,
156
- bashCalls: 0,
157
- };
149
+ // the loop-break (mid-loop) callback.
150
+ hookStats = { toolCalls: 0, toolFailures: 0, turns: 0 };
158
151
  hookText = "";
159
152
  hookConsecutiveFailures = 0;
160
153
  hookRepeatedNoProgressCalls = 0;
@@ -164,20 +157,6 @@ export class AgentSession {
164
157
  hookFileEditCounts = new Map();
165
158
  hookToolCalls = new Map();
166
159
  backgroundVerification = new Map();
167
- idealReviewPhase = "idle";
168
- /** Runtime-only suppression while Nolan owns verification in autopilot mode. */
169
- idealReviewSuppressed = false;
170
- /** Mirror of the last `hook_armed` value broadcast this run, so the event
171
- * fires only on a real edge. */
172
- idealReviewArmed = false;
173
- /** Cached test-drift probe, keyed by the size of the edited-file set. Drift
174
- * depends only on WHICH files were edited and that set only grows, so this
175
- * keeps the arming check off the filesystem on most tool results — the probe
176
- * is several sync existsSync calls per edited file. */
177
- idealDriftProbe = null;
178
- reviewCoverage;
179
- /** Coverage follow-ups spent this run, capped by MAX_REVIEW_COVERAGE_INJECTIONS. */
180
- reviewCoverageInjected = 0;
181
160
  /** 0 = none; 1 = first nudge sent; 2 = final stop-and-report injected. */
182
161
  loopBreakInjected = 0;
183
162
  regroundingInjected = false;
@@ -187,8 +166,6 @@ export class AgentSession {
187
166
  * injection at the next steering poll; judge failures fail open (no
188
167
  * injection) and still consume budget + cooldown. */
189
168
  semanticLoop = { checksUsed: 0, lastCheckTurn: 0, pending: false, verdict: null, injected: false };
190
- /** Independent Ideal reviewer spawned once per run (score-gated). */
191
- independentReviewStarted = false;
192
169
  /**
193
170
  * The environment as the cached system prompt currently describes it.
194
171
  * Re-recorded on every prompt build, so a rebuild (e.g. `/add-dir`) needs no
@@ -199,11 +176,12 @@ export class AgentSession {
199
176
  runStartedAt = 0;
200
177
  /** Gate injections spent this run, capped by MAX_PROCESS_GATE_INJECTIONS. */
201
178
  processGateInjected = 0;
202
- /** Verification gate: code edited this run, nothing proved it since. */
179
+ /** Verification gate: code edited this run, nothing proved it since. Always
180
+ * tracks evidence for run status; the `verificationGateEnabled` setting
181
+ * decides whether an unverified stop is also continued once. */
203
182
  verificationGate = new VerificationGate();
204
- /** Mirror of the last verification `hook_armed` value, so the event fires
205
- * only on a real edge. */
206
- verificationArmed = false;
183
+ /** Mirror of the last `hook_armed` value, so the event fires only on an edge. */
184
+ preFinalArmed = false;
207
185
  compactionOccurred = false;
208
186
  /**
209
187
  * Re-grounding carry-over for post-turn compaction. `resetHookState` clears
@@ -216,6 +194,25 @@ export class AgentSession {
216
194
  postTurnCompaction;
217
195
  lastCompactionCompacted = false;
218
196
  compactionRetryAfter = 0;
197
+ /**
198
+ * SoL-Pi plan-step compaction bookkeeping (see compaction/plan-step-policy.ts).
199
+ * Step progress comes from `[DONE:n]` markers in assistant text — the same
200
+ * contract the approved-plan UI tracks.
201
+ */
202
+ planStepState = {
203
+ planPath: undefined,
204
+ scanIndex: 0,
205
+ doneSteps: new Set(),
206
+ requestsInCompletedSteps: 0,
207
+ requestsThisStep: 0,
208
+ requests: 0,
209
+ grownTokens: 0,
210
+ lastContextTokens: 0,
211
+ compactions: 0,
212
+ writeCostBalance: 0,
213
+ savingPerRequest: 0,
214
+ requestsSinceLastCompaction: 0,
215
+ };
219
216
  /** A restored oversized checkpoint must be canonicalized before its first prompt is persisted. */
220
217
  deferredCompactionPending = false;
221
218
  /** Latest provider count, anchored to the assistant response it measured. */
@@ -231,8 +228,12 @@ export class AgentSession {
231
228
  // different message (or past the end) by the time the cancel arrives.
232
229
  userQueue = [];
233
230
  queueSeq = 0;
231
+ /** Instant interrupt: the running loop's preempt listeners, fired on queueMessage. */
232
+ steeringListeners = new Set();
234
233
  processManager;
235
234
  lspManager;
235
+ testImpact;
236
+ debugManager;
236
237
  subAgentManager;
237
238
  /**
238
239
  * Out-of-band push notifications (finished children, background-process
@@ -336,7 +337,6 @@ export class AgentSession {
336
337
  this.provider = options.provider;
337
338
  this.model = options.model;
338
339
  this.cwd = options.cwd;
339
- this.reviewCoverage = new ReviewCoverageTracker(this.cwd);
340
340
  this.baseUrl = options.baseUrl;
341
341
  this.maxTokens = this.resolveMaxTokens(options.model);
342
342
  this.thinkingLevel = options.thinkingLevel;
@@ -402,7 +402,7 @@ export class AgentSession {
402
402
  : this.opts.globalSubagents
403
403
  ? await discoverAgents({ globalAgentsDir: paths.agentsDir })
404
404
  : [];
405
- const { tools: builtInTools, processManager, rebuildReadTool, clearReadTracker, lspManager, subAgentManager, } = await createTools(this.cwd, {
405
+ const { tools: builtInTools, processManager, rebuildReadTool, clearReadTracker, recordBashReads, lspManager, testImpact, debugManager, subAgentManager, } = await createTools(this.cwd, {
406
406
  agents,
407
407
  skills: this.skills,
408
408
  contextLimits: this.contextLimits,
@@ -433,17 +433,14 @@ export class AgentSession {
433
433
  }),
434
434
  getUseExternalGrep: () => this.settingsManager.get("grepUseRipgrep"),
435
435
  authStorage: this.authStorage,
436
- onFileRead: (filePath) => this.reviewCoverage.recordRead(filePath),
437
436
  onFileMutated: (filePath) => {
438
437
  const relative = path.relative(this.cwd, filePath) || path.basename(filePath);
439
438
  this.hookFileEditCounts.set(relative, (this.hookFileEditCounts.get(relative) ?? 0) + 1);
440
- this.reviewCoverage.recordChanged(filePath);
441
439
  },
442
440
  // Lazy — sessionId/model/provider can change after createTools() runs, so
443
441
  // sub-agent spawns read the current parent state at execution time.
444
442
  getProvider: () => this.provider,
445
443
  getModel: () => this.model,
446
- getThinkingLevel: () => this.thinkingLevel,
447
444
  getBaseUrl: () => this.baseUrl,
448
445
  getCacheKey: () => this.getPromptCacheKey(),
449
446
  getMaxPerModel: () => this.settingsManager.get("subagentMaxPerModel"),
@@ -489,12 +486,16 @@ export class AgentSession {
489
486
  this.mcpCatalog ??= new DeferredToolCatalog(this.contextLimits);
490
487
  this.mcpCatalog.add(deferred);
491
488
  this.ensureToolSearchTool();
489
+ this.promoteWaitAgentAfterSpawn();
492
490
  }
493
491
  }
494
492
  this.rebuildReadTool = rebuildReadTool;
495
493
  this.clearReadTracker = clearReadTracker;
494
+ this.recordBashReads = recordBashReads;
496
495
  this.processManager = processManager;
497
496
  this.lspManager = lspManager;
497
+ this.testImpact = testImpact;
498
+ this.debugManager = debugManager;
498
499
  this.subAgentManager = subAgentManager;
499
500
  this.bindManagerCancellation(this.opts.signal);
500
501
  // Connect MCP servers. Child sessions skip user-configured servers to avoid
@@ -779,6 +780,29 @@ export class AgentSession {
779
780
  : { serverName, ok: false, error: outcome.error };
780
781
  }, this.contextLimits));
781
782
  }
783
+ /**
784
+ * `wait_agent` is deferred, yet nearly every `spawn_agent` is followed by it,
785
+ * so the model spent a whole turn on `tool_search` just to load it (bench 41:
786
+ * ~6 s per fan-out). Promote it as soon as a spawn succeeds instead: the tool
787
+ * list grows exactly as it would after that `tool_search`, one turn earlier,
788
+ * and sessions that never spawn keep the smaller prefix.
789
+ */
790
+ promoteWaitAgentAfterSpawn() {
791
+ const index = this.tools.findIndex((t) => t.name === "spawn_agent");
792
+ const spawn = index >= 0 ? this.tools[index] : undefined;
793
+ if (!spawn)
794
+ return;
795
+ this.tools[index] = {
796
+ ...spawn,
797
+ execute: async (args, context) => {
798
+ const result = await spawn.execute(args, context);
799
+ if (!this.tools.some((t) => t.name === "wait_agent")) {
800
+ this.tools.push(...(this.mcpCatalog?.promote(["wait_agent"]) ?? []));
801
+ }
802
+ return result;
803
+ },
804
+ };
805
+ }
782
806
  /** Append tools, replacing any same-named entry (cached stub → live tool). */
783
807
  replaceOrPushTools(tools) {
784
808
  for (const tool of tools) {
@@ -917,6 +941,7 @@ export class AgentSession {
917
941
  kind: "prompt",
918
942
  visibility: "transcript",
919
943
  }, options = {}) {
944
+ this.prewarmController?.abort();
920
945
  await this.settlePostTurnCompaction();
921
946
  await this.adoptDeferredCheckpointBeforePrompt();
922
947
  const slash = await this.resolveSlashInput(content);
@@ -952,6 +977,7 @@ export class AgentSession {
952
977
  * attachments are always a direct conversational turn.
953
978
  */
954
979
  async promptWithAttachments(text, attachments) {
980
+ this.prewarmController?.abort();
955
981
  await this.settlePostTurnCompaction();
956
982
  await this.adoptDeferredCheckpointBeforePrompt();
957
983
  const parts = this.buildAttachmentParts(text, attachments);
@@ -1050,15 +1076,7 @@ export class AgentSession {
1050
1076
  resetHookState(originalRequest) {
1051
1077
  this.opts.completionReview?.begin(originalRequest);
1052
1078
  this.lspManager?.clearPendingDiagnostics();
1053
- this.hookStats = {
1054
- changedLines: 0,
1055
- toolCalls: 0,
1056
- toolFailures: 0,
1057
- turns: 0,
1058
- writeCalls: 0,
1059
- editCalls: 0,
1060
- bashCalls: 0,
1061
- };
1079
+ this.hookStats = { toolCalls: 0, toolFailures: 0, turns: 0 };
1062
1080
  this.hookText = "";
1063
1081
  this.hookConsecutiveFailures = 0;
1064
1082
  this.hookRepeatedNoProgressCalls = 0;
@@ -1067,13 +1085,6 @@ export class AgentSession {
1067
1085
  this.hookCyclicPattern = null;
1068
1086
  this.hookFileEditCounts.clear();
1069
1087
  this.hookToolCalls.clear();
1070
- this.reviewCoverage.reset();
1071
- this.reviewCoverageInjected = 0;
1072
- this.idealReviewPhase = "idle";
1073
- // No event here: clients reset their own hold on run_start.
1074
- this.idealReviewArmed = false;
1075
- this.verificationArmed = false;
1076
- this.idealDriftProbe = null;
1077
1088
  this.loopBreakInjected = 0;
1078
1089
  this.regroundingInjected = false;
1079
1090
  this.hookRecentCalls = [];
@@ -1084,7 +1095,6 @@ export class AgentSession {
1084
1095
  verdict: null,
1085
1096
  injected: false,
1086
1097
  };
1087
- this.independentReviewStarted = false;
1088
1098
  this.runStartedAt = Date.now();
1089
1099
  this.processGateInjected = 0;
1090
1100
  this.verificationGate.beginRun();
@@ -1105,8 +1115,8 @@ export class AgentSession {
1105
1115
  }
1106
1116
  /**
1107
1117
  * Fold one agent event into the hook stat accumulators. Pure bookkeeping —
1108
- * the same signals the TUI's useAgentLoop collects, so the loop-break and
1109
- * ideal-review decisions match across the CLI and the app.
1118
+ * the same signals the TUI's useAgentLoop collects, so loop-break decisions
1119
+ * match across the CLI and the app.
1110
1120
  */
1111
1121
  async trackHookEvent(event) {
1112
1122
  if (this.opts.completionReview) {
@@ -1172,12 +1182,6 @@ export class AgentSession {
1172
1182
  this.hookStats.toolCalls += 1;
1173
1183
  if (event.isError)
1174
1184
  this.hookStats.toolFailures += 1;
1175
- if (name === "write")
1176
- this.hookStats.writeCalls += 1;
1177
- if (name === "edit")
1178
- this.hookStats.editCalls += 1;
1179
- if (name === "bash")
1180
- this.hookStats.bashCalls += 1;
1181
1185
  this.hookConsecutiveFailures = event.isError ? this.hookConsecutiveFailures + 1 : 0;
1182
1186
  this.hookRepeatedNoProgressCalls = this.hookProgressTracker.record(name, args, event.result, event.isError);
1183
1187
  this.hookCyclicPattern = this.hookCycleDetector.record(name, args, event.result, event.isError);
@@ -1194,12 +1198,6 @@ export class AgentSession {
1194
1198
  if (this.hookRecentCalls.length > MAX_SEMANTIC_LOOP_CALLS) {
1195
1199
  this.hookRecentCalls.splice(0, this.hookRecentCalls.length - MAX_SEMANTIC_LOOP_CALLS);
1196
1200
  }
1197
- if (name === "edit" && !event.isError) {
1198
- const diff = event.details?.diff ?? event.result;
1199
- const added = (diff.match(/^\+[^+]/gm) ?? []).length;
1200
- const removed = (diff.match(/^-[^-]/gm) ?? []).length;
1201
- this.hookStats.changedLines += added + removed;
1202
- }
1203
1201
  // Only host-observed successful mutations and trustworthy check results
1204
1202
  // affect approval. The model's text is never evidence.
1205
1203
  let verificationChanged = false;
@@ -1222,8 +1220,13 @@ export class AgentSession {
1222
1220
  const command = typeof args.command === "string" ? args.command : "";
1223
1221
  const classification = classifyVerificationCommand(command);
1224
1222
  if (classification.accepted || classification.snapshotEligible) {
1225
- if (args.run_in_background === true && !event.isError && args.persist !== true) {
1226
- const id = /^ID:\s*(\S+)/m.exec(event.result)?.[1];
1223
+ // A foreground check that outlived the default budget was moved to
1224
+ // the background, not failed: track it to its real exit the same way.
1225
+ const autoBackgroundId = event.isError ? undefined : autoBackgroundedId(event.result);
1226
+ if ((autoBackgroundId !== undefined || args.run_in_background === true) &&
1227
+ !event.isError &&
1228
+ args.persist !== true) {
1229
+ const id = autoBackgroundId ?? /^ID:\s*(\S+)/m.exec(event.result)?.[1];
1227
1230
  // No parseable ID means the check cannot be tracked to a real exit
1228
1231
  // code — no evidence either way. Recording a FAILURE here made
1229
1232
  // every later green run of a different spelling look owed.
@@ -1294,10 +1297,35 @@ export class AgentSession {
1294
1297
  // Tool results for this step are in the array and their side effects
1295
1298
  // already hit the filesystem. Flushing here is what makes a crash lose
1296
1299
  // at most the in-flight step instead of the entire turn.
1300
+ await this.creditBashReads();
1297
1301
  await this.flushPendingMessages();
1298
1302
  break;
1299
1303
  }
1300
1304
  }
1305
+ /**
1306
+ * Count full-file `cat` output from the step that just finished as reads, so
1307
+ * an edit after `cat` does not cost a second read. Uses the step's results as
1308
+ * stored in the transcript (after per-turn trimming): only bytes the model
1309
+ * actually received count.
1310
+ */
1311
+ async creditBashReads() {
1312
+ if (!this.recordBashReads)
1313
+ return;
1314
+ const messages = this.activeLoopMessages ?? this.messages;
1315
+ const toolMessage = messages.at(-1);
1316
+ const assistant = messages.at(-2);
1317
+ if (toolMessage?.role !== "tool" || assistant?.role !== "assistant")
1318
+ return;
1319
+ if (typeof assistant.content === "string")
1320
+ return;
1321
+ const calls = assistant.content.filter((part) => part.type === "tool_call");
1322
+ try {
1323
+ await this.recordBashReads(calls, toolMessage.content);
1324
+ }
1325
+ catch (err) {
1326
+ log("WARN", "agent-session", "Crediting bash reads failed", { error: String(err) });
1327
+ }
1328
+ }
1301
1329
  /**
1302
1330
  * Append every message added since the last flush to the session file.
1303
1331
  *
@@ -1360,7 +1388,7 @@ export class AgentSession {
1360
1388
  const diagnosticText = this.lspManager?.drainDiagnostics(this.getVerificationProblem() !== null, { deferUnverified: true });
1361
1389
  if (diagnosticText)
1362
1390
  this.eventBus.emit("diagnostics", { text: diagnosticText });
1363
- this.refreshVerificationArmed();
1391
+ this.refreshHookArming();
1364
1392
  const notified = this.notifications.drain();
1365
1393
  const notificationMessage = notified.length > 0 || diagnosticText
1366
1394
  ? {
@@ -1423,6 +1451,7 @@ export class AgentSession {
1423
1451
  }
1424
1452
  if (this.opts.selfCorrectionHooks === false)
1425
1453
  return null;
1454
+ // Legacy key: the user-facing switch for loop-break and re-grounding nudges.
1426
1455
  if (!this.settingsManager.get("idealReviewEnabled"))
1427
1456
  return null;
1428
1457
  // Deterministic stuck verdict, computed once and shared: the semantic
@@ -1622,83 +1651,6 @@ export class AgentSession {
1622
1651
  .join("\n")
1623
1652
  : "";
1624
1653
  }
1625
- /** Independent fresh-context review of the finished work (Codex Guardian
1626
- * pattern). Spawns a READ-ONLY child on the ACTIVE model, waits bounded,
1627
- * and returns findings for the acting agent to address — or nothing when
1628
- * the review passes, is unavailable, or fails (in-thread review remains the
1629
- * fallback; the feature degrades, never blocks).
1630
- *
1631
- * Runs inside the pre-stop poll, so the candidate final answer is already
1632
- * held by arming and this wait cannot race a streamed answer. */
1633
- async runIndependentReview(decision) {
1634
- if (!this.subAgentManager)
1635
- return [];
1636
- if (this.independentReviewStarted)
1637
- return [];
1638
- // An allow-listed session (a subagent worker itself) must not spawn
1639
- // harness-owned grandchildren the tool policy never granted.
1640
- if (this.opts.allowedTools && !this.opts.allowedTools.includes("spawn_agent"))
1641
- return [];
1642
- if (decision.score < INDEPENDENT_REVIEW_SCORE_THRESHOLD)
1643
- return [];
1644
- this.independentReviewStarted = true;
1645
- const taskName = `ideal-reviewer-${Math.random().toString(36).slice(2, 8)}`;
1646
- let agentId;
1647
- try {
1648
- const task = buildReviewerTask({
1649
- originalRequest: this.originalRequest,
1650
- changedFiles: [...this.hookFileEditCounts.keys()],
1651
- stats: this.hookStats,
1652
- triggerReasons: decision.reasons,
1653
- });
1654
- // Active model forced at spawn time — never routed to a fast/review model.
1655
- // The reviewer's own time limit ends it with a verdict on what it read;
1656
- // the wait below is only a backstop against a hung child.
1657
- const snapshot = await this.subAgentManager.spawn(taskName, task, undefined, {
1658
- model: this.model,
1659
- tools: REVIEWER_TOOLS,
1660
- turnTimeoutMs: REVIEWER_TURN_TIMEOUT_MS,
1661
- });
1662
- agentId = snapshot.agent_id;
1663
- const waited = await this.subAgentManager.wait([agentId], "all", REVIEWER_WAIT_MS);
1664
- const agent = waited.agents[0];
1665
- if (!agent || !isTerminalSubAgentState(agent.state)) {
1666
- // Timeout: collect the straggler so the completion gate cannot fire on
1667
- // it later, then fall back to the in-thread review.
1668
- await this.subAgentManager.interrupt(agentId, true).catch(() => { });
1669
- log("WARN", "ideal", "Independent reviewer timed out; falling back to in-thread review", {
1670
- agentId,
1671
- });
1672
- return [];
1673
- }
1674
- const findings = parseReviewerFindings(agent.output ?? "");
1675
- if (!findings) {
1676
- log("WARN", "ideal", "Independent reviewer output unparseable; falling back", {
1677
- agentId,
1678
- state: agent.state,
1679
- ...(agent.error ? { error: agent.error } : {}),
1680
- });
1681
- return [];
1682
- }
1683
- if (findings.clean) {
1684
- log("INFO", "ideal", "Independent reviewer verdict: clean", { agentId });
1685
- return [];
1686
- }
1687
- log("INFO", "ideal", "Independent reviewer flagged findings", {
1688
- agentId,
1689
- count: String(findings.findings.length),
1690
- });
1691
- return [buildIndependentReviewMessage(findings.findings)];
1692
- }
1693
- catch (error) {
1694
- if (agentId)
1695
- await this.subAgentManager.interrupt(agentId, true).catch(() => { });
1696
- log("WARN", "ideal", "Independent reviewer failed; falling back to in-thread review", {
1697
- error: error instanceof Error ? error.message : String(error),
1698
- });
1699
- return [];
1700
- }
1701
- }
1702
1654
  /**
1703
1655
  * Turn-budget extension gate. The loop consults this instead of stopping
1704
1656
  * mid-task when it exhausts `maxTurns`. Grant ONLY on evidence of progress —
@@ -1732,98 +1684,45 @@ export class AgentSession {
1732
1684
  });
1733
1685
  return granted;
1734
1686
  }
1735
- /**
1736
- * Would the stop AFTER the current turn inject the Ideal review? Same inputs
1737
- * as the pre-stop gate below, evaluated early so clients know a candidate
1738
- * final answer is a review draft BEFORE it streams.
1739
- *
1740
- * The turn count is looked ahead by one on purpose. `hookStats.turns` only
1741
- * advances at `turn_end`, so while the model is writing the draft the counter
1742
- * still reads the PREVIOUS turn; the real gate sees one more. Without the
1743
- * lookahead a run sitting on score 3 crosses to 4 on the draft's own
1744
- * `turn_end` — after the text already streamed — which is precisely the
1745
- * appear-then-vanish flash. Over-arming by one turn point costs only live
1746
- * token streaming on a final answer that then shows whole; under-arming costs
1747
- * the flash, so this errs toward arming.
1748
- */
1749
- wouldInjectIdealReview() {
1750
- if (this.opts.completionReview?.armed)
1751
- return true;
1752
- if (this.opts.selfCorrectionHooks === false || this.idealReviewSuppressed)
1753
- return false;
1754
- // Mid-review a stop still injects: the coverage follow-up while files are
1755
- // unread, or its escalation once the budget is spent. Both make the model
1756
- // answer again, so the candidate answer is a draft exactly as it is before
1757
- // the review starts — without arming here it paints and the reviewed answer
1758
- // lands under it as a duplicate.
1759
- if (this.idealReviewPhase === "reviewing") {
1760
- return this.reviewCoverage.evidence().missing.length > 0;
1761
- }
1762
- if (this.idealReviewPhase !== "idle")
1763
- return false;
1764
- if (!this.settingsManager.get("idealReviewEnabled"))
1687
+ /** Is the pre-stop verification gate active for this session? Off by the
1688
+ * `verificationGateEnabled` setting, by `selfCorrectionHooks: false`, and for
1689
+ * allow-listed sessions that cannot run commands at all. */
1690
+ verificationGateActive() {
1691
+ if (this.opts.selfCorrectionHooks === false)
1765
1692
  return false;
1766
- if (evaluateIdealReview({ ...this.hookStats, turns: this.hookStats.turns + 1 }).shouldReview) {
1767
- return true;
1768
- }
1769
- const files = this.hookFileEditCounts.size;
1770
- if (files === 0)
1693
+ if (!this.settingsManager.get("verificationGateEnabled"))
1771
1694
  return false;
1772
- if (this.idealDriftProbe?.files !== files) {
1773
- this.idealDriftProbe = {
1774
- files,
1775
- drifted: detectTestDrift(this.hookFileEditCounts.keys(), this.cwd).length > 0,
1776
- };
1777
- }
1778
- return this.idealDriftProbe.drifted;
1695
+ return !this.opts.allowedTools || this.opts.allowedTools.includes("bash");
1779
1696
  }
1780
- /** Would a stop right now inject the verification gate? Same conditions as
1781
- * the pre-stop branch below, so arming and injection cannot disagree. */
1782
- wouldInjectVerification() {
1697
+ /** Would a stop right now inject a pre-final follow-up? Queued LSP
1698
+ * diagnostics (real errors injected below), a mode-owned completion review,
1699
+ * or the verification gate can, so clients hold the candidate answer only
1700
+ * then. Same conditions as the pre-stop branch, so arming and injection
1701
+ * cannot disagree. */
1702
+ wouldInjectBeforeFinal() {
1703
+ if (this.opts.completionReview?.armed)
1704
+ return true;
1783
1705
  if (this.lspManager?.hasQueuedDiagnostics())
1784
1706
  return true;
1785
- if (this.opts.selfCorrectionHooks === false)
1786
- return false;
1787
- if (!this.settingsManager.get("verificationGateEnabled"))
1788
- return false;
1789
- if (this.opts.allowedTools && !this.opts.allowedTools.includes("bash"))
1790
- return false;
1791
- return this.verificationGate.willInject();
1707
+ return this.verificationGateActive() && this.verificationGate.willInject();
1792
1708
  }
1793
1709
  /** Broadcast pre-final hook arming on change. Both edges matter: armed=false
1794
- * after the hook fires is what lets a client stream the REVIEWED final
1795
- * answer live again.
1796
- *
1797
- * Callable before `initialize()`: the sidecar sets Nolan's review suppression
1798
- * on a freshly constructed session, and every arming predicate below reads
1799
- * settings that `initialize()` has not loaded yet. Nothing can be armed
1800
- * before the session can run a turn, and the first `tool_result`/`turn_end`
1801
- * recomputes both edges — so skipping is the correct answer, not a patch. */
1710
+ * after the hook fires is what lets a client stream the final answer live
1711
+ * again. Callable before `initialize()`, when no manager exists yet. */
1802
1712
  refreshHookArming() {
1713
+ // Before `initialize()` settings are not loaded, so nothing can be armed.
1803
1714
  if (!this.settingsManager)
1804
1715
  return;
1805
- this.refreshIdealReviewArmed();
1806
- this.refreshVerificationArmed();
1807
- }
1808
- refreshVerificationArmed() {
1809
- if (!this.settingsManager)
1810
- return;
1811
- const armed = this.wouldInjectVerification();
1812
- if (armed === this.verificationArmed)
1716
+ const armed = this.wouldInjectBeforeFinal();
1717
+ if (armed === this.preFinalArmed)
1813
1718
  return;
1814
- this.verificationArmed = armed;
1719
+ this.preFinalArmed = armed;
1815
1720
  this.eventBus.emit("hook_armed", { kind: "verification", armed });
1816
1721
  }
1817
- refreshIdealReviewArmed() {
1818
- const armed = this.wouldInjectIdealReview();
1819
- if (armed === this.idealReviewArmed)
1820
- return;
1821
- this.idealReviewArmed = armed;
1822
- this.eventBus.emit("hook_armed", { kind: "ideal", armed });
1823
- }
1824
1722
  /**
1825
- * Pre-stop Ideal review phase machine. Once review starts, completion is
1826
- * blocked until harness-owned post-injection reads cover every changed file.
1723
+ * Pre-stop follow-ups: LSP errors, unread child agents and background
1724
+ * processes, the verification gate (when `verificationGateEnabled`), and a
1725
+ * mode-owned completion review.
1827
1726
  */
1828
1727
  async getHookFollowUpMessages() {
1829
1728
  // Exit notifications and task_output refer to the same host process record.
@@ -1836,14 +1735,15 @@ export class AgentSession {
1836
1735
  if (backgroundChanged)
1837
1736
  await this.persistVerificationState();
1838
1737
  // Edits return immediately; only the completion boundary waits for remaining
1839
- // checks. Queued timeouts stay explicitly unverified, never a false all-clear.
1738
+ // checks. Queued timeouts stay explicitly unverified, never a false
1739
+ // all-clear; they join the verification demand below when the gate is on.
1840
1740
  await this.lspManager?.flushDiagnostics(this.opts.signal);
1841
1741
  if (this.opts.signal?.aborted)
1842
1742
  return null;
1843
- const diagnosticText = this.lspManager?.drainDiagnostics(this.getVerificationProblem() !== null);
1743
+ const diagnosticText = this.lspManager?.drainDiagnostics(this.verificationGateActive() && this.getVerificationProblem() !== null);
1844
1744
  if (diagnosticText)
1845
1745
  this.eventBus.emit("diagnostics", { text: diagnosticText });
1846
- this.refreshVerificationArmed();
1746
+ this.refreshHookArming();
1847
1747
  const diagnosticMessages = diagnosticText
1848
1748
  ? [
1849
1749
  {
@@ -1868,18 +1768,29 @@ export class AgentSession {
1868
1768
  return [...diagnosticMessages, ...processFollowUp];
1869
1769
  }
1870
1770
  // Verification gate: code was edited but nothing verified since the last
1871
- // edit. Above the Ideal review so checks RUN before the read-based review
1872
- // starts; off for allow-listed sessions that cannot run commands at all.
1873
- if (this.opts.selfCorrectionHooks !== false &&
1874
- this.settingsManager.get("verificationGateEnabled") &&
1875
- (!this.opts.allowedTools || this.opts.allowedTools.includes("bash"))) {
1771
+ // edit. Off via the `verificationGateEnabled` setting.
1772
+ if (this.verificationGateActive()) {
1876
1773
  const verificationReason = this.verificationGate.pendingReason();
1774
+ const pendingFiles = this.verificationGate.pendingFiles();
1877
1775
  const verificationFollowUp = this.verificationGate.followUp();
1878
1776
  if (verificationFollowUp) {
1777
+ // Name the tests that actually reach the unverified files, with the
1778
+ // command that runs exactly those, so the check is targeted rather
1779
+ // than guessed. Best-effort: no index or no runner leaves it unchanged.
1780
+ if (this.testImpact &&
1781
+ (verificationReason === "initial" || verificationReason === "recheck")) {
1782
+ const impactLine = await this.testImpact
1783
+ .impactFor(pendingFiles)
1784
+ .then(formatImpactForVerification)
1785
+ .catch(() => "");
1786
+ const first = verificationFollowUp[0];
1787
+ if (impactLine && first?.role === "user" && typeof first.content === "string") {
1788
+ verificationFollowUp[0] = { ...first, content: first.content + impactLine };
1789
+ }
1790
+ }
1879
1791
  log("INFO", "verification-gate", "Injecting verification follow-up", {});
1880
1792
  // Announce, THEN disarm: clients release held text on disarm, so the
1881
- // reverse order paints the draft and immediately deletes it — the exact
1882
- // flash arming exists to prevent.
1793
+ // reverse order paints the draft and immediately deletes it.
1883
1794
  this.eventBus.emit("hook", {
1884
1795
  kind: "verification",
1885
1796
  ...(verificationReason === "tamper"
@@ -1892,7 +1803,7 @@ export class AgentSession {
1892
1803
  return [...diagnosticMessages, ...verificationFollowUp];
1893
1804
  }
1894
1805
  }
1895
- // Address real errors before review; unavailable checks share the existing
1806
+ // Address real errors before review; unavailable checks share the
1896
1807
  // verification demand above instead of manufacturing a separate hook.
1897
1808
  if (diagnosticMessages.length > 0)
1898
1809
  return diagnosticMessages;
@@ -1912,131 +1823,24 @@ export class AgentSession {
1912
1823
  }
1913
1824
  this.refreshHookArming();
1914
1825
  }
1915
- if (this.opts.selfCorrectionHooks === false || this.idealReviewSuppressed)
1916
- return null;
1917
- if (this.idealReviewPhase === "reviewing") {
1918
- const coverage = this.reviewCoverage.evidence();
1919
- const lspEvidence = this.reviewLspEvidence(coverage.expected);
1920
- log("INFO", "ideal", "Ideal review coverage check", {
1921
- covered: coverage.covered,
1922
- missing: coverage.missing,
1923
- lspLowConfidence: lspEvidence.lowConfidence,
1924
- lspMissing: lspEvidence.missing,
1925
- });
1926
- if (coverage.missing.length > 0) {
1927
- // Announce like any other pre-final injection: this follow-up makes the
1928
- // model answer again, so the answer it interrupts is a draft and the
1929
- // hook event is what tells clients to discard it. Injecting silently is
1930
- // what let the pre-coverage answer paint above the reviewed one.
1931
- this.eventBus.emit("hook", {
1932
- kind: "ideal",
1933
- coverageExpected: coverage.expected,
1934
- coverageMissing: coverage.missing,
1935
- });
1936
- if (this.reviewCoverageInjected < MAX_REVIEW_COVERAGE_INJECTIONS) {
1937
- this.reviewCoverageInjected += 1;
1938
- // Stays armed (coverage is still outstanding) — this call is here so a
1939
- // client that missed the earlier edge is armed before the next draft.
1940
- this.refreshIdealReviewArmed();
1941
- return [
1942
- this.withReviewLspEvidence(buildReviewCoverageMessage(coverage.missing), lspEvidence),
1943
- ];
1944
- }
1945
- // Budget spent: close the gate so the run cannot spin on a file that
1946
- // never becomes readable, and require the gap be reported to the user.
1947
- this.idealReviewPhase = "complete";
1948
- // The gate is shut, so this is the real disarm: the answer to the
1949
- // escalation is final and streams live.
1950
- this.refreshIdealReviewArmed();
1951
- log("INFO", "ideal", "Ideal review coverage escalated after retry budget", {
1952
- injected: String(this.reviewCoverageInjected),
1953
- missing: coverage.missing,
1954
- });
1955
- return [buildReviewCoverageEscalationMessage(coverage.missing)];
1956
- }
1957
- this.idealReviewPhase = "complete";
1958
- return null;
1826
+ return null;
1827
+ }
1828
+ /** Wraps the real run: cancels an in-flight cache prewarm and tracks run
1829
+ * activity / last real request time for {@link prewarm}. */
1830
+ async runLoop(options = {}) {
1831
+ this.prewarmController?.abort();
1832
+ this.runLoopDepth++;
1833
+ try {
1834
+ await this.runLoopInner(options);
1835
+ }
1836
+ finally {
1837
+ this.runLoopDepth--;
1838
+ this.lastRealRequestAt = Date.now();
1959
1839
  }
1960
- if (this.idealReviewPhase === "complete")
1961
- return null;
1962
- if (!this.settingsManager.get("idealReviewEnabled"))
1963
- return null;
1964
- const decision = evaluateIdealReview(this.hookStats);
1965
- // Test drift fires the review even on a small change the score would skip:
1966
- // a green-but-stale test is exactly what the volume gate sleeps through.
1967
- const driftedFiles = detectTestDrift(this.hookFileEditCounts.keys(), this.cwd).slice(0, 5);
1968
- if (!decision.shouldReview && driftedFiles.length === 0)
1969
- return null;
1970
- // Independent reviewer first (async, bounded): its findings ride in the
1971
- // SAME follow-up batch as the in-thread review + coverage requirements, so
1972
- // addressing everything still costs one extra turn.
1973
- this.reviewCoverage.start(this.hookFileEditCounts.keys());
1974
- this.idealReviewPhase = "reviewing";
1975
- const coverage = this.reviewCoverage.evidence();
1976
- const lspEvidence = this.reviewLspEvidence(coverage.expected);
1977
- this.eventBus.emit("hook", {
1978
- kind: "ideal",
1979
- coverageExpected: coverage.expected,
1980
- coverageMissing: coverage.missing,
1981
- });
1982
- // Recompute strictly AFTER the hook event: clients release held text on
1983
- // disarm, so the reverse order would paint the draft and then delete it —
1984
- // the exact flash arming exists to prevent. Arming normally PERSISTS here,
1985
- // because review starts with every changed file uncovered and a stop while
1986
- // coverage is outstanding injects again. Disarm lands later, on the read
1987
- // that closes the last gap (or when the retry budget escalates).
1988
- this.refreshIdealReviewArmed();
1989
- // Announce the phase before the reviewer starts, not after its bounded wait.
1990
- const independentMessages = await this.runIndependentReview(decision);
1991
- log("INFO", "ideal", "Injecting ideal review before final response", {
1992
- coverageExpected: coverage.expected,
1993
- coverageMissing: coverage.missing,
1994
- lspLowConfidence: lspEvidence.lowConfidence,
1995
- lspMissing: lspEvidence.missing,
1996
- });
1997
- return [
1998
- ...independentMessages,
1999
- this.withReviewLspEvidence(withReviewCoverageRequirements(buildIdealReviewMessage(decision.reasons, driftedFiles), coverage.missing), lspEvidence),
2000
- ];
2001
- }
2002
- reviewLspEvidence(files) {
2003
- const lowConfidence = [];
2004
- const missing = [];
2005
- for (const filePath of files) {
2006
- const outcome = this.lspManager?.getLatestOutcome(filePath);
2007
- if (outcome?.kind === "low_confidence")
2008
- lowConfidence.push(filePath);
2009
- else if (outcome?.kind !== "clean" && outcome?.kind !== "diagnostics")
2010
- missing.push(filePath);
2011
- }
2012
- return { lowConfidence, missing };
2013
- }
2014
- withReviewLspEvidence(message, evidence) {
2015
- if (evidence.lowConfidence.length === 0 && evidence.missing.length === 0)
2016
- return message;
2017
- const notes = [
2018
- ...(evidence.lowConfidence.length > 0
2019
- ? [`Diagnostics are low confidence while indexing: ${evidence.lowConfidence.join(", ")}.`]
2020
- : []),
2021
- ...(evidence.missing.length > 0
2022
- ? [`Diagnostics evidence is unavailable or missing: ${evidence.missing.join(", ")}.`]
2023
- : []),
2024
- "Do not describe those files as compiler-clean without other evidence.",
2025
- ];
2026
- return {
2027
- role: "user",
2028
- provenance: message.provenance,
2029
- content: `${String(message.content)}\n\n${notes.join(" ")}`,
2030
- };
2031
1840
  }
2032
1841
  /** Auto-compact if needed, run agent loop with auth retry, and persist messages. */
2033
- async runLoop(options = {}) {
2034
- // Languages are re-detected at each task boundary so a project scaffolded
2035
- // during the previous turn gets its packs; the prompt is rebuilt only when
2036
- // the set grows, keeping the cached prefix stable otherwise.
2037
- if (this.refreshActiveLanguages())
2038
- await this.rebuildSystemPromptInPlace();
2039
- this.refreshSystemPromptTail();
1842
+ async runLoopInner(options = {}) {
1843
+ await this.prepareSystemPromptForRequest();
2040
1844
  // One-shot cache-key marker per session so turn_end cacheRead numbers
2041
1845
  // in the log can be traced back to a specific routing namespace —
2042
1846
  // particularly useful when sub-agents inherit `parentKey:subagent`.
@@ -2135,11 +1939,14 @@ export class AgentSession {
2135
1939
  // so the new value transparently yields a correctly-identified client.
2136
1940
  let userAgent = this.provider === "anthropic" ? await getClaudeCliUserAgent() : undefined;
2137
1941
  const loopMessages = await this.prepareDynamicContext();
1942
+ // Re-read every run so rule-file edits apply on the next turn; a few small files.
1943
+ const streamRules = await loadStreamRules(this.cwd);
2138
1944
  const runAgentLoop = async (apiKey, accountId, projectId) => {
2139
1945
  lastResolvedAccessToken = apiKey;
2140
1946
  const modelInfo = getModel(this.model);
2141
1947
  const effectiveBaseUrl = this.baseUrl ?? creds.baseUrl;
2142
1948
  const generator = agentLoop(loopMessages, {
1949
+ ...(streamRules.length > 0 ? { streamRules: { rules: streamRules } } : {}),
2143
1950
  provider: this.provider,
2144
1951
  model: this.model,
2145
1952
  tools: options.disableTools ? [] : this.tools,
@@ -2187,6 +1994,33 @@ export class AgentSession {
2187
1994
  // + pre-warm before the first turn. "baseline": current 5-min default.
2188
1995
  cacheRetention: this.isSpeedOptimized() ? "long" : "short",
2189
1996
  promptCacheKey: this.getPromptCacheKey(),
1997
+ onContextPrepared: (context) => {
1998
+ const report = this.cacheDiagnostics.prepare(context, {
1999
+ provider: this.provider,
2000
+ model: this.model,
2001
+ at: Date.now(),
2002
+ cacheRetention: this.isSpeedOptimized() ? "long" : "short",
2003
+ route: { baseUrl: effectiveBaseUrl, accountId: this.lastAccountId ?? accountId },
2004
+ settings: {
2005
+ thinking: this.planModeRef.current || options.capThinking
2006
+ ? clampThinkingForPlanMode(this.thinkingLevel)
2007
+ : this.thinkingLevel,
2008
+ webSearch: !options.disableTools,
2009
+ supportsImages: modelInfo?.supportsImages,
2010
+ promptCacheKey: this.getPromptCacheKey(),
2011
+ },
2012
+ });
2013
+ log("INFO", "cache", "Prepared context", {
2014
+ sessionId: this.sessionId || this.transportSessionId,
2015
+ data: JSON.stringify(report),
2016
+ });
2017
+ if (report.thinkingPrefixRiskBlocks > 0) {
2018
+ log("WARN", "cache", "Possible signed-thinking prefix mismatch; not server verified", {
2019
+ sessionId: this.sessionId || this.transportSessionId,
2020
+ blocks: String(report.thinkingPrefixRiskBlocks),
2021
+ });
2022
+ }
2023
+ },
2190
2024
  supportsImages: modelInfo?.supportsImages,
2191
2025
  supportsVideo: modelInfo?.supportsVideo,
2192
2026
  userAgent,
@@ -2195,9 +2029,15 @@ export class AgentSession {
2195
2029
  maxToolResultChars: resolveSessionToolResultCharLimit(this.model, this.provider, accountId),
2196
2030
  // Aggregate per-turn budget across parallel tool results (fan-out guard).
2197
2031
  maxTurnToolResultChars: resolveSessionTurnToolResultCharLimit(this.model, this.provider, accountId),
2032
+ // Warn when web/MCP output contains instruction-like text (see injection-detect.ts).
2033
+ transformToolResult: flagUntrustedToolResult,
2198
2034
  // Self-correction hooks (same as the TUI): loop-break + re-grounding are
2199
2035
  // polled mid-loop; the ideal review is polled when the agent would stop.
2200
2036
  getSteeringMessages: () => this.getHookSteeringMessages(),
2037
+ onSteeringAvailable: (listener) => {
2038
+ this.steeringListeners.add(listener);
2039
+ return () => this.steeringListeners.delete(listener);
2040
+ },
2201
2041
  getFollowUpMessages: () => this.getHookFollowUpMessages(),
2202
2042
  onTurnBudgetExhausted: (ctx) => this.shouldExtendTurnBudget(ctx),
2203
2043
  // Check authoritative provider usage before every model/tool step.
@@ -2223,6 +2063,7 @@ export class AgentSession {
2223
2063
  // retained usage afterwards since it counted the pruned content.
2224
2064
  const pruneResult = pruneStaleToolResults(messages);
2225
2065
  if (pruneResult.pruned) {
2066
+ this.cacheDiagnostics.noteEdit("tool_prune", pruneResult.freedTokens);
2226
2067
  this.providerContext = null;
2227
2068
  log("INFO", "compaction", "Pruned stale tool outputs", {
2228
2069
  prunedResults: String(pruneResult.prunedResults),
@@ -2263,6 +2104,9 @@ export class AgentSession {
2263
2104
  usage,
2264
2105
  pendingMessages,
2265
2106
  });
2107
+ // An approved plan runs as ONE run, so step boundaries are only
2108
+ // visible here, between model steps — not after the run ends.
2109
+ const planStep = this.observePlanStepProgress(messages, contextWindow, activeTokens);
2266
2110
  log("INFO", "compaction", "In-flight compaction decision", {
2267
2111
  provider: this.provider,
2268
2112
  model: this.model,
@@ -2270,8 +2114,10 @@ export class AgentSession {
2270
2114
  contextWindow: String(contextWindow),
2271
2115
  activeTokens: String(activeTokens),
2272
2116
  triggerLimit: String(policy.targetTokens),
2117
+ ...(planStep ? { planStep: `${planStep.compact} (${planStep.reason})` } : {}),
2273
2118
  });
2274
- if (!shouldCompact(messages, contextWindow, policy.threshold, activeTokens, policy.targetTokens))
2119
+ if (!shouldCompact(messages, contextWindow, policy.threshold, activeTokens, policy.targetTokens) &&
2120
+ !planStep?.compact)
2275
2121
  return messages;
2276
2122
  }
2277
2123
  // compact() operates on this.messages, while an earlier transform may
@@ -2604,7 +2450,7 @@ export class AgentSession {
2604
2450
  const canonicalPath = await this.sessionManager.resolveCanonicalSession(this.conversationId, this.cwd);
2605
2451
  if (!canonicalPath || canonicalPath === this.sessionPath)
2606
2452
  return;
2607
- await this.adoptCompactionCheckpoint(await this.sessionManager.load(canonicalPath));
2453
+ await this.adoptCompactionCheckpoint(await this.sessionManager.load(canonicalPath, { canonical: true }));
2608
2454
  }
2609
2455
  async persistCompactionCheckpoint(sourceFingerprint, result) {
2610
2456
  const parentSessionId = this.sessionId || undefined;
@@ -2653,6 +2499,99 @@ export class AgentSession {
2653
2499
  if (this.postTurnCompaction)
2654
2500
  await this.postTurnCompaction;
2655
2501
  }
2502
+ /**
2503
+ * Advance plan-step bookkeeping over the messages added since the last
2504
+ * observation and, when a plan step was newly completed (`[DONE:n]`), run
2505
+ * the SoL-Pi cost rule. Called between model steps (in-flight) and once
2506
+ * after the run; `scanIndex` and `doneSteps` make each message and each
2507
+ * step count once across both paths. Returns undefined when no step
2508
+ * completed since the last observation.
2509
+ */
2510
+ observePlanStepProgress(messages, contextWindow, activeTokens) {
2511
+ const st = this.planStepState;
2512
+ const planPath = this.approvedPlanPath;
2513
+ if (st.planPath !== planPath) {
2514
+ st.planPath = planPath;
2515
+ st.doneSteps = new Set();
2516
+ st.requestsInCompletedSteps = 0;
2517
+ st.requestsThisStep = 0;
2518
+ }
2519
+ if (st.scanIndex > messages.length)
2520
+ st.scanIndex = 0;
2521
+ let requests = 0;
2522
+ const newlyDone = [];
2523
+ for (let i = st.scanIndex; i < messages.length; i++) {
2524
+ const msg = messages[i];
2525
+ if (msg?.role !== "assistant")
2526
+ continue;
2527
+ requests++;
2528
+ const text = typeof msg.content === "string"
2529
+ ? msg.content
2530
+ : msg.content.map((part) => (part.type === "text" ? part.text : "")).join("\n");
2531
+ for (const step of findCompletedMarkers(text)) {
2532
+ if (!st.doneSteps.has(step))
2533
+ newlyDone.push(step);
2534
+ }
2535
+ }
2536
+ st.scanIndex = messages.length;
2537
+ const contextTokens = activeTokens ?? estimateConversationTokens(messages);
2538
+ if (st.lastContextTokens > 0)
2539
+ st.grownTokens += Math.max(0, contextTokens - st.lastContextTokens);
2540
+ st.lastContextTokens = contextTokens;
2541
+ st.requests += requests;
2542
+ st.requestsThisStep += requests;
2543
+ st.requestsSinceLastCompaction += requests;
2544
+ if (st.writeCostBalance > 0)
2545
+ st.writeCostBalance -= st.savingPerRequest * requests;
2546
+ if (!planPath || newlyDone.length === 0)
2547
+ return undefined;
2548
+ for (const step of newlyDone)
2549
+ st.doneSteps.add(step);
2550
+ st.requestsInCompletedSteps += st.requestsThisStep;
2551
+ st.requestsThisStep = 0;
2552
+ let planText;
2553
+ try {
2554
+ planText = readFileSync(planPath, "utf8");
2555
+ }
2556
+ catch {
2557
+ return undefined;
2558
+ }
2559
+ const totalSteps = extractPlanSteps(planText).length;
2560
+ const decision = decidePlanStepCompaction({
2561
+ contextTokens,
2562
+ keptTailTokens: PLAN_STEP_KEEP_TOKENS,
2563
+ contextWindow,
2564
+ // The model registry carries no cache price fields; use the SoL-Pi default.
2565
+ cacheWriteReadRatio: DEFAULT_CACHE_WRITE_READ_RATIO,
2566
+ stepsCompleted: st.doneSteps.size,
2567
+ stepsRemaining: Math.max(0, totalSteps - st.doneSteps.size),
2568
+ requestsInCompletedSteps: st.requestsInCompletedSteps,
2569
+ requests: st.requests,
2570
+ grownTokens: st.grownTokens,
2571
+ priorCompactions: st.compactions,
2572
+ writeCostBalance: st.writeCostBalance,
2573
+ requestsSinceLastCompaction: st.compactions > 0 ? st.requestsSinceLastCompaction : undefined,
2574
+ });
2575
+ log("INFO", "compaction", "Plan-step compaction decision", {
2576
+ compact: String(decision.compact),
2577
+ reason: decision.reason,
2578
+ });
2579
+ return decision;
2580
+ }
2581
+ /**
2582
+ * Plan-step bookkeeping after ANY successful compaction (as in the bench:
2583
+ * every compaction leaves a cache-write cost to be repaid by later savings).
2584
+ */
2585
+ recordPlanStepCompaction(contextBefore) {
2586
+ const st = this.planStepState;
2587
+ const after = estimateConversationTokens(this.messages);
2588
+ st.scanIndex = this.messages.length;
2589
+ st.lastContextTokens = after;
2590
+ st.requestsSinceLastCompaction = 0;
2591
+ st.compactions++;
2592
+ st.writeCostBalance += after * (DEFAULT_CACHE_WRITE_READ_RATIO - 1);
2593
+ st.savingPerRequest = Math.max(0, contextBefore - after);
2594
+ }
2656
2595
  /**
2657
2596
  * Post-turn compaction: once the final response has been delivered, compact
2658
2597
  * in the background while the user reads the answer, instead of making the
@@ -2661,7 +2600,9 @@ export class AgentSession {
2661
2600
  * Codex `model_post_turn_compact_threshold_percent` guards: skip when user
2662
2601
  * input is already queued (it would race the next turn), when the run was
2663
2602
  * aborted, or during the failure cooldown — and never let a compaction
2664
- * error surface in the completed turn.
2603
+ * error surface in the completed turn. Besides the size trigger, a newly
2604
+ * completed approved-plan step may compact when the cache-cost rule in
2605
+ * compaction/plan-step-policy.ts says the shrink pays for itself.
2665
2606
  */
2666
2607
  maybeCompactPostTurn(creds) {
2667
2608
  if (!this.settingsManager.get("autoCompact"))
@@ -2674,15 +2615,18 @@ export class AgentSession {
2674
2615
  return;
2675
2616
  if (Date.now() < this.compactionRetryAfter)
2676
2617
  return;
2677
- // One compaction per turn boundary: a pre-run or overflow-recovery
2678
- // compaction already shrank this run's history — re-probing right after
2679
- // the final response would only re-derive that decision.
2680
- if (this.compactionOccurred)
2681
- return;
2682
2618
  const contextWindow = getContextWindow(this.model, {
2683
2619
  provider: this.provider,
2684
2620
  accountId: creds.accountId,
2685
2621
  });
2622
+ // One compaction per turn boundary: a pre-run, in-flight or overflow
2623
+ // compaction already shrank this run's history — re-probing right after
2624
+ // the final response would only re-derive that decision. Still record the
2625
+ // final response's `[DONE:n]` steps so plan bookkeeping stays current.
2626
+ if (this.compactionOccurred) {
2627
+ this.observePlanStepProgress(this.messages, contextWindow, undefined);
2628
+ return;
2629
+ }
2686
2630
  const policy = resolveCompactionPolicy({
2687
2631
  provider: this.provider,
2688
2632
  model: this.model,
@@ -2701,9 +2645,12 @@ export class AgentSession {
2701
2645
  });
2702
2646
  }
2703
2647
  }
2704
- if (!shouldCompact(this.messages, contextWindow, policy.threshold, activeTokens, policy.targetTokens))
2648
+ const planStep = this.observePlanStepProgress(this.messages, contextWindow, activeTokens);
2649
+ if (!shouldCompact(this.messages, contextWindow, policy.threshold, activeTokens, policy.targetTokens) &&
2650
+ !planStep?.compact)
2705
2651
  return;
2706
2652
  log("INFO", "compaction", "Post-turn compaction decision — compacting in background", {
2653
+ trigger: planStep?.compact ? `plan-step (${planStep.reason})` : "size",
2707
2654
  provider: this.provider,
2708
2655
  model: this.model,
2709
2656
  transport: this.provider === "openai" && creds.accountId ? "codex_oauth" : "public_api",
@@ -2757,6 +2704,7 @@ export class AgentSession {
2757
2704
  approvedPlanPath: this.approvedPlanPath,
2758
2705
  });
2759
2706
  const originalCount = this.messages.length;
2707
+ const contextTokensBefore = estimateConversationTokens(this.messages);
2760
2708
  this.eventBus.emit("compaction_start", { messageCount: originalCount });
2761
2709
  let contextSelection;
2762
2710
  const runCompactor = async () => {
@@ -2794,7 +2742,7 @@ export class AgentSession {
2794
2742
  let sourceFingerprint = computeSourceFingerprint(this.messages);
2795
2743
  const canonicalPath = await this.sessionManager.resolveCanonicalSession(conversationId, this.cwd);
2796
2744
  if (canonicalPath && canonicalPath !== this.sessionPath) {
2797
- const newest = await this.sessionManager.load(canonicalPath);
2745
+ const newest = await this.sessionManager.load(canonicalPath, { canonical: true });
2798
2746
  if (newest.header.sourceFingerprint === sourceFingerprint) {
2799
2747
  await this.adoptCompactionCheckpoint(newest);
2800
2748
  this.lastCompactionCompacted = true;
@@ -2868,6 +2816,10 @@ export class AgentSession {
2868
2816
  }
2869
2817
  });
2870
2818
  }
2819
+ if (this.lastCompactionCompacted) {
2820
+ this.cacheDiagnostics.noteEdit("compaction");
2821
+ this.recordPlanStepCompaction(contextTokensBefore);
2822
+ }
2871
2823
  this.eventBus.emit("compaction_end", {
2872
2824
  compacted: this.lastCompactionCompacted,
2873
2825
  originalCount,
@@ -2887,6 +2839,7 @@ export class AgentSession {
2887
2839
  });
2888
2840
  }
2889
2841
  async newSession(preserveConversation = false) {
2842
+ this.cacheDiagnostics.reset();
2890
2843
  // Approved-plan execution is a clean checkpoint of the same conversation;
2891
2844
  // explicit new sessions reset the conversation identity.
2892
2845
  if (!preserveConversation) {
@@ -2934,6 +2887,7 @@ export class AgentSession {
2934
2887
  }
2935
2888
  async loadSession(sessionPath) {
2936
2889
  await this.loadExistingSession(sessionPath);
2890
+ this.cacheDiagnostics.reset();
2937
2891
  if (this.sessionId)
2938
2892
  await this.subAgentManager?.hydrate(this.sessionId);
2939
2893
  this.eventBus.emit("session_start", { sessionId: this.sessionId });
@@ -2964,6 +2918,7 @@ export class AgentSession {
2964
2918
  const branchMessages = this.sessionManager.getMessages(loaded.entries, this.currentLeafId);
2965
2919
  const systemMsg = this.messages[0];
2966
2920
  this.messages = [systemMsg, ...branchMessages];
2921
+ this.cacheDiagnostics.reset();
2967
2922
  this.lastPersistedIndex = this.messages.length;
2968
2923
  // Reads made in the dropped messages are no longer in the model's context.
2969
2924
  this.clearReadTracker?.();
@@ -3031,23 +2986,33 @@ export class AgentSession {
3031
2986
  : undefined;
3032
2987
  return costUsd === undefined ? { used, size } : { used, size, costUsd };
3033
2988
  }
3034
- getPlanMode() {
3035
- return this.planModeRef.current;
3036
- }
3037
2989
  /**
3038
- * Suppress only the pre-final Ideal self-review for this live session.
3039
- * Autopilot uses this while Nolan independently owns verification; loop-break
3040
- * and post-compaction re-grounding remain active.
2990
+ * Whether the provider's prompt cache has likely lapsed since the last
2991
+ * successful request, and how many tokens the next message would re-read at
2992
+ * full price. Null when the route has no known TTL or the chat is empty.
3041
2993
  */
3042
- setIdealReviewSuppressed(suppressed) {
3043
- this.idealReviewSuppressed = suppressed;
3044
- if (suppressed) {
3045
- this.idealReviewPhase = "idle";
3046
- this.reviewCoverage.reset();
3047
- }
3048
- // Suppression flips mid-run (autopilot takes over verification), so a client
3049
- // holding a draft under a stale arming must be released.
3050
- this.refreshHookArming();
2994
+ getCacheExpiryStatus(now = Date.now()) {
2995
+ const status = assessCacheExpiry({
2996
+ current: {
2997
+ provider: this.provider,
2998
+ model: this.model,
2999
+ policy: resolveCacheTtl({
3000
+ provider: this.provider,
3001
+ model: this.model,
3002
+ cacheRetention: this.isSpeedOptimized() ? "long" : "short",
3003
+ baseUrl: this.baseUrl,
3004
+ accountId: this.lastAccountId,
3005
+ }),
3006
+ },
3007
+ lastTouch: this.cacheDiagnostics.lastCacheTouch(),
3008
+ now,
3009
+ prefixTokens: this.getContextUsage().used,
3010
+ hasHistory: this.messages.some((m) => m.role === "user"),
3011
+ });
3012
+ return status && { ...status, sessionId: this.sessionId || this.transportSessionId };
3013
+ }
3014
+ getPlanMode() {
3015
+ return this.planModeRef.current;
3051
3016
  }
3052
3017
  /** Queue a user message (optionally with attachments) to be injected mid-run
3053
3018
  * as steering. Returns the new queue length. No-op semantics are the caller's
@@ -3055,6 +3020,9 @@ export class AgentSession {
3055
3020
  queueMessage(text, attachments = []) {
3056
3021
  this.queueSeq += 1;
3057
3022
  this.userQueue.push({ id: `q${this.queueSeq}`, text, attachments });
3023
+ // Instant interrupt: preempt running tools so the steer lands right away.
3024
+ for (const listener of [...this.steeringListeners])
3025
+ listener();
3058
3026
  return this.userQueue.length;
3059
3027
  }
3060
3028
  /** Pending queued messages (id + text), oldest first, for client display. */
@@ -3107,6 +3075,15 @@ export class AgentSession {
3107
3075
  return `No background process with id "${id}"`;
3108
3076
  return this.processManager.stop(id);
3109
3077
  }
3078
+ /**
3079
+ * Force-stop every background process tree, synchronously. Background
3080
+ * commands run in their own process group, so the daemon's group kill on
3081
+ * quit never reaches them: this is the only thing that does. Callers on a
3082
+ * shutdown deadline run it before awaiting anything that can hang.
3083
+ */
3084
+ stopBackgroundProcesses() {
3085
+ this.processManager?.shutdownAll();
3086
+ }
3110
3087
  /** Replace a host-owned system prompt in place without resetting conversation history. */
3111
3088
  setCustomSystemPrompt(systemPrompt, promptCacheKeyPrefix) {
3112
3089
  this.customSystemPrompt = systemPrompt;
@@ -3253,6 +3230,18 @@ export class AgentSession {
3253
3230
  * the standard prompt therefore needs a rebuild. Custom and sub-agent
3254
3231
  * prompts never render packs, so detection is skipped for them.
3255
3232
  */
3233
+ /**
3234
+ * Bring the system prompt to the exact state the next provider request will
3235
+ * send. Shared by real runs and {@link prewarm} so both see one prefix.
3236
+ */
3237
+ async prepareSystemPromptForRequest() {
3238
+ // Languages are re-detected at each task boundary so a project scaffolded
3239
+ // during the previous turn gets its packs; the prompt is rebuilt only when
3240
+ // the set grows, keeping the cached prefix stable otherwise.
3241
+ if (this.refreshActiveLanguages())
3242
+ await this.rebuildSystemPromptInPlace();
3243
+ this.refreshSystemPromptTail();
3244
+ }
3256
3245
  refreshActiveLanguages() {
3257
3246
  if (this.customSystemPrompt || this.agentPrompt !== undefined)
3258
3247
  return false;
@@ -3351,6 +3340,17 @@ export class AgentSession {
3351
3340
  }));
3352
3341
  }
3353
3342
  async persistTurnMetric(event) {
3343
+ if (event.stopReason === "error")
3344
+ this.cacheDiagnostics.discardAttempt();
3345
+ const cache = this.cacheDiagnostics.complete(event.usage, event.timing);
3346
+ if (cache) {
3347
+ log("INFO", "cache", "Context cache outcome", {
3348
+ sessionId: this.sessionId || this.transportSessionId,
3349
+ provider: this.provider,
3350
+ model: this.model,
3351
+ data: JSON.stringify(cache),
3352
+ });
3353
+ }
3354
3354
  const payload = {
3355
3355
  version: 1,
3356
3356
  turn: event.turn,
@@ -3736,6 +3736,138 @@ export class AgentSession {
3736
3736
  if (signal?.aborted)
3737
3737
  this.managerAbortHandler();
3738
3738
  }
3739
+ runLoopDepth = 0;
3740
+ lastRealRequestAt = 0;
3741
+ lastPrewarmAt = 0;
3742
+ prewarmController = null;
3743
+ /**
3744
+ * Best-effort Anthropic prompt-cache prewarm before the user's next turn
3745
+ * (desktop app calls this on the first keystroke after opening a chat or an
3746
+ * idle pause). Sends the exact request prefix the next real turn will use —
3747
+ * same system/tools/thinking/cache options — with `max_tokens: 1`, so the
3748
+ * first real reply is a cache read instead of a cold write.
3749
+ */
3750
+ async prewarm(signal) {
3751
+ if (this.provider !== "anthropic")
3752
+ return { ok: false, reason: "provider" };
3753
+ if (this.settingsManager?.get("cachePrewarm") === false) {
3754
+ return { ok: false, reason: "disabled" };
3755
+ }
3756
+ if (this.runLoopDepth > 0)
3757
+ return { ok: false, reason: "run_active" };
3758
+ if (this.prewarmController)
3759
+ return { ok: false, reason: "in_flight" };
3760
+ const cacheRetention = this.isSpeedOptimized() ? "long" : "short";
3761
+ const ttlMs = (cacheRetention === "long" ? 60 : 5) * 60_000;
3762
+ const now = Date.now();
3763
+ if (now - Math.max(this.lastPrewarmAt, this.lastRealRequestAt) < ttlMs) {
3764
+ return { ok: false, reason: "cache_fresh" };
3765
+ }
3766
+ // Same preparation as the real run, BEFORE copying history: otherwise the
3767
+ // warmed system block differs from the next request (language packs are
3768
+ // detected at run start) and everything after it misses the cache.
3769
+ await this.prepareSystemPromptForRequest();
3770
+ // The await above yields: re-check that no run or other prewarm began.
3771
+ if (this.runLoopDepth > 0)
3772
+ return { ok: false, reason: "run_active" };
3773
+ if (this.prewarmController)
3774
+ return { ok: false, reason: "in_flight" };
3775
+ // End at the last user/tool message: the next real turn appends a new user
3776
+ // message after the trailing assistant reply, and its cache lookback hits
3777
+ // the entry written at this boundary. A request must end on a user turn.
3778
+ let end = this.messages.length;
3779
+ while (end > 0 && this.messages[end - 1]?.role === "assistant")
3780
+ end--;
3781
+ // The loop repairs tool pairing in place before every request; apply the
3782
+ // same repair to a copy so the warmed prefix is byte-identical to it.
3783
+ const messages = structuredClone(this.messages.slice(0, end));
3784
+ repairToolPairingAdjacent(messages);
3785
+ if (!messages.some((m) => m.role === "user" || m.role === "tool")) {
3786
+ return { ok: false, reason: "no_history" };
3787
+ }
3788
+ const tokens = estimateConversationTokens(messages);
3789
+ if (tokens < 4_000)
3790
+ return { ok: false, reason: "too_small" };
3791
+ const controller = new AbortController();
3792
+ const onAbort = () => controller.abort();
3793
+ signal?.addEventListener("abort", onAbort, { once: true });
3794
+ this.prewarmController = controller;
3795
+ const started = Date.now();
3796
+ try {
3797
+ const creds = await this.authStorage.resolveCredentials(this.provider, {
3798
+ storageKeys: this.currentAuthStorageKeys(),
3799
+ });
3800
+ if (controller.signal.aborted || this.runLoopDepth > 0) {
3801
+ return { ok: false, reason: "aborted" };
3802
+ }
3803
+ const modelInfo = getModel(this.model);
3804
+ const result = stream({
3805
+ provider: this.provider,
3806
+ model: this.model,
3807
+ messages,
3808
+ tools: this.tools,
3809
+ webSearch: true,
3810
+ maxTokens: this.maxTokens,
3811
+ thinking: this.planModeRef.current
3812
+ ? clampThinkingForPlanMode(this.thinkingLevel)
3813
+ : this.thinkingLevel,
3814
+ apiKey: creds.accessToken,
3815
+ baseUrl: this.baseUrl ?? creds.baseUrl,
3816
+ accountId: creds.accountId,
3817
+ transportSessionId: this.sessionId || this.transportSessionId,
3818
+ cacheRetention,
3819
+ promptCacheKey: this.getPromptCacheKey(),
3820
+ supportsImages: modelInfo?.supportsImages,
3821
+ supportsVideo: modelInfo?.supportsVideo,
3822
+ userAgent: await getClaudeCliUserAgent(),
3823
+ prewarm: true,
3824
+ signal: controller.signal,
3825
+ });
3826
+ const response = await result.response;
3827
+ const usage = response.usage;
3828
+ if (usage.inputTokens === 0 && usage.outputTokens === 0) {
3829
+ // @prestyj/ai sent nothing: budget thinking can't stay identical at max_tokens 1.
3830
+ log("INFO", "prewarm", "Cache prewarm skipped: budget thinking", {
3831
+ model: this.model,
3832
+ });
3833
+ return { ok: false, reason: "thinking_budget_incompatible" };
3834
+ }
3835
+ this.lastPrewarmAt = Date.now();
3836
+ const warmedPolicy = resolveCacheTtl({
3837
+ provider: this.provider,
3838
+ model: this.model,
3839
+ cacheRetention,
3840
+ baseUrl: this.baseUrl ?? creds.baseUrl,
3841
+ accountId: creds.accountId,
3842
+ });
3843
+ if (warmedPolicy) {
3844
+ this.cacheDiagnostics.noteCacheTouch({
3845
+ at: started,
3846
+ provider: this.provider,
3847
+ model: this.model,
3848
+ policy: warmedPolicy,
3849
+ });
3850
+ }
3851
+ log("INFO", "prewarm", "Cache prewarm complete", {
3852
+ tokens: String(tokens),
3853
+ cacheRead: String(usage.cacheRead ?? 0),
3854
+ cacheWrite: String(usage.cacheWrite ?? 0),
3855
+ ms: String(Date.now() - started),
3856
+ });
3857
+ return { ok: true, reason: "warmed", usage };
3858
+ }
3859
+ catch (error) {
3860
+ if (isAbortError(error) || controller.signal.aborted)
3861
+ return { ok: false, reason: "aborted" };
3862
+ log("WARN", "prewarm", `Cache prewarm failed: ${error instanceof Error ? error.message : String(error)}`);
3863
+ return { ok: false, reason: "error" };
3864
+ }
3865
+ finally {
3866
+ signal?.removeEventListener("abort", onAbort);
3867
+ if (this.prewarmController === controller)
3868
+ this.prewarmController = null;
3869
+ }
3870
+ }
3739
3871
  /** True when speedProfile is "optimized" (1-h cache TTL + pre-warm), or the
3740
3872
  * session was constructed with `forceLongCacheRetention` (Nolan sessions). */
3741
3873
  isSpeedOptimized() {
@@ -3764,16 +3896,22 @@ export class AgentSession {
3764
3896
  return this.getPromptCacheKey();
3765
3897
  }
3766
3898
  async dispose() {
3899
+ // First and synchronous: nothing below may delay this, or a hung teardown
3900
+ // step leaves background commands running after the app has quit.
3901
+ this.stopBackgroundProcesses();
3767
3902
  // Quiesce any in-flight post-turn compaction BEFORE tearing down state:
3768
3903
  // the background compact() snapshots and replaces `this.messages`, so
3769
3904
  // letting it run past this point would checkpoint a near-empty history
3770
3905
  // and leak a junk session file after teardown.
3771
3906
  if (this.postTurnCompaction)
3772
3907
  await this.postTurnCompaction;
3908
+ this.cacheDiagnostics.reset();
3773
3909
  this.diagnosticsRecorder?.finalize();
3774
3910
  this.managerAbortSignal?.removeEventListener("abort", this.managerAbortHandler);
3775
- this.processManager?.shutdownAll();
3911
+ // Again, in case a turn racing teardown started one while we awaited.
3912
+ this.stopBackgroundProcesses();
3776
3913
  this.lspManager?.shutdownAll();
3914
+ this.debugManager?.shutdown();
3777
3915
  await Promise.all([this.subAgentManager?.shutdownAll(), this.mcpManager?.dispose()]);
3778
3916
  await this.extensionLoader.deactivateAll();
3779
3917
  this.setSessionPath("");
@@ -3809,8 +3947,10 @@ export class AgentSession {
3809
3947
  // A stale physical checkpoint is only an address, not the conversation tip.
3810
3948
  // Resolve every resume—not just over-threshold/deferred compaction resumes—
3811
3949
  // before reading history so the next prompt cannot continue an old branch.
3812
- const canonicalPath = (await this.sessionManager.resolveCanonicalSession(sessionPath, this.cwd)) ?? sessionPath;
3813
- const loaded = await this.sessionManager.load(canonicalPath);
3950
+ const resolvedPath = await this.sessionManager.resolveCanonicalSession(sessionPath, this.cwd);
3951
+ const loaded = resolvedPath
3952
+ ? await this.sessionManager.load(resolvedPath, { canonical: true })
3953
+ : await this.sessionManager.load(sessionPath);
3814
3954
  // Use the leaf from the header to walk the correct branch
3815
3955
  const loadedMessages = this.sessionManager.getMessages(loaded.entries, loaded.header.leafId);
3816
3956
  const savedCompletionReview = [...loaded.entries]