@prestyj/cli 5.28.0 → 5.29.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (398) hide show
  1. package/README.md +2 -2
  2. package/assets/motion/THIRD-PARTY.md +14 -19
  3. package/assets/motion/bin/contact-sheet.mjs +9 -3
  4. package/assets/motion/bin/cues.mjs +337 -0
  5. package/assets/motion/bin/flash-check.mjs +314 -0
  6. package/assets/motion/bin/fonts.mjs +17 -13
  7. package/assets/motion/bin/library.mjs +61 -18
  8. package/assets/motion/bin/motion-blur.mjs +943 -0
  9. package/assets/motion/bin/motion-check.mjs +366 -9
  10. package/assets/motion/bin/music-fit.mjs +436 -0
  11. package/assets/motion/bin/pdf-extract.mjs +22 -10
  12. package/assets/motion/bin/reference-study.mjs +346 -0
  13. package/assets/motion/bin/score-synth.mjs +1052 -93
  14. package/assets/motion/fonts/finger-paint/OFL.txt +93 -0
  15. package/assets/motion/fonts/finger-paint/finger-paint-normal.woff2 +0 -0
  16. package/assets/motion/fonts/fonts.json +56 -0
  17. package/assets/motion/fonts/short-stack/OFL.txt +94 -0
  18. package/assets/motion/fonts/short-stack/short-stack-normal.woff2 +0 -0
  19. package/assets/motion/fonts/sora/OFL.txt +93 -0
  20. package/assets/motion/fonts/sora/sora-normal.woff2 +0 -0
  21. package/assets/motion/fonts/specimen.jpg +0 -0
  22. package/assets/motion/fonts/unbounded/OFL.txt +93 -0
  23. package/assets/motion/fonts/unbounded/unbounded-normal.woff2 +0 -0
  24. package/assets/motion/library/README.md +52 -14
  25. package/assets/motion/library/kit/moves.js +1981 -0
  26. package/assets/motion/library/library.json +232 -0
  27. package/assets/motion/library/pieces/camera-rig/meta.json +13 -0
  28. package/assets/motion/library/pieces/camera-rig/piece.html +153 -0
  29. package/assets/motion/library/pieces/camera-rig/preview.jpg +0 -0
  30. package/assets/motion/library/pieces/chain-knock/meta.json +13 -0
  31. package/assets/motion/library/pieces/chain-knock/piece.html +195 -0
  32. package/assets/motion/library/pieces/chain-knock/preview.jpg +0 -0
  33. package/assets/motion/library/pieces/gather-to-logo/meta.json +13 -0
  34. package/assets/motion/library/pieces/gather-to-logo/piece.html +159 -0
  35. package/assets/motion/library/pieces/gather-to-logo/preview.jpg +0 -0
  36. package/assets/motion/library/pieces/morph-carry/meta.json +13 -0
  37. package/assets/motion/library/pieces/morph-carry/piece.html +173 -0
  38. package/assets/motion/library/pieces/morph-carry/preview.jpg +0 -0
  39. package/assets/motion/library/pieces/one-shape-journey/meta.json +13 -0
  40. package/assets/motion/library/pieces/one-shape-journey/piece.html +195 -0
  41. package/assets/motion/library/pieces/one-shape-journey/preview.jpg +0 -0
  42. package/assets/motion/library/pieces/open-from-subject/meta.json +13 -0
  43. package/assets/motion/library/pieces/open-from-subject/piece.html +168 -0
  44. package/assets/motion/library/pieces/open-from-subject/preview.jpg +0 -0
  45. package/assets/motion/library/pieces/request-to-result/meta.json +13 -0
  46. package/assets/motion/library/pieces/request-to-result/piece.html +212 -0
  47. package/assets/motion/library/pieces/request-to-result/preview.jpg +0 -0
  48. package/assets/motion/library/pieces/scale-dive/meta.json +13 -0
  49. package/assets/motion/library/pieces/scale-dive/piece.html +321 -0
  50. package/assets/motion/library/pieces/scale-dive/preview.jpg +0 -0
  51. package/assets/motion/library/pieces/screen-replica-steps/meta.json +13 -0
  52. package/assets/motion/library/pieces/screen-replica-steps/piece.html +366 -0
  53. package/assets/motion/library/pieces/screen-replica-steps/preview.jpg +0 -0
  54. package/assets/motion/library/pieces/zoom-into-card/meta.json +13 -0
  55. package/assets/motion/library/pieces/zoom-into-card/piece.html +179 -0
  56. package/assets/motion/library/pieces/zoom-into-card/preview.jpg +0 -0
  57. package/assets/motion/library/sheets/diagram.jpg +0 -0
  58. package/assets/motion/library/sheets/frame.jpg +0 -0
  59. package/assets/motion/library/sheets/transition.jpg +0 -0
  60. package/assets/motion/library/sheets/ui.jpg +0 -0
  61. package/assets/motion/plugin.json +1 -1
  62. package/assets/motion/references/build-sheet.md +206 -0
  63. package/assets/motion/references/runtime/determinism-rules.md +3 -3
  64. package/assets/motion/references/runtime/gsap-easing-and-stagger.md +28 -28
  65. package/assets/motion/references/runtime/gsap.md +3 -3
  66. package/assets/motion/references/runtime/inputs-and-assets.md +21 -16
  67. package/assets/motion/references/runtime/lint-validate-inspect.md +3 -3
  68. package/assets/motion/references/runtime/minimal-composition.md +6 -0
  69. package/assets/motion/references/runtime/preview-render.md +3 -3
  70. package/assets/motion/skills/app-walkthrough/SKILL.md +66 -0
  71. package/assets/motion/skills/before-after/SKILL.md +53 -0
  72. package/assets/motion/skills/brand-kit/SKILL.md +17 -16
  73. package/assets/motion/skills/dev-tool-video/SKILL.md +57 -0
  74. package/assets/motion/skills/launch-video/SKILL.md +62 -0
  75. package/assets/motion/skills/match-reference/SKILL.md +58 -0
  76. package/assets/motion/skills/motion/SKILL.md +167 -98
  77. package/assets/motion/skills/source-ingest/SKILL.md +23 -15
  78. package/assets/motion/skills/website-video/SKILL.md +59 -0
  79. package/assets/skills/bulletproof/SKILL.md +36 -11
  80. package/assets/skills/bulletproof/references/agent-surface.md +19 -9
  81. package/assets/skills/bulletproof/references/audit-protocol.md +20 -5
  82. package/assets/skills/bulletproof/references/platform-playbooks.md +5 -4
  83. package/assets/skills/bulletproof/references/provenance.md +26 -1
  84. package/assets/skills/bulletproof/references/secure-defaults.md +6 -5
  85. package/assets/skills/bulletproof/references/supply-chain.md +21 -17
  86. package/assets/skills/bulletproof/references/threat-landscape.md +28 -26
  87. package/assets/skills/bulletproof/references/verification.md +2 -0
  88. package/assets/skills/clarify/SKILL.md +25 -16
  89. package/assets/skills/code-review/SKILL.md +71 -13
  90. package/assets/skills/code-review/references/agent-diffs.md +27 -0
  91. package/assets/skills/code-review/references/tests.md +19 -0
  92. package/assets/skills/compliance-guard/SKILL.md +20 -5
  93. package/assets/skills/compliance-guard/references/artifacts.md +1 -1
  94. package/assets/skills/compliance-guard/references/eu-uk.md +16 -16
  95. package/assets/skills/compliance-guard/references/lawsuit-vectors.md +5 -5
  96. package/assets/skills/compliance-guard/references/provenance.md +41 -2
  97. package/assets/skills/compliance-guard/references/sector-gates.md +3 -3
  98. package/assets/skills/compliance-guard/references/security-baseline.md +2 -2
  99. package/assets/skills/compliance-guard/references/trigger-map.md +5 -5
  100. package/assets/skills/compliance-guard/references/us.md +27 -21
  101. package/assets/skills/durable/SKILL.md +87 -79
  102. package/assets/skills/durable/references/agent-db-safety.md +69 -0
  103. package/assets/skills/durable/references/backups-and-runtime.md +19 -12
  104. package/assets/skills/durable/references/migrations-and-schema.md +13 -6
  105. package/assets/skills/evidence-led-ui/SKILL.md +69 -127
  106. package/assets/skills/evidence-led-ui/references/anti-defaults.md +107 -208
  107. package/assets/skills/evidence-led-ui/references/direction.md +124 -0
  108. package/assets/skills/evidence-led-ui/references/production-contract.md +8 -0
  109. package/assets/skills/evidence-led-ui/references/provenance.md +24 -1
  110. package/assets/skills/lean/SKILL.md +90 -71
  111. package/assets/skills/lean/references/memory-and-processes.md +3 -2
  112. package/assets/skills/lean/references/playbooks.md +37 -12
  113. package/assets/skills/refactoring/SKILL.md +24 -3
  114. package/assets/skills/refactoring/references/agent-pitfalls.md +4 -1
  115. package/assets/skills/refactoring/references/legacy.md +21 -0
  116. package/assets/skills/root-cause/SKILL.md +20 -10
  117. package/assets/skills/shared-language/SKILL.md +16 -14
  118. package/assets/skills/tdd/SKILL.md +27 -15
  119. package/dist/app-sidecar.js +203 -47
  120. package/dist/app-sidecar.js.map +1 -1
  121. package/dist/cli.js +20 -28
  122. package/dist/cli.js.map +1 -1
  123. package/dist/core/acceptance-checks.d.ts +48 -0
  124. package/dist/core/acceptance-checks.js +144 -0
  125. package/dist/core/acceptance-checks.js.map +1 -0
  126. package/dist/core/agent-session.d.ts +119 -75
  127. package/dist/core/agent-session.js +561 -395
  128. package/dist/core/agent-session.js.map +1 -1
  129. package/dist/core/agents.d.ts +6 -5
  130. package/dist/core/agents.js +12 -3
  131. package/dist/core/agents.js.map +1 -1
  132. package/dist/core/ask-user.d.ts +90 -8
  133. package/dist/core/ask-user.js +124 -13
  134. package/dist/core/ask-user.js.map +1 -1
  135. package/dist/core/auth-providers.js +1 -1
  136. package/dist/core/auth-providers.js.map +1 -1
  137. package/dist/core/autopilot-verdict.d.ts +5 -1
  138. package/dist/core/autopilot-verdict.js +30 -13
  139. package/dist/core/autopilot-verdict.js.map +1 -1
  140. package/dist/core/bundled-agents.js +1 -3
  141. package/dist/core/bundled-agents.js.map +1 -1
  142. package/dist/core/cache-diagnostics.d.ts +68 -0
  143. package/dist/core/cache-diagnostics.js +196 -0
  144. package/dist/core/cache-diagnostics.js.map +1 -0
  145. package/dist/core/cache-expiry.d.ts +87 -0
  146. package/dist/core/cache-expiry.js +111 -0
  147. package/dist/core/cache-expiry.js.map +1 -0
  148. package/dist/core/compaction/compactor.js +78 -48
  149. package/dist/core/compaction/compactor.js.map +1 -1
  150. package/dist/core/compaction/plan-step-policy.d.ts +46 -0
  151. package/dist/core/compaction/plan-step-policy.js +57 -0
  152. package/dist/core/compaction/plan-step-policy.js.map +1 -0
  153. package/dist/core/destructive-git-guard.d.ts +90 -0
  154. package/dist/core/destructive-git-guard.js +871 -0
  155. package/dist/core/destructive-git-guard.js.map +1 -0
  156. package/dist/core/event-bus.d.ts +3 -0
  157. package/dist/core/event-bus.js +5 -0
  158. package/dist/core/event-bus.js.map +1 -1
  159. package/dist/core/fast-apply-benchmark.d.ts +1 -1
  160. package/dist/core/fast-apply-benchmark.js +2 -2
  161. package/dist/core/fast-apply-benchmark.js.map +1 -1
  162. package/dist/core/injection-detect.d.ts +38 -0
  163. package/dist/core/injection-detect.js +232 -0
  164. package/dist/core/injection-detect.js.map +1 -0
  165. package/dist/core/keep-awake.d.ts +88 -0
  166. package/dist/core/keep-awake.js +251 -0
  167. package/dist/core/keep-awake.js.map +1 -0
  168. package/dist/core/mcp/client.d.ts +72 -0
  169. package/dist/core/mcp/client.js +264 -41
  170. package/dist/core/mcp/client.js.map +1 -1
  171. package/dist/core/mcp/content.js +6 -2
  172. package/dist/core/mcp/content.js.map +1 -1
  173. package/dist/core/mcp/store.d.ts +6 -1
  174. package/dist/core/mcp/store.js +12 -1
  175. package/dist/core/mcp/store.js.map +1 -1
  176. package/dist/core/mcp/types.d.ts +18 -0
  177. package/dist/core/model-unavailable.d.ts +14 -0
  178. package/dist/core/model-unavailable.js +23 -0
  179. package/dist/core/model-unavailable.js.map +1 -0
  180. package/dist/core/node-debugger.d.ts +148 -0
  181. package/dist/core/node-debugger.js +642 -0
  182. package/dist/core/node-debugger.js.map +1 -0
  183. package/dist/core/nolan-context.d.ts +7 -5
  184. package/dist/core/nolan-context.js +106 -16
  185. package/dist/core/nolan-context.js.map +1 -1
  186. package/dist/core/nolan-prompt.js +24 -21
  187. package/dist/core/nolan-prompt.js.map +1 -1
  188. package/dist/core/package-threats.d.ts +18 -0
  189. package/dist/core/package-threats.js +168 -0
  190. package/dist/core/package-threats.js.map +1 -0
  191. package/dist/core/persistent-shell.d.ts +58 -6
  192. package/dist/core/persistent-shell.js +331 -49
  193. package/dist/core/persistent-shell.js.map +1 -1
  194. package/dist/core/process-manager.d.ts +14 -0
  195. package/dist/core/process-manager.js +61 -0
  196. package/dist/core/process-manager.js.map +1 -1
  197. package/dist/core/progress/git-xp.js +8 -14
  198. package/dist/core/progress/git-xp.js.map +1 -1
  199. package/dist/core/project-discovery.js +77 -26
  200. package/dist/core/project-discovery.js.map +1 -1
  201. package/dist/core/semantic-search-benchmark.d.ts +1 -1
  202. package/dist/core/semantic-search-benchmark.js +2 -2
  203. package/dist/core/semantic-search-benchmark.js.map +1 -1
  204. package/dist/core/session-history.d.ts +12 -0
  205. package/dist/core/session-history.js +27 -0
  206. package/dist/core/session-history.js.map +1 -1
  207. package/dist/core/session-manager.d.ts +13 -1
  208. package/dist/core/session-manager.js +38 -18
  209. package/dist/core/session-manager.js.map +1 -1
  210. package/dist/core/session-summary-index.d.ts +37 -0
  211. package/dist/core/session-summary-index.js +172 -0
  212. package/dist/core/session-summary-index.js.map +1 -0
  213. package/dist/core/settings-manager.d.ts +2 -0
  214. package/dist/core/settings-manager.js +10 -0
  215. package/dist/core/settings-manager.js.map +1 -1
  216. package/dist/core/shell-threats-popular-packages.d.ts +11 -0
  217. package/dist/core/shell-threats-popular-packages.js +675 -0
  218. package/dist/core/shell-threats-popular-packages.js.map +1 -0
  219. package/dist/core/shell-threats.d.ts +8 -0
  220. package/dist/core/shell-threats.js +186 -0
  221. package/dist/core/shell-threats.js.map +1 -0
  222. package/dist/core/skills.js +16 -4
  223. package/dist/core/skills.js.map +1 -1
  224. package/dist/core/stream-rules.d.ts +30 -0
  225. package/dist/core/stream-rules.js +151 -0
  226. package/dist/core/stream-rules.js.map +1 -0
  227. package/dist/core/subagent-manager.d.ts +23 -6
  228. package/dist/core/subagent-manager.js +25 -9
  229. package/dist/core/subagent-manager.js.map +1 -1
  230. package/dist/core/subagent-policy.js +1 -1
  231. package/dist/core/subagent-policy.js.map +1 -1
  232. package/dist/core/subagent-receipt.d.ts +54 -0
  233. package/dist/core/subagent-receipt.js +276 -0
  234. package/dist/core/subagent-receipt.js.map +1 -0
  235. package/dist/core/subagent-turn-record.d.ts +2 -0
  236. package/dist/core/subagent-turn-record.js.map +1 -1
  237. package/dist/core/test-impact.d.ts +73 -0
  238. package/dist/core/test-impact.js +467 -0
  239. package/dist/core/test-impact.js.map +1 -0
  240. package/dist/core/thinking-level.d.ts +1 -1
  241. package/dist/core/thinking-level.js +1 -1
  242. package/dist/core/thinking-level.js.map +1 -1
  243. package/dist/core/verification-gate.d.ts +2 -0
  244. package/dist/core/verification-gate.js +4 -0
  245. package/dist/core/verification-gate.js.map +1 -1
  246. package/dist/core/verification-snapshot.js +3 -5
  247. package/dist/core/verification-snapshot.js.map +1 -1
  248. package/dist/core/workspace-guard.d.ts +18 -7
  249. package/dist/core/workspace-guard.js +227 -60
  250. package/dist/core/workspace-guard.js.map +1 -1
  251. package/dist/core/worktree-setup.d.ts +23 -0
  252. package/dist/core/worktree-setup.js +128 -9
  253. package/dist/core/worktree-setup.js.map +1 -1
  254. package/dist/core/worktree.d.ts +20 -2
  255. package/dist/core/worktree.js +74 -31
  256. package/dist/core/worktree.js.map +1 -1
  257. package/dist/interactive.js +2 -1
  258. package/dist/interactive.js.map +1 -1
  259. package/dist/modes/json-mode.js +11 -2
  260. package/dist/modes/json-mode.js.map +1 -1
  261. package/dist/modes/subagent-worker-mode.d.ts +36 -1
  262. package/dist/modes/subagent-worker-mode.js +100 -42
  263. package/dist/modes/subagent-worker-mode.js.map +1 -1
  264. package/dist/motion-agent/motion-agent.d.ts +7 -3
  265. package/dist/motion-agent/motion-agent.js +6 -8
  266. package/dist/motion-agent/motion-agent.js.map +1 -1
  267. package/dist/motion-agent/motion-prompt.d.ts +1 -1
  268. package/dist/motion-agent/motion-prompt.js +16 -25
  269. package/dist/motion-agent/motion-prompt.js.map +1 -1
  270. package/dist/motion-agent/motion-review-session.js +1 -1
  271. package/dist/motion-agent/motion-review-session.js.map +1 -1
  272. package/dist/motion-agent/motion-review.d.ts +10 -3
  273. package/dist/motion-agent/motion-review.js +15 -7
  274. package/dist/motion-agent/motion-review.js.map +1 -1
  275. package/dist/motion-agent/motion-studio-context.js +1 -1
  276. package/dist/motion-agent/motion-studio-context.js.map +1 -1
  277. package/dist/system-prompt.d.ts +2 -1
  278. package/dist/system-prompt.js +15 -4
  279. package/dist/system-prompt.js.map +1 -1
  280. package/dist/test-support/keep-alive.d.ts +14 -0
  281. package/dist/test-support/keep-alive.js +17 -0
  282. package/dist/test-support/keep-alive.js.map +1 -0
  283. package/dist/tools/ask-user.js +3 -3
  284. package/dist/tools/ask-user.js.map +1 -1
  285. package/dist/tools/bash-read-evidence.d.ts +10 -0
  286. package/dist/tools/bash-read-evidence.js +133 -0
  287. package/dist/tools/bash-read-evidence.js.map +1 -0
  288. package/dist/tools/bash.d.ts +10 -1
  289. package/dist/tools/bash.js +179 -7
  290. package/dist/tools/bash.js.map +1 -1
  291. package/dist/tools/debug.d.ts +54 -0
  292. package/dist/tools/debug.js +233 -0
  293. package/dist/tools/debug.js.map +1 -0
  294. package/dist/tools/edit.js +13 -5
  295. package/dist/tools/edit.js.map +1 -1
  296. package/dist/tools/goals.d.ts +1 -1
  297. package/dist/tools/index.d.ts +25 -2
  298. package/dist/tools/index.js +48 -7
  299. package/dist/tools/index.js.map +1 -1
  300. package/dist/tools/prompt-hints.js +2 -0
  301. package/dist/tools/prompt-hints.js.map +1 -1
  302. package/dist/tools/read-tracker.d.ts +35 -2
  303. package/dist/tools/read-tracker.js +108 -11
  304. package/dist/tools/read-tracker.js.map +1 -1
  305. package/dist/tools/read.js +39 -7
  306. package/dist/tools/read.js.map +1 -1
  307. package/dist/tools/skill.js +5 -0
  308. package/dist/tools/skill.js.map +1 -1
  309. package/dist/tools/subagent-control.js +44 -8
  310. package/dist/tools/subagent-control.js.map +1 -1
  311. package/dist/tools/subagent-shared.d.ts +48 -8
  312. package/dist/tools/subagent-shared.js +75 -14
  313. package/dist/tools/subagent-shared.js.map +1 -1
  314. package/dist/tools/subagent.d.ts +8 -2
  315. package/dist/tools/subagent.js +28 -10
  316. package/dist/tools/subagent.js.map +1 -1
  317. package/dist/tools/task-output.js +3 -2
  318. package/dist/tools/task-output.js.map +1 -1
  319. package/dist/tools/task-send.d.ts +1 -1
  320. package/dist/tools/task-send.js +15 -1
  321. package/dist/tools/task-send.js.map +1 -1
  322. package/dist/tools/tool-tiers.d.ts +2 -2
  323. package/dist/tools/tool-tiers.js +3 -2
  324. package/dist/tools/tool-tiers.js.map +1 -1
  325. package/dist/tools/truncate.d.ts +21 -0
  326. package/dist/tools/truncate.js +187 -0
  327. package/dist/tools/truncate.js.map +1 -1
  328. package/dist/tools/ui-adopt.js +2 -0
  329. package/dist/tools/ui-adopt.js.map +1 -1
  330. package/dist/tools/write.js +4 -3
  331. package/dist/tools/write.js.map +1 -1
  332. package/dist/ui/App.d.ts +0 -4
  333. package/dist/ui/App.js +5 -28
  334. package/dist/ui/App.js.map +1 -1
  335. package/dist/ui/components/ActivityIndicator.js +1 -0
  336. package/dist/ui/components/ActivityIndicator.js.map +1 -1
  337. package/dist/ui/components/Footer.js +1 -1
  338. package/dist/ui/components/Footer.js.map +1 -1
  339. package/dist/ui/hooks/useAgentLoop.d.ts +1 -8
  340. package/dist/ui/hooks/useAgentLoop.js +1 -119
  341. package/dist/ui/hooks/useAgentLoop.js.map +1 -1
  342. package/dist/ui/render.d.ts +2 -4
  343. package/dist/ui/render.js +3 -2
  344. package/dist/ui/render.js.map +1 -1
  345. package/dist/utils/git.d.ts +77 -0
  346. package/dist/utils/git.js +285 -21
  347. package/dist/utils/git.js.map +1 -1
  348. package/dist/utils/github-ci.js +2 -1
  349. package/dist/utils/github-ci.js.map +1 -1
  350. package/dist/utils/github.js +11 -9
  351. package/dist/utils/github.js.map +1 -1
  352. package/dist/utils/image.d.ts +14 -0
  353. package/dist/utils/image.js +16 -0
  354. package/dist/utils/image.js.map +1 -1
  355. package/dist/utils/process.d.ts +20 -0
  356. package/dist/utils/process.js +98 -0
  357. package/dist/utils/process.js.map +1 -1
  358. package/dist/utils/text.d.ts +12 -0
  359. package/dist/utils/text.js +11 -0
  360. package/dist/utils/text.js.map +1 -1
  361. package/package.json +6 -6
  362. package/assets/motion/skills/mixkit-split-text-617/SKILL.md +0 -115
  363. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-480.json +0 -1
  364. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-494.json +0 -1
  365. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-5.json +0 -1
  366. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-508.json +0 -1
  367. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6525.json +0 -1
  368. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6539.json +0 -1
  369. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6647.json +0 -1
  370. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6663.json +0 -1
  371. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6677.json +0 -1
  372. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6691.json +0 -1
  373. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6706.json +0 -1
  374. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6722.json +0 -1
  375. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6736.json +0 -1
  376. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6750.json +0 -1
  377. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6764.json +0 -1
  378. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6780.json +0 -1
  379. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6794.json +0 -1
  380. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6810.json +0 -1
  381. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6823.json +0 -1
  382. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6872.json +0 -1
  383. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-6936.json +0 -1
  384. package/assets/motion/skills/mixkit-split-text-617/data/compositions/comp-76.json +0 -1
  385. package/assets/motion/skills/mixkit-split-text-617/data/manifest.json +0 -495
  386. package/assets/motion/skills/mixkit-split-text-617/references/RECONSTRUCTION.md +0 -201
  387. package/assets/motion/skills/mixkit-split-text-617/references/VERIFICATION.md +0 -107
  388. package/assets/motion/skills/mixkit-split-text-617/tools/inspect_motion.py +0 -130
  389. package/assets/motion/skills/video-qa/SKILL.md +0 -66
  390. package/dist/core/ideal-review-subagent.d.ts +0 -42
  391. package/dist/core/ideal-review-subagent.js +0 -95
  392. package/dist/core/ideal-review-subagent.js.map +0 -1
  393. package/dist/core/ideal-review.d.ts +0 -82
  394. package/dist/core/ideal-review.js +0 -242
  395. package/dist/core/ideal-review.js.map +0 -1
  396. package/dist/motion-agent/motion-check-tool.d.ts +0 -21
  397. package/dist/motion-agent/motion-check-tool.js +0 -206
  398. package/dist/motion-agent/motion-check-tool.js.map +0 -1
@@ -1,6 +1,7 @@
1
- import { agentLoop, isAbortError, isUsageLimitError, } from "@prestyj/agent";
1
+ import { agentLoop, isAbortError, isUsageLimitError, repairToolPairingAdjacent, } from "@prestyj/agent";
2
2
  import { ProviderError, stream, } from "@prestyj/ai";
3
3
  import { EventBus } from "./event-bus.js";
4
+ import { flagUntrustedToolResult } from "./injection-detect.js";
4
5
  import { COMPLETION_REVIEW_STATE_KIND, } from "./completion-review.js";
5
6
  import { SlashCommandRegistry, createBuiltinCommands, } from "./slash-commands.js";
6
7
  import { PROMPT_COMMANDS, getPromptCommand } from "./prompt-commands.js";
@@ -23,8 +24,10 @@ import { ensureAppDirs } from "../config.js";
23
24
  import { buildSubAgentSystemPrompt, buildSystemPrompt, } from "../system-prompt.js";
24
25
  import { createTools, createWebSearchTool, } from "../tools/index.js";
25
26
  import { partitionToolsByTier } from "../tools/tool-tiers.js";
27
+ import { formatImpactForVerification } from "./test-impact.js";
28
+ import { autoBackgroundedId } from "../tools/bash.js";
26
29
  import { buildProcessCompletionFollowUp } from "./process-gate.js";
27
- import { buildSubAgentCompletionFollowUp, } from "./subagent-manager.js";
30
+ import { buildSubAgentCompletionFollowUp } from "./subagent-manager.js";
28
31
  import { applyAsyncSubagentPolicy } from "./subagent-policy.js";
29
32
  import { z } from "zod";
30
33
  import { MCPClientManager, getAllMcpServers } from "./mcp/index.js";
@@ -36,27 +39,31 @@ import { createToolSearchTool } from "../tools/tool-search.js";
36
39
  import { createSessionStatsTool } from "../tools/session-stats.js";
37
40
  import { createDiagnoseCommand, isInternalDiagnosticsEnabled, SessionDiagnosticsRecorder, } from "./internal-diagnostics.js";
38
41
  import { log } from "./logger.js";
39
- import { setEstimatorModel, calibrateEstimatorFromUsage } from "./compaction/token-estimator.js";
42
+ import { CacheDiagnostics } from "./cache-diagnostics.js";
43
+ import { assessCacheExpiry, resolveCacheTtl } from "./cache-expiry.js";
44
+ import { setEstimatorModel, calibrateEstimatorFromUsage, estimateConversationTokens, } from "./compaction/token-estimator.js";
40
45
  import { calculateActiveContextTokens } from "./compaction/active-context.js";
41
46
  import { resolveCompactionPolicy } from "./compaction/policy.js";
47
+ import { decidePlanStepCompaction, DEFAULT_CACHE_WRITE_READ_RATIO, PLAN_STEP_KEEP_TOKENS, } from "./compaction/plan-step-policy.js";
48
+ import { extractPlanSteps, findCompletedMarkers } from "../utils/plan-steps.js";
49
+ import { readFileSync } from "node:fs";
42
50
  import { clampThinkingForPlanMode } from "./thinking-level.js";
43
51
  import { pruneStaleToolResults } from "./compaction/tool-result-pruner.js";
44
52
  import { discoverAgents } from "./agents.js";
45
53
  import { enhancePrompt } from "../utils/prompt-enhancer.js";
46
54
  import { detectLanguages, detectProjectStack } from "./language-detector.js";
47
- import { evaluateIdealReview, buildIdealReviewMessage, buildReviewCoverageEscalationMessage, buildReviewCoverageMessage, MAX_REVIEW_COVERAGE_INJECTIONS, withReviewCoverageRequirements, detectTestDrift, ReviewCoverageTracker, } from "./ideal-review.js";
48
55
  import { evaluateLoopBreak, buildLoopBreakMessage, CycleDetector, ToolCallProgressTracker, detectTextRepetition, } from "./loop-breaker.js";
49
56
  import { buildRegroundingMessage, requestTextForRegrounding } from "./regrounding.js";
50
57
  import { buildSemanticLoopJudgePrompt, buildSemanticLoopMessage, MAX_SEMANTIC_LOOP_CALLS, parseSemanticLoopVerdict, shouldRunSemanticLoopCheck, SEMANTIC_LOOP_JUDGE_TIMEOUT_MS, withJudgeTimeout, } from "./semantic-loop-check.js";
51
- import { buildIndependentReviewMessage, buildReviewerTask, INDEPENDENT_REVIEW_SCORE_THRESHOLD, parseReviewerFindings, REVIEWER_TOOLS, REVIEWER_WAIT_MS, } from "./ideal-review-subagent.js";
52
58
  import { buildEnvDeltaMessage } from "./env-delta.js";
53
59
  import { wrapSteeringText, buildNotificationSteeringText, STEERING_PREFIX } from "./steering.js";
54
60
  import { AgentNotificationQueue } from "./agent-notifications.js";
55
- import { VerificationGate, extractAddedLines, isCheckOwnFile, isCodeFilePath, VERIFICATION_STATE_KIND, isVerificationCommand, } from "./verification-gate.js";
61
+ import { VerificationGate, isCheckOwnFile, extractAddedLines, isCodeFilePath, VERIFICATION_STATE_KIND, isVerificationCommand, } from "./verification-gate.js";
56
62
  import { classifyVerificationCommand } from "./verification-evidence.js";
57
63
  import { captureVerificationSnapshot } from "./verification-snapshot.js";
58
64
  import { findUserSessionPrompt, getUserSessionPrompt } from "./session-preview.js";
59
65
  import { normalizeMessageImages } from "./message-images.js";
66
+ import { loadStreamRules } from "./stream-rules.js";
60
67
  import crypto from "node:crypto";
61
68
  import fs from "node:fs/promises";
62
69
  import os from "node:os";
@@ -66,14 +73,6 @@ import path from "node:path";
66
73
  * progressing — refuse to extend its turn budget.
67
74
  */
68
75
  const TURN_EXTENSION_MAX_FAILURE_RATIO = 0.5;
69
- /** Terminal subagent states — mirrors SubAgentManager's private isTerminal. */
70
- function isTerminalSubAgentState(state) {
71
- return (state === "completed" ||
72
- state === "failed" ||
73
- state === "interrupted" ||
74
- state === "closed" ||
75
- state === "reaped");
76
- }
77
76
  // ── Tool-result policy ─────────────────────────────────────
78
77
  /** Resolve the per-result cap passed to the agent loop for the active transport. */
79
78
  export function resolveSessionToolResultCharLimit(model, provider, accountId) {
@@ -130,6 +129,7 @@ export class AgentSession {
130
129
  // transcript rows the live run showed.
131
130
  appMarkers = [];
132
131
  turnMetrics = [];
132
+ cacheDiagnostics = new CacheDiagnostics();
133
133
  /** Internal-only (EZ_INTERNAL): live per-session cost/reliability recorder.
134
134
  * Absent entirely in public builds — see core/internal-diagnostics.ts. */
135
135
  diagnosticsRecorder;
@@ -138,20 +138,16 @@ export class AgentSession {
138
138
  * creation). Called from switchModel so video-capable models get the
139
139
  * read-tool's native-video path after a mid-session model change. */
140
140
  rebuildReadTool;
141
+ /** Forgets every file read; called whenever the conversation is replaced or
142
+ * rewound, so the model must re-read a file before changing it. */
143
+ clearReadTracker;
144
+ recordBashReads;
141
145
  skills = [];
142
146
  cacheKeyLogged = false;
143
147
  // ── Self-correction hook state (mirrors the TUI's useAgentLoop refs) ──
144
148
  // Reset at the start of every run; observed from the event stream; read by
145
- // the loop-break (mid-loop) and ideal-review (pre-stop) callbacks.
146
- hookStats = {
147
- changedLines: 0,
148
- toolCalls: 0,
149
- toolFailures: 0,
150
- turns: 0,
151
- writeCalls: 0,
152
- editCalls: 0,
153
- bashCalls: 0,
154
- };
149
+ // the loop-break (mid-loop) callback.
150
+ hookStats = { toolCalls: 0, toolFailures: 0, turns: 0 };
155
151
  hookText = "";
156
152
  hookConsecutiveFailures = 0;
157
153
  hookRepeatedNoProgressCalls = 0;
@@ -161,20 +157,6 @@ export class AgentSession {
161
157
  hookFileEditCounts = new Map();
162
158
  hookToolCalls = new Map();
163
159
  backgroundVerification = new Map();
164
- idealReviewPhase = "idle";
165
- /** Runtime-only suppression while Nolan owns verification in autopilot mode. */
166
- idealReviewSuppressed = false;
167
- /** Mirror of the last `hook_armed` value broadcast this run, so the event
168
- * fires only on a real edge. */
169
- idealReviewArmed = false;
170
- /** Cached test-drift probe, keyed by the size of the edited-file set. Drift
171
- * depends only on WHICH files were edited and that set only grows, so this
172
- * keeps the arming check off the filesystem on most tool results — the probe
173
- * is several sync existsSync calls per edited file. */
174
- idealDriftProbe = null;
175
- reviewCoverage;
176
- /** Coverage follow-ups spent this run, capped by MAX_REVIEW_COVERAGE_INJECTIONS. */
177
- reviewCoverageInjected = 0;
178
160
  /** 0 = none; 1 = first nudge sent; 2 = final stop-and-report injected. */
179
161
  loopBreakInjected = 0;
180
162
  regroundingInjected = false;
@@ -184,8 +166,6 @@ export class AgentSession {
184
166
  * injection at the next steering poll; judge failures fail open (no
185
167
  * injection) and still consume budget + cooldown. */
186
168
  semanticLoop = { checksUsed: 0, lastCheckTurn: 0, pending: false, verdict: null, injected: false };
187
- /** Independent Ideal reviewer spawned once per run (score-gated). */
188
- independentReviewStarted = false;
189
169
  /**
190
170
  * The environment as the cached system prompt currently describes it.
191
171
  * Re-recorded on every prompt build, so a rebuild (e.g. `/add-dir`) needs no
@@ -196,11 +176,12 @@ export class AgentSession {
196
176
  runStartedAt = 0;
197
177
  /** Gate injections spent this run, capped by MAX_PROCESS_GATE_INJECTIONS. */
198
178
  processGateInjected = 0;
199
- /** Verification gate: code edited this run, nothing proved it since. */
179
+ /** Verification gate: code edited this run, nothing proved it since. Always
180
+ * tracks evidence for run status; the `verificationGateEnabled` setting
181
+ * decides whether an unverified stop is also continued once. */
200
182
  verificationGate = new VerificationGate();
201
- /** Mirror of the last verification `hook_armed` value, so the event fires
202
- * only on a real edge. */
203
- verificationArmed = false;
183
+ /** Mirror of the last `hook_armed` value, so the event fires only on an edge. */
184
+ preFinalArmed = false;
204
185
  compactionOccurred = false;
205
186
  /**
206
187
  * Re-grounding carry-over for post-turn compaction. `resetHookState` clears
@@ -213,6 +194,25 @@ export class AgentSession {
213
194
  postTurnCompaction;
214
195
  lastCompactionCompacted = false;
215
196
  compactionRetryAfter = 0;
197
+ /**
198
+ * SoL-Pi plan-step compaction bookkeeping (see compaction/plan-step-policy.ts).
199
+ * Step progress comes from `[DONE:n]` markers in assistant text — the same
200
+ * contract the approved-plan UI tracks.
201
+ */
202
+ planStepState = {
203
+ planPath: undefined,
204
+ scanIndex: 0,
205
+ doneSteps: new Set(),
206
+ requestsInCompletedSteps: 0,
207
+ requestsThisStep: 0,
208
+ requests: 0,
209
+ grownTokens: 0,
210
+ lastContextTokens: 0,
211
+ compactions: 0,
212
+ writeCostBalance: 0,
213
+ savingPerRequest: 0,
214
+ requestsSinceLastCompaction: 0,
215
+ };
216
216
  /** A restored oversized checkpoint must be canonicalized before its first prompt is persisted. */
217
217
  deferredCompactionPending = false;
218
218
  /** Latest provider count, anchored to the assistant response it measured. */
@@ -228,8 +228,12 @@ export class AgentSession {
228
228
  // different message (or past the end) by the time the cancel arrives.
229
229
  userQueue = [];
230
230
  queueSeq = 0;
231
+ /** Instant interrupt: the running loop's preempt listeners, fired on queueMessage. */
232
+ steeringListeners = new Set();
231
233
  processManager;
232
234
  lspManager;
235
+ testImpact;
236
+ debugManager;
233
237
  subAgentManager;
234
238
  /**
235
239
  * Out-of-band push notifications (finished children, background-process
@@ -333,7 +337,6 @@ export class AgentSession {
333
337
  this.provider = options.provider;
334
338
  this.model = options.model;
335
339
  this.cwd = options.cwd;
336
- this.reviewCoverage = new ReviewCoverageTracker(this.cwd);
337
340
  this.baseUrl = options.baseUrl;
338
341
  this.maxTokens = this.resolveMaxTokens(options.model);
339
342
  this.thinkingLevel = options.thinkingLevel;
@@ -399,7 +402,7 @@ export class AgentSession {
399
402
  : this.opts.globalSubagents
400
403
  ? await discoverAgents({ globalAgentsDir: paths.agentsDir })
401
404
  : [];
402
- const { tools: builtInTools, processManager, rebuildReadTool, lspManager, subAgentManager, } = await createTools(this.cwd, {
405
+ const { tools: builtInTools, processManager, rebuildReadTool, clearReadTracker, recordBashReads, lspManager, testImpact, debugManager, subAgentManager, } = await createTools(this.cwd, {
403
406
  agents,
404
407
  skills: this.skills,
405
408
  contextLimits: this.contextLimits,
@@ -430,17 +433,14 @@ export class AgentSession {
430
433
  }),
431
434
  getUseExternalGrep: () => this.settingsManager.get("grepUseRipgrep"),
432
435
  authStorage: this.authStorage,
433
- onFileRead: (filePath) => this.reviewCoverage.recordRead(filePath),
434
436
  onFileMutated: (filePath) => {
435
437
  const relative = path.relative(this.cwd, filePath) || path.basename(filePath);
436
438
  this.hookFileEditCounts.set(relative, (this.hookFileEditCounts.get(relative) ?? 0) + 1);
437
- this.reviewCoverage.recordChanged(filePath);
438
439
  },
439
440
  // Lazy — sessionId/model/provider can change after createTools() runs, so
440
441
  // sub-agent spawns read the current parent state at execution time.
441
442
  getProvider: () => this.provider,
442
443
  getModel: () => this.model,
443
- getThinkingLevel: () => this.thinkingLevel,
444
444
  getBaseUrl: () => this.baseUrl,
445
445
  getCacheKey: () => this.getPromptCacheKey(),
446
446
  getMaxPerModel: () => this.settingsManager.get("subagentMaxPerModel"),
@@ -486,11 +486,16 @@ export class AgentSession {
486
486
  this.mcpCatalog ??= new DeferredToolCatalog(this.contextLimits);
487
487
  this.mcpCatalog.add(deferred);
488
488
  this.ensureToolSearchTool();
489
+ this.promoteWaitAgentAfterSpawn();
489
490
  }
490
491
  }
491
492
  this.rebuildReadTool = rebuildReadTool;
493
+ this.clearReadTracker = clearReadTracker;
494
+ this.recordBashReads = recordBashReads;
492
495
  this.processManager = processManager;
493
496
  this.lspManager = lspManager;
497
+ this.testImpact = testImpact;
498
+ this.debugManager = debugManager;
494
499
  this.subAgentManager = subAgentManager;
495
500
  this.bindManagerCancellation(this.opts.signal);
496
501
  // Connect MCP servers. Child sessions skip user-configured servers to avoid
@@ -775,6 +780,29 @@ export class AgentSession {
775
780
  : { serverName, ok: false, error: outcome.error };
776
781
  }, this.contextLimits));
777
782
  }
783
+ /**
784
+ * `wait_agent` is deferred, yet nearly every `spawn_agent` is followed by it,
785
+ * so the model spent a whole turn on `tool_search` just to load it (bench 41:
786
+ * ~6 s per fan-out). Promote it as soon as a spawn succeeds instead: the tool
787
+ * list grows exactly as it would after that `tool_search`, one turn earlier,
788
+ * and sessions that never spawn keep the smaller prefix.
789
+ */
790
+ promoteWaitAgentAfterSpawn() {
791
+ const index = this.tools.findIndex((t) => t.name === "spawn_agent");
792
+ const spawn = index >= 0 ? this.tools[index] : undefined;
793
+ if (!spawn)
794
+ return;
795
+ this.tools[index] = {
796
+ ...spawn,
797
+ execute: async (args, context) => {
798
+ const result = await spawn.execute(args, context);
799
+ if (!this.tools.some((t) => t.name === "wait_agent")) {
800
+ this.tools.push(...(this.mcpCatalog?.promote(["wait_agent"]) ?? []));
801
+ }
802
+ return result;
803
+ },
804
+ };
805
+ }
778
806
  /** Append tools, replacing any same-named entry (cached stub → live tool). */
779
807
  replaceOrPushTools(tools) {
780
808
  for (const tool of tools) {
@@ -905,12 +933,15 @@ export class AgentSession {
905
933
  }
906
934
  /**
907
935
  * Process user input. Handles slash commands or runs agent loop.
936
+ * `capThinking` holds reasoning effort at the plan-mode ceiling for this
937
+ * prompt — for turns that must answer quickly from what is already known.
908
938
  */
909
939
  async prompt(content, provenance = {
910
940
  source: "human",
911
941
  kind: "prompt",
912
942
  visibility: "transcript",
913
943
  }, options = {}) {
944
+ this.prewarmController?.abort();
914
945
  await this.settlePostTurnCompaction();
915
946
  await this.adoptDeferredCheckpointBeforePrompt();
916
947
  const slash = await this.resolveSlashInput(content);
@@ -946,6 +977,7 @@ export class AgentSession {
946
977
  * attachments are always a direct conversational turn.
947
978
  */
948
979
  async promptWithAttachments(text, attachments) {
980
+ this.prewarmController?.abort();
949
981
  await this.settlePostTurnCompaction();
950
982
  await this.adoptDeferredCheckpointBeforePrompt();
951
983
  const parts = this.buildAttachmentParts(text, attachments);
@@ -1044,15 +1076,7 @@ export class AgentSession {
1044
1076
  resetHookState(originalRequest) {
1045
1077
  this.opts.completionReview?.begin(originalRequest);
1046
1078
  this.lspManager?.clearPendingDiagnostics();
1047
- this.hookStats = {
1048
- changedLines: 0,
1049
- toolCalls: 0,
1050
- toolFailures: 0,
1051
- turns: 0,
1052
- writeCalls: 0,
1053
- editCalls: 0,
1054
- bashCalls: 0,
1055
- };
1079
+ this.hookStats = { toolCalls: 0, toolFailures: 0, turns: 0 };
1056
1080
  this.hookText = "";
1057
1081
  this.hookConsecutiveFailures = 0;
1058
1082
  this.hookRepeatedNoProgressCalls = 0;
@@ -1061,13 +1085,6 @@ export class AgentSession {
1061
1085
  this.hookCyclicPattern = null;
1062
1086
  this.hookFileEditCounts.clear();
1063
1087
  this.hookToolCalls.clear();
1064
- this.reviewCoverage.reset();
1065
- this.reviewCoverageInjected = 0;
1066
- this.idealReviewPhase = "idle";
1067
- // No event here: clients reset their own hold on run_start.
1068
- this.idealReviewArmed = false;
1069
- this.verificationArmed = false;
1070
- this.idealDriftProbe = null;
1071
1088
  this.loopBreakInjected = 0;
1072
1089
  this.regroundingInjected = false;
1073
1090
  this.hookRecentCalls = [];
@@ -1078,7 +1095,6 @@ export class AgentSession {
1078
1095
  verdict: null,
1079
1096
  injected: false,
1080
1097
  };
1081
- this.independentReviewStarted = false;
1082
1098
  this.runStartedAt = Date.now();
1083
1099
  this.processGateInjected = 0;
1084
1100
  this.verificationGate.beginRun();
@@ -1099,8 +1115,8 @@ export class AgentSession {
1099
1115
  }
1100
1116
  /**
1101
1117
  * Fold one agent event into the hook stat accumulators. Pure bookkeeping —
1102
- * the same signals the TUI's useAgentLoop collects, so the loop-break and
1103
- * ideal-review decisions match across the CLI and the app.
1118
+ * the same signals the TUI's useAgentLoop collects, so loop-break decisions
1119
+ * match across the CLI and the app.
1104
1120
  */
1105
1121
  async trackHookEvent(event) {
1106
1122
  if (this.opts.completionReview) {
@@ -1145,7 +1161,14 @@ export class AgentSession {
1145
1161
  if (call.sourceSnapshot === null)
1146
1162
  this.verificationGate.requireFreshVerification(true, event.args.command);
1147
1163
  }
1148
- else {
1164
+ else if ((classification.accepted && event.args.persist !== true) ||
1165
+ (!classification.accepted && classification.mayMutate)) {
1166
+ // Flag the workspace unknown only when tool_call_end can resolve
1167
+ // it: a bounded check records pass/fail, a file-rewriting command
1168
+ // bumps the revision. An unrecognized read-only check (`biome ci`)
1169
+ // or a persistent-shell run records nothing at the end, so
1170
+ // flagging it left verified work Unverified forever — and
1171
+ // autopilot silently refused every later turn.
1149
1172
  this.verificationGate.requireFreshVerification(!classification.accepted && classification.mayMutate, event.args.command);
1150
1173
  }
1151
1174
  await this.persistVerificationState();
@@ -1159,12 +1182,6 @@ export class AgentSession {
1159
1182
  this.hookStats.toolCalls += 1;
1160
1183
  if (event.isError)
1161
1184
  this.hookStats.toolFailures += 1;
1162
- if (name === "write")
1163
- this.hookStats.writeCalls += 1;
1164
- if (name === "edit")
1165
- this.hookStats.editCalls += 1;
1166
- if (name === "bash")
1167
- this.hookStats.bashCalls += 1;
1168
1185
  this.hookConsecutiveFailures = event.isError ? this.hookConsecutiveFailures + 1 : 0;
1169
1186
  this.hookRepeatedNoProgressCalls = this.hookProgressTracker.record(name, args, event.result, event.isError);
1170
1187
  this.hookCyclicPattern = this.hookCycleDetector.record(name, args, event.result, event.isError);
@@ -1181,12 +1198,6 @@ export class AgentSession {
1181
1198
  if (this.hookRecentCalls.length > MAX_SEMANTIC_LOOP_CALLS) {
1182
1199
  this.hookRecentCalls.splice(0, this.hookRecentCalls.length - MAX_SEMANTIC_LOOP_CALLS);
1183
1200
  }
1184
- if (name === "edit" && !event.isError) {
1185
- const diff = event.details?.diff ?? event.result;
1186
- const added = (diff.match(/^\+[^+]/gm) ?? []).length;
1187
- const removed = (diff.match(/^-[^-]/gm) ?? []).length;
1188
- this.hookStats.changedLines += added + removed;
1189
- }
1190
1201
  // Only host-observed successful mutations and trustworthy check results
1191
1202
  // affect approval. The model's text is never evidence.
1192
1203
  let verificationChanged = false;
@@ -1209,8 +1220,13 @@ export class AgentSession {
1209
1220
  const command = typeof args.command === "string" ? args.command : "";
1210
1221
  const classification = classifyVerificationCommand(command);
1211
1222
  if (classification.accepted || classification.snapshotEligible) {
1212
- if (args.run_in_background === true && !event.isError && args.persist !== true) {
1213
- const id = /^ID:\s*(\S+)/m.exec(event.result)?.[1];
1223
+ // A foreground check that outlived the default budget was moved to
1224
+ // the background, not failed: track it to its real exit the same way.
1225
+ const autoBackgroundId = event.isError ? undefined : autoBackgroundedId(event.result);
1226
+ if ((autoBackgroundId !== undefined || args.run_in_background === true) &&
1227
+ !event.isError &&
1228
+ args.persist !== true) {
1229
+ const id = autoBackgroundId ?? /^ID:\s*(\S+)/m.exec(event.result)?.[1];
1214
1230
  // No parseable ID means the check cannot be tracked to a real exit
1215
1231
  // code — no evidence either way. Recording a FAILURE here made
1216
1232
  // every later green run of a different spelling look owed.
@@ -1281,10 +1297,35 @@ export class AgentSession {
1281
1297
  // Tool results for this step are in the array and their side effects
1282
1298
  // already hit the filesystem. Flushing here is what makes a crash lose
1283
1299
  // at most the in-flight step instead of the entire turn.
1300
+ await this.creditBashReads();
1284
1301
  await this.flushPendingMessages();
1285
1302
  break;
1286
1303
  }
1287
1304
  }
1305
+ /**
1306
+ * Count full-file `cat` output from the step that just finished as reads, so
1307
+ * an edit after `cat` does not cost a second read. Uses the step's results as
1308
+ * stored in the transcript (after per-turn trimming): only bytes the model
1309
+ * actually received count.
1310
+ */
1311
+ async creditBashReads() {
1312
+ if (!this.recordBashReads)
1313
+ return;
1314
+ const messages = this.activeLoopMessages ?? this.messages;
1315
+ const toolMessage = messages.at(-1);
1316
+ const assistant = messages.at(-2);
1317
+ if (toolMessage?.role !== "tool" || assistant?.role !== "assistant")
1318
+ return;
1319
+ if (typeof assistant.content === "string")
1320
+ return;
1321
+ const calls = assistant.content.filter((part) => part.type === "tool_call");
1322
+ try {
1323
+ await this.recordBashReads(calls, toolMessage.content);
1324
+ }
1325
+ catch (err) {
1326
+ log("WARN", "agent-session", "Crediting bash reads failed", { error: String(err) });
1327
+ }
1328
+ }
1288
1329
  /**
1289
1330
  * Append every message added since the last flush to the session file.
1290
1331
  *
@@ -1347,7 +1388,7 @@ export class AgentSession {
1347
1388
  const diagnosticText = this.lspManager?.drainDiagnostics(this.getVerificationProblem() !== null, { deferUnverified: true });
1348
1389
  if (diagnosticText)
1349
1390
  this.eventBus.emit("diagnostics", { text: diagnosticText });
1350
- this.refreshVerificationArmed();
1391
+ this.refreshHookArming();
1351
1392
  const notified = this.notifications.drain();
1352
1393
  const notificationMessage = notified.length > 0 || diagnosticText
1353
1394
  ? {
@@ -1410,6 +1451,7 @@ export class AgentSession {
1410
1451
  }
1411
1452
  if (this.opts.selfCorrectionHooks === false)
1412
1453
  return null;
1454
+ // Legacy key: the user-facing switch for loop-break and re-grounding nudges.
1413
1455
  if (!this.settingsManager.get("idealReviewEnabled"))
1414
1456
  return null;
1415
1457
  // Deterministic stuck verdict, computed once and shared: the semantic
@@ -1609,76 +1651,6 @@ export class AgentSession {
1609
1651
  .join("\n")
1610
1652
  : "";
1611
1653
  }
1612
- /** Independent fresh-context review of the finished work (Codex Guardian
1613
- * pattern). Spawns a READ-ONLY child on the ACTIVE model, waits bounded,
1614
- * and returns findings for the acting agent to address — or nothing when
1615
- * the review passes, is unavailable, or fails (in-thread review remains the
1616
- * fallback; the feature degrades, never blocks).
1617
- *
1618
- * Runs inside the pre-stop poll, so the candidate final answer is already
1619
- * held by arming and this wait cannot race a streamed answer. */
1620
- async runIndependentReview(decision) {
1621
- if (!this.subAgentManager)
1622
- return [];
1623
- if (this.independentReviewStarted)
1624
- return [];
1625
- // An allow-listed session (a subagent worker itself) must not spawn
1626
- // harness-owned grandchildren the tool policy never granted.
1627
- if (this.opts.allowedTools && !this.opts.allowedTools.includes("spawn_agent"))
1628
- return [];
1629
- if (decision.score < INDEPENDENT_REVIEW_SCORE_THRESHOLD)
1630
- return [];
1631
- this.independentReviewStarted = true;
1632
- const taskName = `ideal-reviewer-${Math.random().toString(36).slice(2, 8)}`;
1633
- let agentId;
1634
- try {
1635
- const task = buildReviewerTask({
1636
- originalRequest: this.originalRequest,
1637
- changedFiles: [...this.hookFileEditCounts.keys()],
1638
- stats: this.hookStats,
1639
- triggerReasons: decision.reasons,
1640
- });
1641
- // Active model forced at spawn time — never routed to a fast/review model.
1642
- const snapshot = await this.subAgentManager.spawn(taskName, task, undefined, {
1643
- model: this.model,
1644
- tools: REVIEWER_TOOLS,
1645
- });
1646
- agentId = snapshot.agent_id;
1647
- const waited = await this.subAgentManager.wait([agentId], "all", REVIEWER_WAIT_MS);
1648
- const agent = waited.agents[0];
1649
- if (!agent || !isTerminalSubAgentState(agent.state)) {
1650
- // Timeout: collect the straggler so the completion gate cannot fire on
1651
- // it later, then fall back to the in-thread review.
1652
- await this.subAgentManager.interrupt(agentId, true).catch(() => { });
1653
- log("WARN", "ideal", "Independent reviewer timed out; falling back to in-thread review", {
1654
- agentId,
1655
- });
1656
- return [];
1657
- }
1658
- const findings = parseReviewerFindings(agent.output ?? "");
1659
- if (!findings) {
1660
- log("WARN", "ideal", "Independent reviewer output unparseable; falling back", { agentId });
1661
- return [];
1662
- }
1663
- if (findings.clean) {
1664
- log("INFO", "ideal", "Independent reviewer verdict: clean", { agentId });
1665
- return [];
1666
- }
1667
- log("INFO", "ideal", "Independent reviewer flagged findings", {
1668
- agentId,
1669
- count: String(findings.findings.length),
1670
- });
1671
- return [buildIndependentReviewMessage(findings.findings)];
1672
- }
1673
- catch (error) {
1674
- if (agentId)
1675
- await this.subAgentManager.interrupt(agentId, true).catch(() => { });
1676
- log("WARN", "ideal", "Independent reviewer failed; falling back to in-thread review", {
1677
- error: error instanceof Error ? error.message : String(error),
1678
- });
1679
- return [];
1680
- }
1681
- }
1682
1654
  /**
1683
1655
  * Turn-budget extension gate. The loop consults this instead of stopping
1684
1656
  * mid-task when it exhausts `maxTurns`. Grant ONLY on evidence of progress —
@@ -1712,98 +1684,45 @@ export class AgentSession {
1712
1684
  });
1713
1685
  return granted;
1714
1686
  }
1715
- /**
1716
- * Would the stop AFTER the current turn inject the Ideal review? Same inputs
1717
- * as the pre-stop gate below, evaluated early so clients know a candidate
1718
- * final answer is a review draft BEFORE it streams.
1719
- *
1720
- * The turn count is looked ahead by one on purpose. `hookStats.turns` only
1721
- * advances at `turn_end`, so while the model is writing the draft the counter
1722
- * still reads the PREVIOUS turn; the real gate sees one more. Without the
1723
- * lookahead a run sitting on score 3 crosses to 4 on the draft's own
1724
- * `turn_end` — after the text already streamed — which is precisely the
1725
- * appear-then-vanish flash. Over-arming by one turn point costs only live
1726
- * token streaming on a final answer that then shows whole; under-arming costs
1727
- * the flash, so this errs toward arming.
1728
- */
1729
- wouldInjectIdealReview() {
1730
- if (this.opts.completionReview?.armed)
1731
- return true;
1732
- if (this.opts.selfCorrectionHooks === false || this.idealReviewSuppressed)
1733
- return false;
1734
- // Mid-review a stop still injects: the coverage follow-up while files are
1735
- // unread, or its escalation once the budget is spent. Both make the model
1736
- // answer again, so the candidate answer is a draft exactly as it is before
1737
- // the review starts — without arming here it paints and the reviewed answer
1738
- // lands under it as a duplicate.
1739
- if (this.idealReviewPhase === "reviewing") {
1740
- return this.reviewCoverage.evidence().missing.length > 0;
1741
- }
1742
- if (this.idealReviewPhase !== "idle")
1743
- return false;
1744
- if (!this.settingsManager.get("idealReviewEnabled"))
1687
+ /** Is the pre-stop verification gate active for this session? Off by the
1688
+ * `verificationGateEnabled` setting, by `selfCorrectionHooks: false`, and for
1689
+ * allow-listed sessions that cannot run commands at all. */
1690
+ verificationGateActive() {
1691
+ if (this.opts.selfCorrectionHooks === false)
1745
1692
  return false;
1746
- if (evaluateIdealReview({ ...this.hookStats, turns: this.hookStats.turns + 1 }).shouldReview) {
1747
- return true;
1748
- }
1749
- const files = this.hookFileEditCounts.size;
1750
- if (files === 0)
1693
+ if (!this.settingsManager.get("verificationGateEnabled"))
1751
1694
  return false;
1752
- if (this.idealDriftProbe?.files !== files) {
1753
- this.idealDriftProbe = {
1754
- files,
1755
- drifted: detectTestDrift(this.hookFileEditCounts.keys(), this.cwd).length > 0,
1756
- };
1757
- }
1758
- return this.idealDriftProbe.drifted;
1695
+ return !this.opts.allowedTools || this.opts.allowedTools.includes("bash");
1759
1696
  }
1760
- /** Would a stop right now inject the verification gate? Same conditions as
1761
- * the pre-stop branch below, so arming and injection cannot disagree. */
1762
- wouldInjectVerification() {
1697
+ /** Would a stop right now inject a pre-final follow-up? Queued LSP
1698
+ * diagnostics (real errors injected below), a mode-owned completion review,
1699
+ * or the verification gate can, so clients hold the candidate answer only
1700
+ * then. Same conditions as the pre-stop branch, so arming and injection
1701
+ * cannot disagree. */
1702
+ wouldInjectBeforeFinal() {
1703
+ if (this.opts.completionReview?.armed)
1704
+ return true;
1763
1705
  if (this.lspManager?.hasQueuedDiagnostics())
1764
1706
  return true;
1765
- if (this.opts.selfCorrectionHooks === false)
1766
- return false;
1767
- if (!this.settingsManager.get("verificationGateEnabled"))
1768
- return false;
1769
- if (this.opts.allowedTools && !this.opts.allowedTools.includes("bash"))
1770
- return false;
1771
- return this.verificationGate.willInject();
1707
+ return this.verificationGateActive() && this.verificationGate.willInject();
1772
1708
  }
1773
1709
  /** Broadcast pre-final hook arming on change. Both edges matter: armed=false
1774
- * after the hook fires is what lets a client stream the REVIEWED final
1775
- * answer live again.
1776
- *
1777
- * Callable before `initialize()`: the sidecar sets Nolan's review suppression
1778
- * on a freshly constructed session, and every arming predicate below reads
1779
- * settings that `initialize()` has not loaded yet. Nothing can be armed
1780
- * before the session can run a turn, and the first `tool_result`/`turn_end`
1781
- * recomputes both edges — so skipping is the correct answer, not a patch. */
1710
+ * after the hook fires is what lets a client stream the final answer live
1711
+ * again. Callable before `initialize()`, when no manager exists yet. */
1782
1712
  refreshHookArming() {
1713
+ // Before `initialize()` settings are not loaded, so nothing can be armed.
1783
1714
  if (!this.settingsManager)
1784
1715
  return;
1785
- this.refreshIdealReviewArmed();
1786
- this.refreshVerificationArmed();
1787
- }
1788
- refreshVerificationArmed() {
1789
- if (!this.settingsManager)
1716
+ const armed = this.wouldInjectBeforeFinal();
1717
+ if (armed === this.preFinalArmed)
1790
1718
  return;
1791
- const armed = this.wouldInjectVerification();
1792
- if (armed === this.verificationArmed)
1793
- return;
1794
- this.verificationArmed = armed;
1719
+ this.preFinalArmed = armed;
1795
1720
  this.eventBus.emit("hook_armed", { kind: "verification", armed });
1796
1721
  }
1797
- refreshIdealReviewArmed() {
1798
- const armed = this.wouldInjectIdealReview();
1799
- if (armed === this.idealReviewArmed)
1800
- return;
1801
- this.idealReviewArmed = armed;
1802
- this.eventBus.emit("hook_armed", { kind: "ideal", armed });
1803
- }
1804
1722
  /**
1805
- * Pre-stop Ideal review phase machine. Once review starts, completion is
1806
- * blocked until harness-owned post-injection reads cover every changed file.
1723
+ * Pre-stop follow-ups: LSP errors, unread child agents and background
1724
+ * processes, the verification gate (when `verificationGateEnabled`), and a
1725
+ * mode-owned completion review.
1807
1726
  */
1808
1727
  async getHookFollowUpMessages() {
1809
1728
  // Exit notifications and task_output refer to the same host process record.
@@ -1816,14 +1735,15 @@ export class AgentSession {
1816
1735
  if (backgroundChanged)
1817
1736
  await this.persistVerificationState();
1818
1737
  // Edits return immediately; only the completion boundary waits for remaining
1819
- // checks. Queued timeouts stay explicitly unverified, never a false all-clear.
1738
+ // checks. Queued timeouts stay explicitly unverified, never a false
1739
+ // all-clear; they join the verification demand below when the gate is on.
1820
1740
  await this.lspManager?.flushDiagnostics(this.opts.signal);
1821
1741
  if (this.opts.signal?.aborted)
1822
1742
  return null;
1823
- const diagnosticText = this.lspManager?.drainDiagnostics(this.getVerificationProblem() !== null);
1743
+ const diagnosticText = this.lspManager?.drainDiagnostics(this.verificationGateActive() && this.getVerificationProblem() !== null);
1824
1744
  if (diagnosticText)
1825
1745
  this.eventBus.emit("diagnostics", { text: diagnosticText });
1826
- this.refreshVerificationArmed();
1746
+ this.refreshHookArming();
1827
1747
  const diagnosticMessages = diagnosticText
1828
1748
  ? [
1829
1749
  {
@@ -1848,18 +1768,29 @@ export class AgentSession {
1848
1768
  return [...diagnosticMessages, ...processFollowUp];
1849
1769
  }
1850
1770
  // Verification gate: code was edited but nothing verified since the last
1851
- // edit. Above the Ideal review so checks RUN before the read-based review
1852
- // starts; off for allow-listed sessions that cannot run commands at all.
1853
- if (this.opts.selfCorrectionHooks !== false &&
1854
- this.settingsManager.get("verificationGateEnabled") &&
1855
- (!this.opts.allowedTools || this.opts.allowedTools.includes("bash"))) {
1771
+ // edit. Off via the `verificationGateEnabled` setting.
1772
+ if (this.verificationGateActive()) {
1856
1773
  const verificationReason = this.verificationGate.pendingReason();
1774
+ const pendingFiles = this.verificationGate.pendingFiles();
1857
1775
  const verificationFollowUp = this.verificationGate.followUp();
1858
1776
  if (verificationFollowUp) {
1777
+ // Name the tests that actually reach the unverified files, with the
1778
+ // command that runs exactly those, so the check is targeted rather
1779
+ // than guessed. Best-effort: no index or no runner leaves it unchanged.
1780
+ if (this.testImpact &&
1781
+ (verificationReason === "initial" || verificationReason === "recheck")) {
1782
+ const impactLine = await this.testImpact
1783
+ .impactFor(pendingFiles)
1784
+ .then(formatImpactForVerification)
1785
+ .catch(() => "");
1786
+ const first = verificationFollowUp[0];
1787
+ if (impactLine && first?.role === "user" && typeof first.content === "string") {
1788
+ verificationFollowUp[0] = { ...first, content: first.content + impactLine };
1789
+ }
1790
+ }
1859
1791
  log("INFO", "verification-gate", "Injecting verification follow-up", {});
1860
1792
  // Announce, THEN disarm: clients release held text on disarm, so the
1861
- // reverse order paints the draft and immediately deletes it — the exact
1862
- // flash arming exists to prevent.
1793
+ // reverse order paints the draft and immediately deletes it.
1863
1794
  this.eventBus.emit("hook", {
1864
1795
  kind: "verification",
1865
1796
  ...(verificationReason === "tamper"
@@ -1872,7 +1803,7 @@ export class AgentSession {
1872
1803
  return [...diagnosticMessages, ...verificationFollowUp];
1873
1804
  }
1874
1805
  }
1875
- // Address real errors before review; unavailable checks share the existing
1806
+ // Address real errors before review; unavailable checks share the
1876
1807
  // verification demand above instead of manufacturing a separate hook.
1877
1808
  if (diagnosticMessages.length > 0)
1878
1809
  return diagnosticMessages;
@@ -1892,131 +1823,24 @@ export class AgentSession {
1892
1823
  }
1893
1824
  this.refreshHookArming();
1894
1825
  }
1895
- if (this.opts.selfCorrectionHooks === false || this.idealReviewSuppressed)
1896
- return null;
1897
- if (this.idealReviewPhase === "reviewing") {
1898
- const coverage = this.reviewCoverage.evidence();
1899
- const lspEvidence = this.reviewLspEvidence(coverage.expected);
1900
- log("INFO", "ideal", "Ideal review coverage check", {
1901
- covered: coverage.covered,
1902
- missing: coverage.missing,
1903
- lspLowConfidence: lspEvidence.lowConfidence,
1904
- lspMissing: lspEvidence.missing,
1905
- });
1906
- if (coverage.missing.length > 0) {
1907
- // Announce like any other pre-final injection: this follow-up makes the
1908
- // model answer again, so the answer it interrupts is a draft and the
1909
- // hook event is what tells clients to discard it. Injecting silently is
1910
- // what let the pre-coverage answer paint above the reviewed one.
1911
- this.eventBus.emit("hook", {
1912
- kind: "ideal",
1913
- coverageExpected: coverage.expected,
1914
- coverageMissing: coverage.missing,
1915
- });
1916
- if (this.reviewCoverageInjected < MAX_REVIEW_COVERAGE_INJECTIONS) {
1917
- this.reviewCoverageInjected += 1;
1918
- // Stays armed (coverage is still outstanding) — this call is here so a
1919
- // client that missed the earlier edge is armed before the next draft.
1920
- this.refreshIdealReviewArmed();
1921
- return [
1922
- this.withReviewLspEvidence(buildReviewCoverageMessage(coverage.missing), lspEvidence),
1923
- ];
1924
- }
1925
- // Budget spent: close the gate so the run cannot spin on a file that
1926
- // never becomes readable, and require the gap be reported to the user.
1927
- this.idealReviewPhase = "complete";
1928
- // The gate is shut, so this is the real disarm: the answer to the
1929
- // escalation is final and streams live.
1930
- this.refreshIdealReviewArmed();
1931
- log("INFO", "ideal", "Ideal review coverage escalated after retry budget", {
1932
- injected: String(this.reviewCoverageInjected),
1933
- missing: coverage.missing,
1934
- });
1935
- return [buildReviewCoverageEscalationMessage(coverage.missing)];
1936
- }
1937
- this.idealReviewPhase = "complete";
1938
- return null;
1826
+ return null;
1827
+ }
1828
+ /** Wraps the real run: cancels an in-flight cache prewarm and tracks run
1829
+ * activity / last real request time for {@link prewarm}. */
1830
+ async runLoop(options = {}) {
1831
+ this.prewarmController?.abort();
1832
+ this.runLoopDepth++;
1833
+ try {
1834
+ await this.runLoopInner(options);
1835
+ }
1836
+ finally {
1837
+ this.runLoopDepth--;
1838
+ this.lastRealRequestAt = Date.now();
1939
1839
  }
1940
- if (this.idealReviewPhase === "complete")
1941
- return null;
1942
- if (!this.settingsManager.get("idealReviewEnabled"))
1943
- return null;
1944
- const decision = evaluateIdealReview(this.hookStats);
1945
- // Test drift fires the review even on a small change the score would skip:
1946
- // a green-but-stale test is exactly what the volume gate sleeps through.
1947
- const driftedFiles = detectTestDrift(this.hookFileEditCounts.keys(), this.cwd).slice(0, 5);
1948
- if (!decision.shouldReview && driftedFiles.length === 0)
1949
- return null;
1950
- // Independent reviewer first (async, bounded): its findings ride in the
1951
- // SAME follow-up batch as the in-thread review + coverage requirements, so
1952
- // addressing everything still costs one extra turn.
1953
- this.reviewCoverage.start(this.hookFileEditCounts.keys());
1954
- this.idealReviewPhase = "reviewing";
1955
- const coverage = this.reviewCoverage.evidence();
1956
- const lspEvidence = this.reviewLspEvidence(coverage.expected);
1957
- this.eventBus.emit("hook", {
1958
- kind: "ideal",
1959
- coverageExpected: coverage.expected,
1960
- coverageMissing: coverage.missing,
1961
- });
1962
- // Recompute strictly AFTER the hook event: clients release held text on
1963
- // disarm, so the reverse order would paint the draft and then delete it —
1964
- // the exact flash arming exists to prevent. Arming normally PERSISTS here,
1965
- // because review starts with every changed file uncovered and a stop while
1966
- // coverage is outstanding injects again. Disarm lands later, on the read
1967
- // that closes the last gap (or when the retry budget escalates).
1968
- this.refreshIdealReviewArmed();
1969
- // Announce the phase before the reviewer starts, not after its bounded wait.
1970
- const independentMessages = await this.runIndependentReview(decision);
1971
- log("INFO", "ideal", "Injecting ideal review before final response", {
1972
- coverageExpected: coverage.expected,
1973
- coverageMissing: coverage.missing,
1974
- lspLowConfidence: lspEvidence.lowConfidence,
1975
- lspMissing: lspEvidence.missing,
1976
- });
1977
- return [
1978
- ...independentMessages,
1979
- this.withReviewLspEvidence(withReviewCoverageRequirements(buildIdealReviewMessage(decision.reasons, driftedFiles), coverage.missing), lspEvidence),
1980
- ];
1981
- }
1982
- reviewLspEvidence(files) {
1983
- const lowConfidence = [];
1984
- const missing = [];
1985
- for (const filePath of files) {
1986
- const outcome = this.lspManager?.getLatestOutcome(filePath);
1987
- if (outcome?.kind === "low_confidence")
1988
- lowConfidence.push(filePath);
1989
- else if (outcome?.kind !== "clean" && outcome?.kind !== "diagnostics")
1990
- missing.push(filePath);
1991
- }
1992
- return { lowConfidence, missing };
1993
- }
1994
- withReviewLspEvidence(message, evidence) {
1995
- if (evidence.lowConfidence.length === 0 && evidence.missing.length === 0)
1996
- return message;
1997
- const notes = [
1998
- ...(evidence.lowConfidence.length > 0
1999
- ? [`Diagnostics are low confidence while indexing: ${evidence.lowConfidence.join(", ")}.`]
2000
- : []),
2001
- ...(evidence.missing.length > 0
2002
- ? [`Diagnostics evidence is unavailable or missing: ${evidence.missing.join(", ")}.`]
2003
- : []),
2004
- "Do not describe those files as compiler-clean without other evidence.",
2005
- ];
2006
- return {
2007
- role: "user",
2008
- provenance: message.provenance,
2009
- content: `${String(message.content)}\n\n${notes.join(" ")}`,
2010
- };
2011
1840
  }
2012
1841
  /** Auto-compact if needed, run agent loop with auth retry, and persist messages. */
2013
- async runLoop(options = {}) {
2014
- // Languages are re-detected at each task boundary so a project scaffolded
2015
- // during the previous turn gets its packs; the prompt is rebuilt only when
2016
- // the set grows, keeping the cached prefix stable otherwise.
2017
- if (this.refreshActiveLanguages())
2018
- await this.rebuildSystemPromptInPlace();
2019
- this.refreshSystemPromptTail();
1842
+ async runLoopInner(options = {}) {
1843
+ await this.prepareSystemPromptForRequest();
2020
1844
  // One-shot cache-key marker per session so turn_end cacheRead numbers
2021
1845
  // in the log can be traced back to a specific routing namespace —
2022
1846
  // particularly useful when sub-agents inherit `parentKey:subagent`.
@@ -2115,11 +1939,14 @@ export class AgentSession {
2115
1939
  // so the new value transparently yields a correctly-identified client.
2116
1940
  let userAgent = this.provider === "anthropic" ? await getClaudeCliUserAgent() : undefined;
2117
1941
  const loopMessages = await this.prepareDynamicContext();
1942
+ // Re-read every run so rule-file edits apply on the next turn; a few small files.
1943
+ const streamRules = await loadStreamRules(this.cwd);
2118
1944
  const runAgentLoop = async (apiKey, accountId, projectId) => {
2119
1945
  lastResolvedAccessToken = apiKey;
2120
1946
  const modelInfo = getModel(this.model);
2121
1947
  const effectiveBaseUrl = this.baseUrl ?? creds.baseUrl;
2122
1948
  const generator = agentLoop(loopMessages, {
1949
+ ...(streamRules.length > 0 ? { streamRules: { rules: streamRules } } : {}),
2123
1950
  provider: this.provider,
2124
1951
  model: this.model,
2125
1952
  tools: options.disableTools ? [] : this.tools,
@@ -2130,8 +1957,9 @@ export class AgentSession {
2130
1957
  // Plan mode caps effort at medium (Codex `plan_mode_reasoning_effort`
2131
1958
  // preset): read-only exploration doesn't need xhigh/max reasoning, and
2132
1959
  // deep-reasoning models left at the ceiling burn enormous thinking
2133
- // budgets re-deriving context they cannot act on.
2134
- thinking: this.planModeRef.current
1960
+ // budgets re-deriving context they cannot act on. A capped prompt
1961
+ // (a sub-agent's timed answer) gets the same ceiling for the same reason.
1962
+ thinking: this.planModeRef.current || options.capThinking
2135
1963
  ? clampThinkingForPlanMode(this.thinkingLevel)
2136
1964
  : this.thinkingLevel,
2137
1965
  apiKey,
@@ -2166,6 +1994,33 @@ export class AgentSession {
2166
1994
  // + pre-warm before the first turn. "baseline": current 5-min default.
2167
1995
  cacheRetention: this.isSpeedOptimized() ? "long" : "short",
2168
1996
  promptCacheKey: this.getPromptCacheKey(),
1997
+ onContextPrepared: (context) => {
1998
+ const report = this.cacheDiagnostics.prepare(context, {
1999
+ provider: this.provider,
2000
+ model: this.model,
2001
+ at: Date.now(),
2002
+ cacheRetention: this.isSpeedOptimized() ? "long" : "short",
2003
+ route: { baseUrl: effectiveBaseUrl, accountId: this.lastAccountId ?? accountId },
2004
+ settings: {
2005
+ thinking: this.planModeRef.current || options.capThinking
2006
+ ? clampThinkingForPlanMode(this.thinkingLevel)
2007
+ : this.thinkingLevel,
2008
+ webSearch: !options.disableTools,
2009
+ supportsImages: modelInfo?.supportsImages,
2010
+ promptCacheKey: this.getPromptCacheKey(),
2011
+ },
2012
+ });
2013
+ log("INFO", "cache", "Prepared context", {
2014
+ sessionId: this.sessionId || this.transportSessionId,
2015
+ data: JSON.stringify(report),
2016
+ });
2017
+ if (report.thinkingPrefixRiskBlocks > 0) {
2018
+ log("WARN", "cache", "Possible signed-thinking prefix mismatch; not server verified", {
2019
+ sessionId: this.sessionId || this.transportSessionId,
2020
+ blocks: String(report.thinkingPrefixRiskBlocks),
2021
+ });
2022
+ }
2023
+ },
2169
2024
  supportsImages: modelInfo?.supportsImages,
2170
2025
  supportsVideo: modelInfo?.supportsVideo,
2171
2026
  userAgent,
@@ -2174,9 +2029,15 @@ export class AgentSession {
2174
2029
  maxToolResultChars: resolveSessionToolResultCharLimit(this.model, this.provider, accountId),
2175
2030
  // Aggregate per-turn budget across parallel tool results (fan-out guard).
2176
2031
  maxTurnToolResultChars: resolveSessionTurnToolResultCharLimit(this.model, this.provider, accountId),
2032
+ // Warn when web/MCP output contains instruction-like text (see injection-detect.ts).
2033
+ transformToolResult: flagUntrustedToolResult,
2177
2034
  // Self-correction hooks (same as the TUI): loop-break + re-grounding are
2178
2035
  // polled mid-loop; the ideal review is polled when the agent would stop.
2179
2036
  getSteeringMessages: () => this.getHookSteeringMessages(),
2037
+ onSteeringAvailable: (listener) => {
2038
+ this.steeringListeners.add(listener);
2039
+ return () => this.steeringListeners.delete(listener);
2040
+ },
2180
2041
  getFollowUpMessages: () => this.getHookFollowUpMessages(),
2181
2042
  onTurnBudgetExhausted: (ctx) => this.shouldExtendTurnBudget(ctx),
2182
2043
  // Check authoritative provider usage before every model/tool step.
@@ -2202,6 +2063,7 @@ export class AgentSession {
2202
2063
  // retained usage afterwards since it counted the pruned content.
2203
2064
  const pruneResult = pruneStaleToolResults(messages);
2204
2065
  if (pruneResult.pruned) {
2066
+ this.cacheDiagnostics.noteEdit("tool_prune", pruneResult.freedTokens);
2205
2067
  this.providerContext = null;
2206
2068
  log("INFO", "compaction", "Pruned stale tool outputs", {
2207
2069
  prunedResults: String(pruneResult.prunedResults),
@@ -2242,6 +2104,9 @@ export class AgentSession {
2242
2104
  usage,
2243
2105
  pendingMessages,
2244
2106
  });
2107
+ // An approved plan runs as ONE run, so step boundaries are only
2108
+ // visible here, between model steps — not after the run ends.
2109
+ const planStep = this.observePlanStepProgress(messages, contextWindow, activeTokens);
2245
2110
  log("INFO", "compaction", "In-flight compaction decision", {
2246
2111
  provider: this.provider,
2247
2112
  model: this.model,
@@ -2249,8 +2114,10 @@ export class AgentSession {
2249
2114
  contextWindow: String(contextWindow),
2250
2115
  activeTokens: String(activeTokens),
2251
2116
  triggerLimit: String(policy.targetTokens),
2117
+ ...(planStep ? { planStep: `${planStep.compact} (${planStep.reason})` } : {}),
2252
2118
  });
2253
- if (!shouldCompact(messages, contextWindow, policy.threshold, activeTokens, policy.targetTokens))
2119
+ if (!shouldCompact(messages, contextWindow, policy.threshold, activeTokens, policy.targetTokens) &&
2120
+ !planStep?.compact)
2254
2121
  return messages;
2255
2122
  }
2256
2123
  // compact() operates on this.messages, while an earlier transform may
@@ -2583,7 +2450,7 @@ export class AgentSession {
2583
2450
  const canonicalPath = await this.sessionManager.resolveCanonicalSession(this.conversationId, this.cwd);
2584
2451
  if (!canonicalPath || canonicalPath === this.sessionPath)
2585
2452
  return;
2586
- await this.adoptCompactionCheckpoint(await this.sessionManager.load(canonicalPath));
2453
+ await this.adoptCompactionCheckpoint(await this.sessionManager.load(canonicalPath, { canonical: true }));
2587
2454
  }
2588
2455
  async persistCompactionCheckpoint(sourceFingerprint, result) {
2589
2456
  const parentSessionId = this.sessionId || undefined;
@@ -2632,6 +2499,99 @@ export class AgentSession {
2632
2499
  if (this.postTurnCompaction)
2633
2500
  await this.postTurnCompaction;
2634
2501
  }
2502
+ /**
2503
+ * Advance plan-step bookkeeping over the messages added since the last
2504
+ * observation and, when a plan step was newly completed (`[DONE:n]`), run
2505
+ * the SoL-Pi cost rule. Called between model steps (in-flight) and once
2506
+ * after the run; `scanIndex` and `doneSteps` make each message and each
2507
+ * step count once across both paths. Returns undefined when no step
2508
+ * completed since the last observation.
2509
+ */
2510
+ observePlanStepProgress(messages, contextWindow, activeTokens) {
2511
+ const st = this.planStepState;
2512
+ const planPath = this.approvedPlanPath;
2513
+ if (st.planPath !== planPath) {
2514
+ st.planPath = planPath;
2515
+ st.doneSteps = new Set();
2516
+ st.requestsInCompletedSteps = 0;
2517
+ st.requestsThisStep = 0;
2518
+ }
2519
+ if (st.scanIndex > messages.length)
2520
+ st.scanIndex = 0;
2521
+ let requests = 0;
2522
+ const newlyDone = [];
2523
+ for (let i = st.scanIndex; i < messages.length; i++) {
2524
+ const msg = messages[i];
2525
+ if (msg?.role !== "assistant")
2526
+ continue;
2527
+ requests++;
2528
+ const text = typeof msg.content === "string"
2529
+ ? msg.content
2530
+ : msg.content.map((part) => (part.type === "text" ? part.text : "")).join("\n");
2531
+ for (const step of findCompletedMarkers(text)) {
2532
+ if (!st.doneSteps.has(step))
2533
+ newlyDone.push(step);
2534
+ }
2535
+ }
2536
+ st.scanIndex = messages.length;
2537
+ const contextTokens = activeTokens ?? estimateConversationTokens(messages);
2538
+ if (st.lastContextTokens > 0)
2539
+ st.grownTokens += Math.max(0, contextTokens - st.lastContextTokens);
2540
+ st.lastContextTokens = contextTokens;
2541
+ st.requests += requests;
2542
+ st.requestsThisStep += requests;
2543
+ st.requestsSinceLastCompaction += requests;
2544
+ if (st.writeCostBalance > 0)
2545
+ st.writeCostBalance -= st.savingPerRequest * requests;
2546
+ if (!planPath || newlyDone.length === 0)
2547
+ return undefined;
2548
+ for (const step of newlyDone)
2549
+ st.doneSteps.add(step);
2550
+ st.requestsInCompletedSteps += st.requestsThisStep;
2551
+ st.requestsThisStep = 0;
2552
+ let planText;
2553
+ try {
2554
+ planText = readFileSync(planPath, "utf8");
2555
+ }
2556
+ catch {
2557
+ return undefined;
2558
+ }
2559
+ const totalSteps = extractPlanSteps(planText).length;
2560
+ const decision = decidePlanStepCompaction({
2561
+ contextTokens,
2562
+ keptTailTokens: PLAN_STEP_KEEP_TOKENS,
2563
+ contextWindow,
2564
+ // The model registry carries no cache price fields; use the SoL-Pi default.
2565
+ cacheWriteReadRatio: DEFAULT_CACHE_WRITE_READ_RATIO,
2566
+ stepsCompleted: st.doneSteps.size,
2567
+ stepsRemaining: Math.max(0, totalSteps - st.doneSteps.size),
2568
+ requestsInCompletedSteps: st.requestsInCompletedSteps,
2569
+ requests: st.requests,
2570
+ grownTokens: st.grownTokens,
2571
+ priorCompactions: st.compactions,
2572
+ writeCostBalance: st.writeCostBalance,
2573
+ requestsSinceLastCompaction: st.compactions > 0 ? st.requestsSinceLastCompaction : undefined,
2574
+ });
2575
+ log("INFO", "compaction", "Plan-step compaction decision", {
2576
+ compact: String(decision.compact),
2577
+ reason: decision.reason,
2578
+ });
2579
+ return decision;
2580
+ }
2581
+ /**
2582
+ * Plan-step bookkeeping after ANY successful compaction (as in the bench:
2583
+ * every compaction leaves a cache-write cost to be repaid by later savings).
2584
+ */
2585
+ recordPlanStepCompaction(contextBefore) {
2586
+ const st = this.planStepState;
2587
+ const after = estimateConversationTokens(this.messages);
2588
+ st.scanIndex = this.messages.length;
2589
+ st.lastContextTokens = after;
2590
+ st.requestsSinceLastCompaction = 0;
2591
+ st.compactions++;
2592
+ st.writeCostBalance += after * (DEFAULT_CACHE_WRITE_READ_RATIO - 1);
2593
+ st.savingPerRequest = Math.max(0, contextBefore - after);
2594
+ }
2635
2595
  /**
2636
2596
  * Post-turn compaction: once the final response has been delivered, compact
2637
2597
  * in the background while the user reads the answer, instead of making the
@@ -2640,7 +2600,9 @@ export class AgentSession {
2640
2600
  * Codex `model_post_turn_compact_threshold_percent` guards: skip when user
2641
2601
  * input is already queued (it would race the next turn), when the run was
2642
2602
  * aborted, or during the failure cooldown — and never let a compaction
2643
- * error surface in the completed turn.
2603
+ * error surface in the completed turn. Besides the size trigger, a newly
2604
+ * completed approved-plan step may compact when the cache-cost rule in
2605
+ * compaction/plan-step-policy.ts says the shrink pays for itself.
2644
2606
  */
2645
2607
  maybeCompactPostTurn(creds) {
2646
2608
  if (!this.settingsManager.get("autoCompact"))
@@ -2653,15 +2615,18 @@ export class AgentSession {
2653
2615
  return;
2654
2616
  if (Date.now() < this.compactionRetryAfter)
2655
2617
  return;
2656
- // One compaction per turn boundary: a pre-run or overflow-recovery
2657
- // compaction already shrank this run's history — re-probing right after
2658
- // the final response would only re-derive that decision.
2659
- if (this.compactionOccurred)
2660
- return;
2661
2618
  const contextWindow = getContextWindow(this.model, {
2662
2619
  provider: this.provider,
2663
2620
  accountId: creds.accountId,
2664
2621
  });
2622
+ // One compaction per turn boundary: a pre-run, in-flight or overflow
2623
+ // compaction already shrank this run's history — re-probing right after
2624
+ // the final response would only re-derive that decision. Still record the
2625
+ // final response's `[DONE:n]` steps so plan bookkeeping stays current.
2626
+ if (this.compactionOccurred) {
2627
+ this.observePlanStepProgress(this.messages, contextWindow, undefined);
2628
+ return;
2629
+ }
2665
2630
  const policy = resolveCompactionPolicy({
2666
2631
  provider: this.provider,
2667
2632
  model: this.model,
@@ -2680,9 +2645,12 @@ export class AgentSession {
2680
2645
  });
2681
2646
  }
2682
2647
  }
2683
- if (!shouldCompact(this.messages, contextWindow, policy.threshold, activeTokens, policy.targetTokens))
2648
+ const planStep = this.observePlanStepProgress(this.messages, contextWindow, activeTokens);
2649
+ if (!shouldCompact(this.messages, contextWindow, policy.threshold, activeTokens, policy.targetTokens) &&
2650
+ !planStep?.compact)
2684
2651
  return;
2685
2652
  log("INFO", "compaction", "Post-turn compaction decision — compacting in background", {
2653
+ trigger: planStep?.compact ? `plan-step (${planStep.reason})` : "size",
2686
2654
  provider: this.provider,
2687
2655
  model: this.model,
2688
2656
  transport: this.provider === "openai" && creds.accountId ? "codex_oauth" : "public_api",
@@ -2736,6 +2704,7 @@ export class AgentSession {
2736
2704
  approvedPlanPath: this.approvedPlanPath,
2737
2705
  });
2738
2706
  const originalCount = this.messages.length;
2707
+ const contextTokensBefore = estimateConversationTokens(this.messages);
2739
2708
  this.eventBus.emit("compaction_start", { messageCount: originalCount });
2740
2709
  let contextSelection;
2741
2710
  const runCompactor = async () => {
@@ -2773,7 +2742,7 @@ export class AgentSession {
2773
2742
  let sourceFingerprint = computeSourceFingerprint(this.messages);
2774
2743
  const canonicalPath = await this.sessionManager.resolveCanonicalSession(conversationId, this.cwd);
2775
2744
  if (canonicalPath && canonicalPath !== this.sessionPath) {
2776
- const newest = await this.sessionManager.load(canonicalPath);
2745
+ const newest = await this.sessionManager.load(canonicalPath, { canonical: true });
2777
2746
  if (newest.header.sourceFingerprint === sourceFingerprint) {
2778
2747
  await this.adoptCompactionCheckpoint(newest);
2779
2748
  this.lastCompactionCompacted = true;
@@ -2847,6 +2816,10 @@ export class AgentSession {
2847
2816
  }
2848
2817
  });
2849
2818
  }
2819
+ if (this.lastCompactionCompacted) {
2820
+ this.cacheDiagnostics.noteEdit("compaction");
2821
+ this.recordPlanStepCompaction(contextTokensBefore);
2822
+ }
2850
2823
  this.eventBus.emit("compaction_end", {
2851
2824
  compacted: this.lastCompactionCompacted,
2852
2825
  originalCount,
@@ -2866,6 +2839,7 @@ export class AgentSession {
2866
2839
  });
2867
2840
  }
2868
2841
  async newSession(preserveConversation = false) {
2842
+ this.cacheDiagnostics.reset();
2869
2843
  // Approved-plan execution is a clean checkpoint of the same conversation;
2870
2844
  // explicit new sessions reset the conversation identity.
2871
2845
  if (!preserveConversation) {
@@ -2889,6 +2863,7 @@ export class AgentSession {
2889
2863
  const basePrompt = await this.buildBasePrompt(false, undefined);
2890
2864
  this.baseSystemPrompt = basePrompt;
2891
2865
  this.messages = [{ role: "system", content: this.withSystemPromptTail(basePrompt) }];
2866
+ this.clearReadTracker?.();
2892
2867
  // Fresh conversation — new entries must not chain onto the old DAG's leaf.
2893
2868
  this.currentLeafId = null;
2894
2869
  // Transient sessions (Nolan chat/autopilot, subagent spawns) never touch the
@@ -2912,6 +2887,7 @@ export class AgentSession {
2912
2887
  }
2913
2888
  async loadSession(sessionPath) {
2914
2889
  await this.loadExistingSession(sessionPath);
2890
+ this.cacheDiagnostics.reset();
2915
2891
  if (this.sessionId)
2916
2892
  await this.subAgentManager?.hydrate(this.sessionId);
2917
2893
  this.eventBus.emit("session_start", { sessionId: this.sessionId });
@@ -2942,7 +2918,10 @@ export class AgentSession {
2942
2918
  const branchMessages = this.sessionManager.getMessages(loaded.entries, this.currentLeafId);
2943
2919
  const systemMsg = this.messages[0];
2944
2920
  this.messages = [systemMsg, ...branchMessages];
2921
+ this.cacheDiagnostics.reset();
2945
2922
  this.lastPersistedIndex = this.messages.length;
2923
+ // Reads made in the dropped messages are no longer in the model's context.
2924
+ this.clearReadTracker?.();
2946
2925
  this.eventBus.emit("branch_created", {
2947
2926
  leafId: this.currentLeafId,
2948
2927
  messagesKept: branchMessages.length,
@@ -3007,23 +2986,33 @@ export class AgentSession {
3007
2986
  : undefined;
3008
2987
  return costUsd === undefined ? { used, size } : { used, size, costUsd };
3009
2988
  }
3010
- getPlanMode() {
3011
- return this.planModeRef.current;
3012
- }
3013
2989
  /**
3014
- * Suppress only the pre-final Ideal self-review for this live session.
3015
- * Autopilot uses this while Nolan independently owns verification; loop-break
3016
- * and post-compaction re-grounding remain active.
2990
+ * Whether the provider's prompt cache has likely lapsed since the last
2991
+ * successful request, and how many tokens the next message would re-read at
2992
+ * full price. Null when the route has no known TTL or the chat is empty.
3017
2993
  */
3018
- setIdealReviewSuppressed(suppressed) {
3019
- this.idealReviewSuppressed = suppressed;
3020
- if (suppressed) {
3021
- this.idealReviewPhase = "idle";
3022
- this.reviewCoverage.reset();
3023
- }
3024
- // Suppression flips mid-run (autopilot takes over verification), so a client
3025
- // holding a draft under a stale arming must be released.
3026
- this.refreshHookArming();
2994
+ getCacheExpiryStatus(now = Date.now()) {
2995
+ const status = assessCacheExpiry({
2996
+ current: {
2997
+ provider: this.provider,
2998
+ model: this.model,
2999
+ policy: resolveCacheTtl({
3000
+ provider: this.provider,
3001
+ model: this.model,
3002
+ cacheRetention: this.isSpeedOptimized() ? "long" : "short",
3003
+ baseUrl: this.baseUrl,
3004
+ accountId: this.lastAccountId,
3005
+ }),
3006
+ },
3007
+ lastTouch: this.cacheDiagnostics.lastCacheTouch(),
3008
+ now,
3009
+ prefixTokens: this.getContextUsage().used,
3010
+ hasHistory: this.messages.some((m) => m.role === "user"),
3011
+ });
3012
+ return status && { ...status, sessionId: this.sessionId || this.transportSessionId };
3013
+ }
3014
+ getPlanMode() {
3015
+ return this.planModeRef.current;
3027
3016
  }
3028
3017
  /** Queue a user message (optionally with attachments) to be injected mid-run
3029
3018
  * as steering. Returns the new queue length. No-op semantics are the caller's
@@ -3031,6 +3020,9 @@ export class AgentSession {
3031
3020
  queueMessage(text, attachments = []) {
3032
3021
  this.queueSeq += 1;
3033
3022
  this.userQueue.push({ id: `q${this.queueSeq}`, text, attachments });
3023
+ // Instant interrupt: preempt running tools so the steer lands right away.
3024
+ for (const listener of [...this.steeringListeners])
3025
+ listener();
3034
3026
  return this.userQueue.length;
3035
3027
  }
3036
3028
  /** Pending queued messages (id + text), oldest first, for client display. */
@@ -3083,6 +3075,15 @@ export class AgentSession {
3083
3075
  return `No background process with id "${id}"`;
3084
3076
  return this.processManager.stop(id);
3085
3077
  }
3078
+ /**
3079
+ * Force-stop every background process tree, synchronously. Background
3080
+ * commands run in their own process group, so the daemon's group kill on
3081
+ * quit never reaches them: this is the only thing that does. Callers on a
3082
+ * shutdown deadline run it before awaiting anything that can hang.
3083
+ */
3084
+ stopBackgroundProcesses() {
3085
+ this.processManager?.shutdownAll();
3086
+ }
3086
3087
  /** Replace a host-owned system prompt in place without resetting conversation history. */
3087
3088
  setCustomSystemPrompt(systemPrompt, promptCacheKeyPrefix) {
3088
3089
  this.customSystemPrompt = systemPrompt;
@@ -3229,6 +3230,18 @@ export class AgentSession {
3229
3230
  * the standard prompt therefore needs a rebuild. Custom and sub-agent
3230
3231
  * prompts never render packs, so detection is skipped for them.
3231
3232
  */
3233
+ /**
3234
+ * Bring the system prompt to the exact state the next provider request will
3235
+ * send. Shared by real runs and {@link prewarm} so both see one prefix.
3236
+ */
3237
+ async prepareSystemPromptForRequest() {
3238
+ // Languages are re-detected at each task boundary so a project scaffolded
3239
+ // during the previous turn gets its packs; the prompt is rebuilt only when
3240
+ // the set grows, keeping the cached prefix stable otherwise.
3241
+ if (this.refreshActiveLanguages())
3242
+ await this.rebuildSystemPromptInPlace();
3243
+ this.refreshSystemPromptTail();
3244
+ }
3232
3245
  refreshActiveLanguages() {
3233
3246
  if (this.customSystemPrompt || this.agentPrompt !== undefined)
3234
3247
  return false;
@@ -3327,6 +3340,17 @@ export class AgentSession {
3327
3340
  }));
3328
3341
  }
3329
3342
  async persistTurnMetric(event) {
3343
+ if (event.stopReason === "error")
3344
+ this.cacheDiagnostics.discardAttempt();
3345
+ const cache = this.cacheDiagnostics.complete(event.usage, event.timing);
3346
+ if (cache) {
3347
+ log("INFO", "cache", "Context cache outcome", {
3348
+ sessionId: this.sessionId || this.transportSessionId,
3349
+ provider: this.provider,
3350
+ model: this.model,
3351
+ data: JSON.stringify(cache),
3352
+ });
3353
+ }
3330
3354
  const payload = {
3331
3355
  version: 1,
3332
3356
  turn: event.turn,
@@ -3712,6 +3736,138 @@ export class AgentSession {
3712
3736
  if (signal?.aborted)
3713
3737
  this.managerAbortHandler();
3714
3738
  }
3739
+ runLoopDepth = 0;
3740
+ lastRealRequestAt = 0;
3741
+ lastPrewarmAt = 0;
3742
+ prewarmController = null;
3743
+ /**
3744
+ * Best-effort Anthropic prompt-cache prewarm before the user's next turn
3745
+ * (desktop app calls this on the first keystroke after opening a chat or an
3746
+ * idle pause). Sends the exact request prefix the next real turn will use —
3747
+ * same system/tools/thinking/cache options — with `max_tokens: 1`, so the
3748
+ * first real reply is a cache read instead of a cold write.
3749
+ */
3750
+ async prewarm(signal) {
3751
+ if (this.provider !== "anthropic")
3752
+ return { ok: false, reason: "provider" };
3753
+ if (this.settingsManager?.get("cachePrewarm") === false) {
3754
+ return { ok: false, reason: "disabled" };
3755
+ }
3756
+ if (this.runLoopDepth > 0)
3757
+ return { ok: false, reason: "run_active" };
3758
+ if (this.prewarmController)
3759
+ return { ok: false, reason: "in_flight" };
3760
+ const cacheRetention = this.isSpeedOptimized() ? "long" : "short";
3761
+ const ttlMs = (cacheRetention === "long" ? 60 : 5) * 60_000;
3762
+ const now = Date.now();
3763
+ if (now - Math.max(this.lastPrewarmAt, this.lastRealRequestAt) < ttlMs) {
3764
+ return { ok: false, reason: "cache_fresh" };
3765
+ }
3766
+ // Same preparation as the real run, BEFORE copying history: otherwise the
3767
+ // warmed system block differs from the next request (language packs are
3768
+ // detected at run start) and everything after it misses the cache.
3769
+ await this.prepareSystemPromptForRequest();
3770
+ // The await above yields: re-check that no run or other prewarm began.
3771
+ if (this.runLoopDepth > 0)
3772
+ return { ok: false, reason: "run_active" };
3773
+ if (this.prewarmController)
3774
+ return { ok: false, reason: "in_flight" };
3775
+ // End at the last user/tool message: the next real turn appends a new user
3776
+ // message after the trailing assistant reply, and its cache lookback hits
3777
+ // the entry written at this boundary. A request must end on a user turn.
3778
+ let end = this.messages.length;
3779
+ while (end > 0 && this.messages[end - 1]?.role === "assistant")
3780
+ end--;
3781
+ // The loop repairs tool pairing in place before every request; apply the
3782
+ // same repair to a copy so the warmed prefix is byte-identical to it.
3783
+ const messages = structuredClone(this.messages.slice(0, end));
3784
+ repairToolPairingAdjacent(messages);
3785
+ if (!messages.some((m) => m.role === "user" || m.role === "tool")) {
3786
+ return { ok: false, reason: "no_history" };
3787
+ }
3788
+ const tokens = estimateConversationTokens(messages);
3789
+ if (tokens < 4_000)
3790
+ return { ok: false, reason: "too_small" };
3791
+ const controller = new AbortController();
3792
+ const onAbort = () => controller.abort();
3793
+ signal?.addEventListener("abort", onAbort, { once: true });
3794
+ this.prewarmController = controller;
3795
+ const started = Date.now();
3796
+ try {
3797
+ const creds = await this.authStorage.resolveCredentials(this.provider, {
3798
+ storageKeys: this.currentAuthStorageKeys(),
3799
+ });
3800
+ if (controller.signal.aborted || this.runLoopDepth > 0) {
3801
+ return { ok: false, reason: "aborted" };
3802
+ }
3803
+ const modelInfo = getModel(this.model);
3804
+ const result = stream({
3805
+ provider: this.provider,
3806
+ model: this.model,
3807
+ messages,
3808
+ tools: this.tools,
3809
+ webSearch: true,
3810
+ maxTokens: this.maxTokens,
3811
+ thinking: this.planModeRef.current
3812
+ ? clampThinkingForPlanMode(this.thinkingLevel)
3813
+ : this.thinkingLevel,
3814
+ apiKey: creds.accessToken,
3815
+ baseUrl: this.baseUrl ?? creds.baseUrl,
3816
+ accountId: creds.accountId,
3817
+ transportSessionId: this.sessionId || this.transportSessionId,
3818
+ cacheRetention,
3819
+ promptCacheKey: this.getPromptCacheKey(),
3820
+ supportsImages: modelInfo?.supportsImages,
3821
+ supportsVideo: modelInfo?.supportsVideo,
3822
+ userAgent: await getClaudeCliUserAgent(),
3823
+ prewarm: true,
3824
+ signal: controller.signal,
3825
+ });
3826
+ const response = await result.response;
3827
+ const usage = response.usage;
3828
+ if (usage.inputTokens === 0 && usage.outputTokens === 0) {
3829
+ // @prestyj/ai sent nothing: budget thinking can't stay identical at max_tokens 1.
3830
+ log("INFO", "prewarm", "Cache prewarm skipped: budget thinking", {
3831
+ model: this.model,
3832
+ });
3833
+ return { ok: false, reason: "thinking_budget_incompatible" };
3834
+ }
3835
+ this.lastPrewarmAt = Date.now();
3836
+ const warmedPolicy = resolveCacheTtl({
3837
+ provider: this.provider,
3838
+ model: this.model,
3839
+ cacheRetention,
3840
+ baseUrl: this.baseUrl ?? creds.baseUrl,
3841
+ accountId: creds.accountId,
3842
+ });
3843
+ if (warmedPolicy) {
3844
+ this.cacheDiagnostics.noteCacheTouch({
3845
+ at: started,
3846
+ provider: this.provider,
3847
+ model: this.model,
3848
+ policy: warmedPolicy,
3849
+ });
3850
+ }
3851
+ log("INFO", "prewarm", "Cache prewarm complete", {
3852
+ tokens: String(tokens),
3853
+ cacheRead: String(usage.cacheRead ?? 0),
3854
+ cacheWrite: String(usage.cacheWrite ?? 0),
3855
+ ms: String(Date.now() - started),
3856
+ });
3857
+ return { ok: true, reason: "warmed", usage };
3858
+ }
3859
+ catch (error) {
3860
+ if (isAbortError(error) || controller.signal.aborted)
3861
+ return { ok: false, reason: "aborted" };
3862
+ log("WARN", "prewarm", `Cache prewarm failed: ${error instanceof Error ? error.message : String(error)}`);
3863
+ return { ok: false, reason: "error" };
3864
+ }
3865
+ finally {
3866
+ signal?.removeEventListener("abort", onAbort);
3867
+ if (this.prewarmController === controller)
3868
+ this.prewarmController = null;
3869
+ }
3870
+ }
3715
3871
  /** True when speedProfile is "optimized" (1-h cache TTL + pre-warm), or the
3716
3872
  * session was constructed with `forceLongCacheRetention` (Nolan sessions). */
3717
3873
  isSpeedOptimized() {
@@ -3740,16 +3896,22 @@ export class AgentSession {
3740
3896
  return this.getPromptCacheKey();
3741
3897
  }
3742
3898
  async dispose() {
3899
+ // First and synchronous: nothing below may delay this, or a hung teardown
3900
+ // step leaves background commands running after the app has quit.
3901
+ this.stopBackgroundProcesses();
3743
3902
  // Quiesce any in-flight post-turn compaction BEFORE tearing down state:
3744
3903
  // the background compact() snapshots and replaces `this.messages`, so
3745
3904
  // letting it run past this point would checkpoint a near-empty history
3746
3905
  // and leak a junk session file after teardown.
3747
3906
  if (this.postTurnCompaction)
3748
3907
  await this.postTurnCompaction;
3908
+ this.cacheDiagnostics.reset();
3749
3909
  this.diagnosticsRecorder?.finalize();
3750
3910
  this.managerAbortSignal?.removeEventListener("abort", this.managerAbortHandler);
3751
- this.processManager?.shutdownAll();
3911
+ // Again, in case a turn racing teardown started one while we awaited.
3912
+ this.stopBackgroundProcesses();
3752
3913
  this.lspManager?.shutdownAll();
3914
+ this.debugManager?.shutdown();
3753
3915
  await Promise.all([this.subAgentManager?.shutdownAll(), this.mcpManager?.dispose()]);
3754
3916
  await this.extensionLoader.deactivateAll();
3755
3917
  this.setSessionPath("");
@@ -3785,8 +3947,10 @@ export class AgentSession {
3785
3947
  // A stale physical checkpoint is only an address, not the conversation tip.
3786
3948
  // Resolve every resume—not just over-threshold/deferred compaction resumes—
3787
3949
  // before reading history so the next prompt cannot continue an old branch.
3788
- const canonicalPath = (await this.sessionManager.resolveCanonicalSession(sessionPath, this.cwd)) ?? sessionPath;
3789
- const loaded = await this.sessionManager.load(canonicalPath);
3950
+ const resolvedPath = await this.sessionManager.resolveCanonicalSession(sessionPath, this.cwd);
3951
+ const loaded = resolvedPath
3952
+ ? await this.sessionManager.load(resolvedPath, { canonical: true })
3953
+ : await this.sessionManager.load(sessionPath);
3790
3954
  // Use the leaf from the header to walk the correct branch
3791
3955
  const loadedMessages = this.sessionManager.getMessages(loaded.entries, loaded.header.leafId);
3792
3956
  const savedCompletionReview = [...loaded.entries]
@@ -3840,6 +4004,8 @@ export class AgentSession {
3840
4004
  // not fail when Anthropic's stricter many-image limit activates later.
3841
4005
  const systemMsg = this.messages[0]; // Already built
3842
4006
  this.messages = [systemMsg, ...loadedMessages];
4007
+ // Reads recorded for the previous conversation don't carry over.
4008
+ this.clearReadTracker?.();
3843
4009
  const normalizedImageCount = await normalizeMessageImages(this.messages);
3844
4010
  if (normalizedImageCount > 0) {
3845
4011
  log("INFO", "session", `Resized ${normalizedImageCount} restored session image(s)`);